mirror of
https://repo.dactyloidae.xyz/Dactyloidae/UXP.git
synced 2026-09-26 02:17:34 +09:00
import FIREFOX_52_6_0esr_RELEASE from mozilla-esr52 hg repo
This commit is contained in:
commit
dcd9973243
150858 changed files with 23884658 additions and 0 deletions
93
media/ffvpx/libavcodec/x86/constants.c
Normal file
93
media/ffvpx/libavcodec/x86/constants.c
Normal file
|
|
@ -0,0 +1,93 @@
|
|||
/*
|
||||
* MMX/SSE/AVX constants used across x86 dsp optimizations.
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#include "libavutil/mem.h"
|
||||
#include "libavutil/x86/asm.h" // for xmm_reg
|
||||
#include "constants.h"
|
||||
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_1) = { 0x0001000100010001ULL, 0x0001000100010001ULL,
|
||||
0x0001000100010001ULL, 0x0001000100010001ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_2) = { 0x0002000200020002ULL, 0x0002000200020002ULL,
|
||||
0x0002000200020002ULL, 0x0002000200020002ULL };
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_3) = { 0x0003000300030003ULL, 0x0003000300030003ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_4) = { 0x0004000400040004ULL, 0x0004000400040004ULL,
|
||||
0x0004000400040004ULL, 0x0004000400040004ULL };
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_5) = { 0x0005000500050005ULL, 0x0005000500050005ULL };
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_8) = { 0x0008000800080008ULL, 0x0008000800080008ULL };
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_9) = { 0x0009000900090009ULL, 0x0009000900090009ULL };
|
||||
DECLARE_ALIGNED(8, const uint64_t, ff_pw_15) = 0x000F000F000F000FULL;
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_16) = { 0x0010001000100010ULL, 0x0010001000100010ULL };
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_17) = { 0x0011001100110011ULL, 0x0011001100110011ULL };
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_18) = { 0x0012001200120012ULL, 0x0012001200120012ULL };
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_20) = { 0x0014001400140014ULL, 0x0014001400140014ULL };
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_32) = { 0x0020002000200020ULL, 0x0020002000200020ULL };
|
||||
DECLARE_ALIGNED(8, const uint64_t, ff_pw_42) = 0x002A002A002A002AULL;
|
||||
DECLARE_ALIGNED(8, const uint64_t, ff_pw_53) = 0x0035003500350035ULL;
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_64) = { 0x0040004000400040ULL, 0x0040004000400040ULL };
|
||||
DECLARE_ALIGNED(8, const uint64_t, ff_pw_96) = 0x0060006000600060ULL;
|
||||
DECLARE_ALIGNED(8, const uint64_t, ff_pw_128) = 0x0080008000800080ULL;
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_255) = { 0x00ff00ff00ff00ffULL, 0x00ff00ff00ff00ffULL,
|
||||
0x00ff00ff00ff00ffULL, 0x00ff00ff00ff00ffULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_256) = { 0x0100010001000100ULL, 0x0100010001000100ULL,
|
||||
0x0100010001000100ULL, 0x0100010001000100ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_512) = { 0x0200020002000200ULL, 0x0200020002000200ULL,
|
||||
0x0200020002000200ULL, 0x0200020002000200ULL };
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pw_1019) = { 0x03FB03FB03FB03FBULL, 0x03FB03FB03FB03FBULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_1023) = { 0x03ff03ff03ff03ffULL, 0x03ff03ff03ff03ffULL,
|
||||
0x03ff03ff03ff03ffULL, 0x03ff03ff03ff03ffULL};
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_1024) = { 0x0400040004000400ULL, 0x0400040004000400ULL,
|
||||
0x0400040004000400ULL, 0x0400040004000400ULL};
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_2048) = { 0x0800080008000800ULL, 0x0800080008000800ULL,
|
||||
0x0800080008000800ULL, 0x0800080008000800ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_4095) = { 0x0fff0fff0fff0fffULL, 0x0fff0fff0fff0fffULL,
|
||||
0x0fff0fff0fff0fffULL, 0x0fff0fff0fff0fffULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_4096) = { 0x1000100010001000ULL, 0x1000100010001000ULL,
|
||||
0x1000100010001000ULL, 0x1000100010001000ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_8192) = { 0x2000200020002000ULL, 0x2000200020002000ULL,
|
||||
0x2000200020002000ULL, 0x2000200020002000ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pw_m1) = { 0xFFFFFFFFFFFFFFFFULL, 0xFFFFFFFFFFFFFFFFULL,
|
||||
0xFFFFFFFFFFFFFFFFULL, 0xFFFFFFFFFFFFFFFFULL };
|
||||
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pb_0) = { 0x0000000000000000ULL, 0x0000000000000000ULL,
|
||||
0x0000000000000000ULL, 0x0000000000000000ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pb_1) = { 0x0101010101010101ULL, 0x0101010101010101ULL,
|
||||
0x0101010101010101ULL, 0x0101010101010101ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pb_2) = { 0x0202020202020202ULL, 0x0202020202020202ULL,
|
||||
0x0202020202020202ULL, 0x0202020202020202ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pb_3) = { 0x0303030303030303ULL, 0x0303030303030303ULL,
|
||||
0x0303030303030303ULL, 0x0303030303030303ULL };
|
||||
DECLARE_ALIGNED(32, const xmm_reg, ff_pb_15) = { 0x0F0F0F0F0F0F0F0FULL, 0x0F0F0F0F0F0F0F0FULL };
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_pb_80) = { 0x8080808080808080ULL, 0x8080808080808080ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pb_FE) = { 0xFEFEFEFEFEFEFEFEULL, 0xFEFEFEFEFEFEFEFEULL,
|
||||
0xFEFEFEFEFEFEFEFEULL, 0xFEFEFEFEFEFEFEFEULL };
|
||||
DECLARE_ALIGNED(8, const uint64_t, ff_pb_FC) = 0xFCFCFCFCFCFCFCFCULL;
|
||||
|
||||
DECLARE_ALIGNED(16, const xmm_reg, ff_ps_neg) = { 0x8000000080000000ULL, 0x8000000080000000ULL };
|
||||
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pd_1) = { 0x0000000100000001ULL, 0x0000000100000001ULL,
|
||||
0x0000000100000001ULL, 0x0000000100000001ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pd_16) = { 0x0000001000000010ULL, 0x0000001000000010ULL,
|
||||
0x0000001000000010ULL, 0x0000001000000010ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pd_32) = { 0x0000002000000020ULL, 0x0000002000000020ULL,
|
||||
0x0000002000000020ULL, 0x0000002000000020ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pd_8192) = { 0x0000200000002000ULL, 0x0000200000002000ULL,
|
||||
0x0000200000002000ULL, 0x0000200000002000ULL };
|
||||
DECLARE_ALIGNED(32, const ymm_reg, ff_pd_65535)= { 0x0000ffff0000ffffULL, 0x0000ffff0000ffffULL,
|
||||
0x0000ffff0000ffffULL, 0x0000ffff0000ffffULL };
|
||||
71
media/ffvpx/libavcodec/x86/constants.h
Normal file
71
media/ffvpx/libavcodec/x86/constants.h
Normal file
|
|
@ -0,0 +1,71 @@
|
|||
/*
|
||||
* MMX/SSE constants used across x86 dsp optimizations.
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#ifndef AVCODEC_X86_CONSTANTS_H
|
||||
#define AVCODEC_X86_CONSTANTS_H
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#include "libavutil/x86/asm.h"
|
||||
|
||||
extern const ymm_reg ff_pw_1;
|
||||
extern const ymm_reg ff_pw_2;
|
||||
extern const xmm_reg ff_pw_3;
|
||||
extern const ymm_reg ff_pw_4;
|
||||
extern const xmm_reg ff_pw_5;
|
||||
extern const xmm_reg ff_pw_8;
|
||||
extern const xmm_reg ff_pw_9;
|
||||
extern const uint64_t ff_pw_15;
|
||||
extern const xmm_reg ff_pw_16;
|
||||
extern const xmm_reg ff_pw_18;
|
||||
extern const xmm_reg ff_pw_20;
|
||||
extern const xmm_reg ff_pw_32;
|
||||
extern const uint64_t ff_pw_42;
|
||||
extern const uint64_t ff_pw_53;
|
||||
extern const xmm_reg ff_pw_64;
|
||||
extern const uint64_t ff_pw_96;
|
||||
extern const uint64_t ff_pw_128;
|
||||
extern const ymm_reg ff_pw_255;
|
||||
extern const ymm_reg ff_pw_512;
|
||||
extern const ymm_reg ff_pw_1023;
|
||||
extern const ymm_reg ff_pw_1024;
|
||||
extern const ymm_reg ff_pw_2048;
|
||||
extern const ymm_reg ff_pw_4095;
|
||||
extern const ymm_reg ff_pw_4096;
|
||||
extern const ymm_reg ff_pw_8192;
|
||||
extern const ymm_reg ff_pw_m1;
|
||||
|
||||
extern const ymm_reg ff_pb_0;
|
||||
extern const ymm_reg ff_pb_1;
|
||||
extern const ymm_reg ff_pb_2;
|
||||
extern const ymm_reg ff_pb_3;
|
||||
extern const xmm_reg ff_pb_80;
|
||||
extern const ymm_reg ff_pb_FE;
|
||||
extern const uint64_t ff_pb_FC;
|
||||
|
||||
extern const xmm_reg ff_ps_neg;
|
||||
|
||||
extern const ymm_reg ff_pd_1;
|
||||
extern const ymm_reg ff_pd_16;
|
||||
extern const ymm_reg ff_pd_32;
|
||||
extern const ymm_reg ff_pd_8192;
|
||||
extern const ymm_reg ff_pd_65535;
|
||||
|
||||
#endif /* AVCODEC_X86_CONSTANTS_H */
|
||||
313
media/ffvpx/libavcodec/x86/flacdsp.asm
Normal file
313
media/ffvpx/libavcodec/x86/flacdsp.asm
Normal file
|
|
@ -0,0 +1,313 @@
|
|||
;******************************************************************************
|
||||
;* FLAC DSP SIMD optimizations
|
||||
;*
|
||||
;* Copyright (C) 2014 Loren Merritt
|
||||
;* Copyright (C) 2014 James Almer
|
||||
;*
|
||||
;* This file is part of FFmpeg.
|
||||
;*
|
||||
;* FFmpeg is free software; you can redistribute it and/or
|
||||
;* modify it under the terms of the GNU Lesser General Public
|
||||
;* License as published by the Free Software Foundation; either
|
||||
;* version 2.1 of the License, or (at your option) any later version.
|
||||
;*
|
||||
;* FFmpeg is distributed in the hope that it will be useful,
|
||||
;* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
;* Lesser General Public License for more details.
|
||||
;*
|
||||
;* You should have received a copy of the GNU Lesser General Public
|
||||
;* License along with FFmpeg; if not, write to the Free Software
|
||||
;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
;******************************************************************************
|
||||
|
||||
%include "libavutil/x86/x86util.asm"
|
||||
|
||||
SECTION .text
|
||||
|
||||
%macro PMACSDQL 5
|
||||
%if cpuflag(xop)
|
||||
pmacsdql %1, %2, %3, %1
|
||||
%else
|
||||
pmuldq %2, %3
|
||||
paddq %1, %2
|
||||
%endif
|
||||
%endmacro
|
||||
|
||||
%macro LPC_32 1
|
||||
INIT_XMM %1
|
||||
cglobal flac_lpc_32, 5,6,5, decoded, coeffs, pred_order, qlevel, len, j
|
||||
sub lend, pred_orderd
|
||||
jle .ret
|
||||
lea decodedq, [decodedq+pred_orderq*4-8]
|
||||
lea coeffsq, [coeffsq+pred_orderq*4]
|
||||
neg pred_orderq
|
||||
movd m4, qlevelm
|
||||
ALIGN 16
|
||||
.loop_sample:
|
||||
movd m0, [decodedq+pred_orderq*4+8]
|
||||
add decodedq, 8
|
||||
movd m1, [coeffsq+pred_orderq*4]
|
||||
pxor m2, m2
|
||||
pxor m3, m3
|
||||
lea jq, [pred_orderq+1]
|
||||
test jq, jq
|
||||
jz .end_order
|
||||
.loop_order:
|
||||
PMACSDQL m2, m0, m1, m2, m0
|
||||
movd m0, [decodedq+jq*4]
|
||||
PMACSDQL m3, m1, m0, m3, m1
|
||||
movd m1, [coeffsq+jq*4]
|
||||
inc jq
|
||||
jl .loop_order
|
||||
.end_order:
|
||||
PMACSDQL m2, m0, m1, m2, m0
|
||||
psrlq m2, m4
|
||||
movd m0, [decodedq]
|
||||
paddd m0, m2
|
||||
movd [decodedq], m0
|
||||
sub lend, 2
|
||||
jl .ret
|
||||
PMACSDQL m3, m1, m0, m3, m1
|
||||
psrlq m3, m4
|
||||
movd m1, [decodedq+4]
|
||||
paddd m1, m3
|
||||
movd [decodedq+4], m1
|
||||
jg .loop_sample
|
||||
.ret:
|
||||
REP_RET
|
||||
%endmacro
|
||||
|
||||
%if HAVE_XOP_EXTERNAL
|
||||
LPC_32 xop
|
||||
%endif
|
||||
LPC_32 sse4
|
||||
|
||||
;----------------------------------------------------------------------------------
|
||||
;void ff_flac_decorrelate_[lrm]s_16_sse2(uint8_t **out, int32_t **in, int channels,
|
||||
; int len, int shift);
|
||||
;----------------------------------------------------------------------------------
|
||||
%macro FLAC_DECORRELATE_16 3-4
|
||||
cglobal flac_decorrelate_%1_16, 2, 4, 4, out, in0, in1, len
|
||||
%if ARCH_X86_32
|
||||
mov lend, lenm
|
||||
%endif
|
||||
movd m3, r4m
|
||||
shl lend, 2
|
||||
mov in1q, [in0q + gprsize]
|
||||
mov in0q, [in0q]
|
||||
mov outq, [outq]
|
||||
add in1q, lenq
|
||||
add in0q, lenq
|
||||
add outq, lenq
|
||||
neg lenq
|
||||
|
||||
align 16
|
||||
.loop:
|
||||
mova m0, [in0q + lenq]
|
||||
mova m1, [in1q + lenq]
|
||||
%ifidn %1, ms
|
||||
psrad m2, m1, 1
|
||||
psubd m0, m2
|
||||
%endif
|
||||
%ifnidn %1, indep2
|
||||
p%4d m2, m0, m1
|
||||
%endif
|
||||
packssdw m%2, m%2
|
||||
packssdw m%3, m%3
|
||||
punpcklwd m%2, m%3
|
||||
psllw m%2, m3
|
||||
mova [outq + lenq], m%2
|
||||
add lenq, 16
|
||||
jl .loop
|
||||
REP_RET
|
||||
%endmacro
|
||||
|
||||
INIT_XMM sse2
|
||||
FLAC_DECORRELATE_16 ls, 0, 2, sub
|
||||
FLAC_DECORRELATE_16 rs, 2, 1, add
|
||||
FLAC_DECORRELATE_16 ms, 2, 0, add
|
||||
|
||||
;----------------------------------------------------------------------------------
|
||||
;void ff_flac_decorrelate_[lrm]s_32_sse2(uint8_t **out, int32_t **in, int channels,
|
||||
; int len, int shift);
|
||||
;----------------------------------------------------------------------------------
|
||||
%macro FLAC_DECORRELATE_32 5
|
||||
cglobal flac_decorrelate_%1_32, 2, 4, 4, out, in0, in1, len
|
||||
%if ARCH_X86_32
|
||||
mov lend, lenm
|
||||
%endif
|
||||
movd m3, r4m
|
||||
mov in1q, [in0q + gprsize]
|
||||
mov in0q, [in0q]
|
||||
mov outq, [outq]
|
||||
sub in1q, in0q
|
||||
|
||||
align 16
|
||||
.loop:
|
||||
mova m0, [in0q]
|
||||
mova m1, [in0q + in1q]
|
||||
%ifidn %1, ms
|
||||
psrad m2, m1, 1
|
||||
psubd m0, m2
|
||||
%endif
|
||||
p%5d m2, m0, m1
|
||||
pslld m%2, m3
|
||||
pslld m%3, m3
|
||||
|
||||
SBUTTERFLY dq, %2, %3, %4
|
||||
|
||||
mova [outq ], m%2
|
||||
mova [outq + mmsize], m%3
|
||||
|
||||
add in0q, mmsize
|
||||
add outq, mmsize*2
|
||||
sub lend, mmsize/4
|
||||
jg .loop
|
||||
REP_RET
|
||||
%endmacro
|
||||
|
||||
INIT_XMM sse2
|
||||
FLAC_DECORRELATE_32 ls, 0, 2, 1, sub
|
||||
FLAC_DECORRELATE_32 rs, 2, 1, 0, add
|
||||
FLAC_DECORRELATE_32 ms, 2, 0, 1, add
|
||||
|
||||
;-----------------------------------------------------------------------------------------
|
||||
;void ff_flac_decorrelate_indep<ch>_<bps>_<opt>(uint8_t **out, int32_t **in, int channels,
|
||||
; int len, int shift);
|
||||
;-----------------------------------------------------------------------------------------
|
||||
;%1 = bps
|
||||
;%2 = channels
|
||||
;%3 = last xmm reg used
|
||||
;%4 = word/dword (shift instruction)
|
||||
%macro FLAC_DECORRELATE_INDEP 4
|
||||
%define REPCOUNT %2/(32/%1) ; 16bits = channels / 2; 32bits = channels
|
||||
cglobal flac_decorrelate_indep%2_%1, 2, %2+2, %3+1, out, in0, in1, len, in2, in3, in4, in5, in6, in7
|
||||
%if ARCH_X86_32
|
||||
%if %2 == 6
|
||||
DEFINE_ARGS out, in0, in1, in2, in3, in4, in5
|
||||
%define lend dword r3m
|
||||
%else
|
||||
mov lend, lenm
|
||||
%endif
|
||||
%endif
|
||||
movd m%3, r4m
|
||||
|
||||
%assign %%i 1
|
||||
%rep %2-1
|
||||
mov in %+ %%i %+ q, [in0q+%%i*gprsize]
|
||||
%assign %%i %%i+1
|
||||
%endrep
|
||||
|
||||
mov in0q, [in0q]
|
||||
mov outq, [outq]
|
||||
|
||||
%assign %%i 1
|
||||
%rep %2-1
|
||||
sub in %+ %%i %+ q, in0q
|
||||
%assign %%i %%i+1
|
||||
%endrep
|
||||
|
||||
align 16
|
||||
.loop:
|
||||
mova m0, [in0q]
|
||||
|
||||
%assign %%i 1
|
||||
%rep REPCOUNT-1
|
||||
mova m %+ %%i, [in0q + in %+ %%i %+ q]
|
||||
%assign %%i %%i+1
|
||||
%endrep
|
||||
|
||||
%if %1 == 32
|
||||
|
||||
%if %2 == 8
|
||||
TRANSPOSE8x4D 0, 1, 2, 3, 4, 5, 6, 7, 8
|
||||
%elif %2 == 6
|
||||
SBUTTERFLY dq, 0, 1, 6
|
||||
SBUTTERFLY dq, 2, 3, 6
|
||||
SBUTTERFLY dq, 4, 5, 6
|
||||
|
||||
punpcklqdq m6, m0, m2
|
||||
punpckhqdq m2, m4
|
||||
shufps m4, m0, 0xe4
|
||||
punpcklqdq m0, m1, m3
|
||||
punpckhqdq m3, m5
|
||||
shufps m5, m1, 0xe4
|
||||
SWAP 0,6,1,4,5,3
|
||||
%elif %2 == 4
|
||||
TRANSPOSE4x4D 0, 1, 2, 3, 4
|
||||
%else ; %2 == 2
|
||||
SBUTTERFLY dq, 0, 1, 2
|
||||
%endif
|
||||
|
||||
%else ; %1 == 16
|
||||
|
||||
%if %2 == 8
|
||||
packssdw m0, [in0q + in4q]
|
||||
packssdw m1, [in0q + in5q]
|
||||
packssdw m2, [in0q + in6q]
|
||||
packssdw m3, [in0q + in7q]
|
||||
TRANSPOSE2x4x4W 0, 1, 2, 3, 4
|
||||
%elif %2 == 6
|
||||
packssdw m0, [in0q + in3q]
|
||||
packssdw m1, [in0q + in4q]
|
||||
packssdw m2, [in0q + in5q]
|
||||
pshufd m3, m0, q1032
|
||||
punpcklwd m0, m1
|
||||
punpckhwd m1, m2
|
||||
punpcklwd m2, m3
|
||||
|
||||
shufps m3, m0, m2, q2020
|
||||
shufps m0, m1, q2031
|
||||
shufps m2, m1, q3131
|
||||
shufps m1, m2, m3, q3120
|
||||
shufps m3, m0, q0220
|
||||
shufps m0, m2, q3113
|
||||
SWAP 2, 0, 3
|
||||
%else ; %2 == 4
|
||||
packssdw m0, [in0q + in2q]
|
||||
packssdw m1, [in0q + in3q]
|
||||
SBUTTERFLY wd, 0, 1, 2
|
||||
SBUTTERFLY dq, 0, 1, 2
|
||||
%endif
|
||||
|
||||
%endif
|
||||
|
||||
%assign %%i 0
|
||||
%rep REPCOUNT
|
||||
psll%4 m %+ %%i, m%3
|
||||
%assign %%i %%i+1
|
||||
%endrep
|
||||
|
||||
%assign %%i 0
|
||||
%rep REPCOUNT
|
||||
mova [outq + %%i*mmsize], m %+ %%i
|
||||
%assign %%i %%i+1
|
||||
%endrep
|
||||
|
||||
add in0q, mmsize
|
||||
add outq, mmsize*REPCOUNT
|
||||
sub lend, mmsize/4
|
||||
jg .loop
|
||||
REP_RET
|
||||
%endmacro
|
||||
|
||||
INIT_XMM sse2
|
||||
FLAC_DECORRELATE_16 indep2, 0, 1 ; Reuse stereo 16bits macro
|
||||
FLAC_DECORRELATE_INDEP 32, 2, 3, d
|
||||
FLAC_DECORRELATE_INDEP 16, 4, 3, w
|
||||
FLAC_DECORRELATE_INDEP 32, 4, 5, d
|
||||
FLAC_DECORRELATE_INDEP 16, 6, 4, w
|
||||
FLAC_DECORRELATE_INDEP 32, 6, 7, d
|
||||
%if ARCH_X86_64
|
||||
FLAC_DECORRELATE_INDEP 16, 8, 5, w
|
||||
FLAC_DECORRELATE_INDEP 32, 8, 9, d
|
||||
%endif
|
||||
|
||||
INIT_XMM avx
|
||||
FLAC_DECORRELATE_INDEP 32, 4, 5, d
|
||||
FLAC_DECORRELATE_INDEP 32, 6, 7, d
|
||||
%if ARCH_X86_64
|
||||
FLAC_DECORRELATE_INDEP 16, 8, 5, w
|
||||
FLAC_DECORRELATE_INDEP 32, 8, 9, d
|
||||
%endif
|
||||
115
media/ffvpx/libavcodec/x86/flacdsp_init.c
Normal file
115
media/ffvpx/libavcodec/x86/flacdsp_init.c
Normal file
|
|
@ -0,0 +1,115 @@
|
|||
/*
|
||||
* Copyright (c) 2014 James Almer
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#include "libavcodec/flacdsp.h"
|
||||
#include "libavutil/x86/cpu.h"
|
||||
#include "config.h"
|
||||
|
||||
void ff_flac_lpc_32_sse4(int32_t *samples, const int coeffs[32], int order,
|
||||
int qlevel, int len);
|
||||
void ff_flac_lpc_32_xop(int32_t *samples, const int coeffs[32], int order,
|
||||
int qlevel, int len);
|
||||
|
||||
void ff_flac_enc_lpc_16_sse4(int32_t *, const int32_t *, int, int, const int32_t *,int);
|
||||
|
||||
#define DECORRELATE_FUNCS(fmt, opt) \
|
||||
void ff_flac_decorrelate_ls_##fmt##_##opt(uint8_t **out, int32_t **in, int channels, \
|
||||
int len, int shift); \
|
||||
void ff_flac_decorrelate_rs_##fmt##_##opt(uint8_t **out, int32_t **in, int channels, \
|
||||
int len, int shift); \
|
||||
void ff_flac_decorrelate_ms_##fmt##_##opt(uint8_t **out, int32_t **in, int channels, \
|
||||
int len, int shift); \
|
||||
void ff_flac_decorrelate_indep2_##fmt##_##opt(uint8_t **out, int32_t **in, int channels, \
|
||||
int len, int shift); \
|
||||
void ff_flac_decorrelate_indep4_##fmt##_##opt(uint8_t **out, int32_t **in, int channels, \
|
||||
int len, int shift); \
|
||||
void ff_flac_decorrelate_indep6_##fmt##_##opt(uint8_t **out, int32_t **in, int channels, \
|
||||
int len, int shift); \
|
||||
void ff_flac_decorrelate_indep8_##fmt##_##opt(uint8_t **out, int32_t **in, int channels, \
|
||||
int len, int shift)
|
||||
|
||||
DECORRELATE_FUNCS(16, sse2);
|
||||
DECORRELATE_FUNCS(16, avx);
|
||||
DECORRELATE_FUNCS(32, sse2);
|
||||
DECORRELATE_FUNCS(32, avx);
|
||||
|
||||
av_cold void ff_flacdsp_init_x86(FLACDSPContext *c, enum AVSampleFormat fmt, int channels,
|
||||
int bps)
|
||||
{
|
||||
#if HAVE_YASM
|
||||
int cpu_flags = av_get_cpu_flags();
|
||||
|
||||
#if CONFIG_FLAC_DECODER
|
||||
if (EXTERNAL_SSE2(cpu_flags)) {
|
||||
if (fmt == AV_SAMPLE_FMT_S16) {
|
||||
if (channels == 2)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep2_16_sse2;
|
||||
else if (channels == 4)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep4_16_sse2;
|
||||
else if (channels == 6)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep6_16_sse2;
|
||||
else if (ARCH_X86_64 && channels == 8)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep8_16_sse2;
|
||||
c->decorrelate[1] = ff_flac_decorrelate_ls_16_sse2;
|
||||
c->decorrelate[2] = ff_flac_decorrelate_rs_16_sse2;
|
||||
c->decorrelate[3] = ff_flac_decorrelate_ms_16_sse2;
|
||||
} else if (fmt == AV_SAMPLE_FMT_S32) {
|
||||
if (channels == 2)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep2_32_sse2;
|
||||
else if (channels == 4)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep4_32_sse2;
|
||||
else if (channels == 6)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep6_32_sse2;
|
||||
else if (ARCH_X86_64 && channels == 8)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep8_32_sse2;
|
||||
c->decorrelate[1] = ff_flac_decorrelate_ls_32_sse2;
|
||||
c->decorrelate[2] = ff_flac_decorrelate_rs_32_sse2;
|
||||
c->decorrelate[3] = ff_flac_decorrelate_ms_32_sse2;
|
||||
}
|
||||
}
|
||||
if (EXTERNAL_SSE4(cpu_flags)) {
|
||||
c->lpc32 = ff_flac_lpc_32_sse4;
|
||||
}
|
||||
if (EXTERNAL_AVX(cpu_flags)) {
|
||||
if (fmt == AV_SAMPLE_FMT_S16) {
|
||||
if (ARCH_X86_64 && channels == 8)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep8_16_avx;
|
||||
} else if (fmt == AV_SAMPLE_FMT_S32) {
|
||||
if (channels == 4)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep4_32_avx;
|
||||
else if (channels == 6)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep6_32_avx;
|
||||
else if (ARCH_X86_64 && channels == 8)
|
||||
c->decorrelate[0] = ff_flac_decorrelate_indep8_32_avx;
|
||||
}
|
||||
}
|
||||
if (EXTERNAL_XOP(cpu_flags)) {
|
||||
c->lpc32 = ff_flac_lpc_32_xop;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if CONFIG_FLAC_ENCODER
|
||||
if (EXTERNAL_SSE4(cpu_flags)) {
|
||||
if (CONFIG_GPL)
|
||||
c->lpc16_encode = ff_flac_enc_lpc_16_sse4;
|
||||
}
|
||||
#endif
|
||||
#endif /* HAVE_YASM */
|
||||
}
|
||||
212
media/ffvpx/libavcodec/x86/h264_i386.h
Normal file
212
media/ffvpx/libavcodec/x86/h264_i386.h
Normal file
|
|
@ -0,0 +1,212 @@
|
|||
/*
|
||||
* H.26L/H.264/AVC/JVT/14496-10/... encoder/decoder
|
||||
* Copyright (c) 2003 Michael Niedermayer <michaelni@gmx.at>
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
/**
|
||||
* @file
|
||||
* H.264 / AVC / MPEG-4 part10 codec.
|
||||
* non-MMX i386-specific optimizations for H.264
|
||||
* @author Michael Niedermayer <michaelni@gmx.at>
|
||||
*/
|
||||
|
||||
#ifndef AVCODEC_X86_H264_I386_H
|
||||
#define AVCODEC_X86_H264_I386_H
|
||||
|
||||
#include <stddef.h>
|
||||
|
||||
#include "libavcodec/cabac.h"
|
||||
#include "cabac.h"
|
||||
|
||||
#if HAVE_INLINE_ASM
|
||||
|
||||
#if ARCH_X86_64
|
||||
#define REG64 "r"
|
||||
#else
|
||||
#define REG64 "m"
|
||||
#endif
|
||||
|
||||
//FIXME use some macros to avoid duplicating get_cabac (cannot be done yet
|
||||
//as that would make optimization work hard)
|
||||
#if HAVE_7REGS && !BROKEN_COMPILER
|
||||
#define decode_significance decode_significance_x86
|
||||
static int decode_significance_x86(CABACContext *c, int max_coeff,
|
||||
uint8_t *significant_coeff_ctx_base,
|
||||
int *index, x86_reg last_off){
|
||||
void *end= significant_coeff_ctx_base + max_coeff - 1;
|
||||
int minusstart= -(intptr_t)significant_coeff_ctx_base;
|
||||
int minusindex= 4-(intptr_t)index;
|
||||
int bit;
|
||||
x86_reg coeff_count;
|
||||
|
||||
#ifdef BROKEN_RELOCATIONS
|
||||
void *tables;
|
||||
|
||||
__asm__ volatile(
|
||||
"lea "MANGLE(ff_h264_cabac_tables)", %0 \n\t"
|
||||
: "=&r"(tables)
|
||||
: NAMED_CONSTRAINTS_ARRAY(ff_h264_cabac_tables)
|
||||
);
|
||||
#endif
|
||||
|
||||
__asm__ volatile(
|
||||
"3: \n\t"
|
||||
|
||||
BRANCHLESS_GET_CABAC("%4", "%q4", "(%1)", "%3", "%w3",
|
||||
"%5", "%q5", "%k0", "%b0",
|
||||
"%c11(%6)", "%c12(%6)",
|
||||
AV_STRINGIFY(H264_NORM_SHIFT_OFFSET),
|
||||
AV_STRINGIFY(H264_LPS_RANGE_OFFSET),
|
||||
AV_STRINGIFY(H264_MLPS_STATE_OFFSET),
|
||||
"%13")
|
||||
|
||||
"test $1, %4 \n\t"
|
||||
" jz 4f \n\t"
|
||||
"add %10, %1 \n\t"
|
||||
|
||||
BRANCHLESS_GET_CABAC("%4", "%q4", "(%1)", "%3", "%w3",
|
||||
"%5", "%q5", "%k0", "%b0",
|
||||
"%c11(%6)", "%c12(%6)",
|
||||
AV_STRINGIFY(H264_NORM_SHIFT_OFFSET),
|
||||
AV_STRINGIFY(H264_LPS_RANGE_OFFSET),
|
||||
AV_STRINGIFY(H264_MLPS_STATE_OFFSET),
|
||||
"%13")
|
||||
|
||||
"sub %10, %1 \n\t"
|
||||
"mov %2, %0 \n\t"
|
||||
"movl %7, %%ecx \n\t"
|
||||
"add %1, %%"REG_c" \n\t"
|
||||
"movl %%ecx, (%0) \n\t"
|
||||
|
||||
"test $1, %4 \n\t"
|
||||
" jnz 5f \n\t"
|
||||
|
||||
"add"OPSIZE" $4, %2 \n\t"
|
||||
|
||||
"4: \n\t"
|
||||
"add $1, %1 \n\t"
|
||||
"cmp %8, %1 \n\t"
|
||||
" jb 3b \n\t"
|
||||
"mov %2, %0 \n\t"
|
||||
"movl %7, %%ecx \n\t"
|
||||
"add %1, %%"REG_c" \n\t"
|
||||
"movl %%ecx, (%0) \n\t"
|
||||
"5: \n\t"
|
||||
"add %9, %k0 \n\t"
|
||||
"shr $2, %k0 \n\t"
|
||||
: "=&q"(coeff_count), "+r"(significant_coeff_ctx_base), "+m"(index),
|
||||
"+&r"(c->low), "=&r"(bit), "+&r"(c->range)
|
||||
: "r"(c), "m"(minusstart), "m"(end), "m"(minusindex), "m"(last_off),
|
||||
"i"(offsetof(CABACContext, bytestream)),
|
||||
"i"(offsetof(CABACContext, bytestream_end))
|
||||
TABLES_ARG
|
||||
: "%"REG_c, "memory"
|
||||
);
|
||||
return coeff_count;
|
||||
}
|
||||
|
||||
#define decode_significance_8x8 decode_significance_8x8_x86
|
||||
static int decode_significance_8x8_x86(CABACContext *c,
|
||||
uint8_t *significant_coeff_ctx_base,
|
||||
int *index, uint8_t *last_coeff_ctx_base, const uint8_t *sig_off){
|
||||
int minusindex= 4-(intptr_t)index;
|
||||
int bit;
|
||||
x86_reg coeff_count;
|
||||
x86_reg last=0;
|
||||
x86_reg state;
|
||||
|
||||
#ifdef BROKEN_RELOCATIONS
|
||||
void *tables;
|
||||
|
||||
__asm__ volatile(
|
||||
"lea "MANGLE(ff_h264_cabac_tables)", %0 \n\t"
|
||||
: "=&r"(tables)
|
||||
: NAMED_CONSTRAINTS_ARRAY(ff_h264_cabac_tables)
|
||||
);
|
||||
#endif
|
||||
|
||||
__asm__ volatile(
|
||||
"mov %1, %6 \n\t"
|
||||
"3: \n\t"
|
||||
|
||||
"mov %10, %0 \n\t"
|
||||
"movzb (%0, %6), %6 \n\t"
|
||||
"add %9, %6 \n\t"
|
||||
|
||||
BRANCHLESS_GET_CABAC("%4", "%q4", "(%6)", "%3", "%w3",
|
||||
"%5", "%q5", "%k0", "%b0",
|
||||
"%c12(%7)", "%c13(%7)",
|
||||
AV_STRINGIFY(H264_NORM_SHIFT_OFFSET),
|
||||
AV_STRINGIFY(H264_LPS_RANGE_OFFSET),
|
||||
AV_STRINGIFY(H264_MLPS_STATE_OFFSET),
|
||||
"%15")
|
||||
|
||||
"mov %1, %6 \n\t"
|
||||
"test $1, %4 \n\t"
|
||||
" jz 4f \n\t"
|
||||
|
||||
#ifdef BROKEN_RELOCATIONS
|
||||
"movzb %c14(%15, %q6), %6\n\t"
|
||||
#else
|
||||
"movzb "MANGLE(ff_h264_cabac_tables)"+%c14(%6), %6\n\t"
|
||||
#endif
|
||||
"add %11, %6 \n\t"
|
||||
|
||||
BRANCHLESS_GET_CABAC("%4", "%q4", "(%6)", "%3", "%w3",
|
||||
"%5", "%q5", "%k0", "%b0",
|
||||
"%c12(%7)", "%c13(%7)",
|
||||
AV_STRINGIFY(H264_NORM_SHIFT_OFFSET),
|
||||
AV_STRINGIFY(H264_LPS_RANGE_OFFSET),
|
||||
AV_STRINGIFY(H264_MLPS_STATE_OFFSET),
|
||||
"%15")
|
||||
|
||||
"mov %2, %0 \n\t"
|
||||
"mov %1, %6 \n\t"
|
||||
"mov %k6, (%0) \n\t"
|
||||
|
||||
"test $1, %4 \n\t"
|
||||
" jnz 5f \n\t"
|
||||
|
||||
"add"OPSIZE" $4, %2 \n\t"
|
||||
|
||||
"4: \n\t"
|
||||
"add $1, %6 \n\t"
|
||||
"mov %6, %1 \n\t"
|
||||
"cmp $63, %6 \n\t"
|
||||
" jb 3b \n\t"
|
||||
"mov %2, %0 \n\t"
|
||||
"mov %k6, (%0) \n\t"
|
||||
"5: \n\t"
|
||||
"addl %8, %k0 \n\t"
|
||||
"shr $2, %k0 \n\t"
|
||||
: "=&q"(coeff_count), "+"REG64(last), "+"REG64(index), "+&r"(c->low),
|
||||
"=&r"(bit), "+&r"(c->range), "=&r"(state)
|
||||
: "r"(c), "m"(minusindex), "m"(significant_coeff_ctx_base),
|
||||
REG64(sig_off), REG64(last_coeff_ctx_base),
|
||||
"i"(offsetof(CABACContext, bytestream)),
|
||||
"i"(offsetof(CABACContext, bytestream_end)),
|
||||
"i"(H264_LAST_COEFF_FLAG_OFFSET_8x8_OFFSET) TABLES_ARG
|
||||
: "%"REG_c, "memory"
|
||||
);
|
||||
return coeff_count;
|
||||
}
|
||||
#endif /* HAVE_7REGS && BROKEN_COMPILER */
|
||||
|
||||
#endif /* HAVE_INLINE_ASM */
|
||||
#endif /* AVCODEC_X86_H264_I386_H */
|
||||
2717
media/ffvpx/libavcodec/x86/h264_intrapred.asm
Normal file
2717
media/ffvpx/libavcodec/x86/h264_intrapred.asm
Normal file
File diff suppressed because it is too large
Load diff
1192
media/ffvpx/libavcodec/x86/h264_intrapred_10bit.asm
Normal file
1192
media/ffvpx/libavcodec/x86/h264_intrapred_10bit.asm
Normal file
File diff suppressed because it is too large
Load diff
403
media/ffvpx/libavcodec/x86/h264_intrapred_init.c
Normal file
403
media/ffvpx/libavcodec/x86/h264_intrapred_init.c
Normal file
|
|
@ -0,0 +1,403 @@
|
|||
/*
|
||||
* Copyright (c) 2010 Fiona Glaser <fiona@x264.com>
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#include "libavutil/attributes.h"
|
||||
#include "libavutil/cpu.h"
|
||||
#include "libavutil/x86/cpu.h"
|
||||
#include "libavcodec/avcodec.h"
|
||||
#include "libavcodec/h264pred.h"
|
||||
|
||||
#define PRED4x4(TYPE, DEPTH, OPT) \
|
||||
void ff_pred4x4_ ## TYPE ## _ ## DEPTH ## _ ## OPT (uint8_t *src, \
|
||||
const uint8_t *topright, \
|
||||
ptrdiff_t stride);
|
||||
|
||||
PRED4x4(dc, 10, mmxext)
|
||||
PRED4x4(down_left, 10, sse2)
|
||||
PRED4x4(down_left, 10, avx)
|
||||
PRED4x4(down_right, 10, sse2)
|
||||
PRED4x4(down_right, 10, ssse3)
|
||||
PRED4x4(down_right, 10, avx)
|
||||
PRED4x4(vertical_left, 10, sse2)
|
||||
PRED4x4(vertical_left, 10, avx)
|
||||
PRED4x4(vertical_right, 10, sse2)
|
||||
PRED4x4(vertical_right, 10, ssse3)
|
||||
PRED4x4(vertical_right, 10, avx)
|
||||
PRED4x4(horizontal_up, 10, mmxext)
|
||||
PRED4x4(horizontal_down, 10, sse2)
|
||||
PRED4x4(horizontal_down, 10, ssse3)
|
||||
PRED4x4(horizontal_down, 10, avx)
|
||||
|
||||
#define PRED8x8(TYPE, DEPTH, OPT) \
|
||||
void ff_pred8x8_ ## TYPE ## _ ## DEPTH ## _ ## OPT (uint8_t *src, \
|
||||
ptrdiff_t stride);
|
||||
|
||||
PRED8x8(dc, 10, mmxext)
|
||||
PRED8x8(dc, 10, sse2)
|
||||
PRED8x8(top_dc, 10, sse2)
|
||||
PRED8x8(plane, 10, sse2)
|
||||
PRED8x8(vertical, 10, sse2)
|
||||
PRED8x8(horizontal, 10, sse2)
|
||||
|
||||
#define PRED8x8L(TYPE, DEPTH, OPT)\
|
||||
void ff_pred8x8l_ ## TYPE ## _ ## DEPTH ## _ ## OPT (uint8_t *src, \
|
||||
int has_topleft, \
|
||||
int has_topright, \
|
||||
ptrdiff_t stride);
|
||||
|
||||
PRED8x8L(dc, 10, sse2)
|
||||
PRED8x8L(dc, 10, avx)
|
||||
PRED8x8L(128_dc, 10, mmxext)
|
||||
PRED8x8L(128_dc, 10, sse2)
|
||||
PRED8x8L(top_dc, 10, sse2)
|
||||
PRED8x8L(top_dc, 10, avx)
|
||||
PRED8x8L(vertical, 10, sse2)
|
||||
PRED8x8L(vertical, 10, avx)
|
||||
PRED8x8L(horizontal, 10, sse2)
|
||||
PRED8x8L(horizontal, 10, ssse3)
|
||||
PRED8x8L(horizontal, 10, avx)
|
||||
PRED8x8L(down_left, 10, sse2)
|
||||
PRED8x8L(down_left, 10, ssse3)
|
||||
PRED8x8L(down_left, 10, avx)
|
||||
PRED8x8L(down_right, 10, sse2)
|
||||
PRED8x8L(down_right, 10, ssse3)
|
||||
PRED8x8L(down_right, 10, avx)
|
||||
PRED8x8L(vertical_right, 10, sse2)
|
||||
PRED8x8L(vertical_right, 10, ssse3)
|
||||
PRED8x8L(vertical_right, 10, avx)
|
||||
PRED8x8L(horizontal_up, 10, sse2)
|
||||
PRED8x8L(horizontal_up, 10, ssse3)
|
||||
PRED8x8L(horizontal_up, 10, avx)
|
||||
|
||||
#define PRED16x16(TYPE, DEPTH, OPT)\
|
||||
void ff_pred16x16_ ## TYPE ## _ ## DEPTH ## _ ## OPT (uint8_t *src, \
|
||||
ptrdiff_t stride);
|
||||
|
||||
PRED16x16(dc, 10, mmxext)
|
||||
PRED16x16(dc, 10, sse2)
|
||||
PRED16x16(top_dc, 10, mmxext)
|
||||
PRED16x16(top_dc, 10, sse2)
|
||||
PRED16x16(128_dc, 10, mmxext)
|
||||
PRED16x16(128_dc, 10, sse2)
|
||||
PRED16x16(left_dc, 10, mmxext)
|
||||
PRED16x16(left_dc, 10, sse2)
|
||||
PRED16x16(vertical, 10, mmxext)
|
||||
PRED16x16(vertical, 10, sse2)
|
||||
PRED16x16(horizontal, 10, mmxext)
|
||||
PRED16x16(horizontal, 10, sse2)
|
||||
|
||||
/* 8-bit versions */
|
||||
PRED16x16(vertical, 8, mmx)
|
||||
PRED16x16(vertical, 8, sse)
|
||||
PRED16x16(horizontal, 8, mmx)
|
||||
PRED16x16(horizontal, 8, mmxext)
|
||||
PRED16x16(horizontal, 8, ssse3)
|
||||
PRED16x16(dc, 8, mmxext)
|
||||
PRED16x16(dc, 8, sse2)
|
||||
PRED16x16(dc, 8, ssse3)
|
||||
PRED16x16(plane_h264, 8, mmx)
|
||||
PRED16x16(plane_h264, 8, mmxext)
|
||||
PRED16x16(plane_h264, 8, sse2)
|
||||
PRED16x16(plane_h264, 8, ssse3)
|
||||
PRED16x16(plane_rv40, 8, mmx)
|
||||
PRED16x16(plane_rv40, 8, mmxext)
|
||||
PRED16x16(plane_rv40, 8, sse2)
|
||||
PRED16x16(plane_rv40, 8, ssse3)
|
||||
PRED16x16(plane_svq3, 8, mmx)
|
||||
PRED16x16(plane_svq3, 8, mmxext)
|
||||
PRED16x16(plane_svq3, 8, sse2)
|
||||
PRED16x16(plane_svq3, 8, ssse3)
|
||||
PRED16x16(tm_vp8, 8, mmx)
|
||||
PRED16x16(tm_vp8, 8, mmxext)
|
||||
PRED16x16(tm_vp8, 8, sse2)
|
||||
|
||||
PRED8x8(top_dc, 8, mmxext)
|
||||
PRED8x8(dc_rv40, 8, mmxext)
|
||||
PRED8x8(dc, 8, mmxext)
|
||||
PRED8x8(vertical, 8, mmx)
|
||||
PRED8x8(horizontal, 8, mmx)
|
||||
PRED8x8(horizontal, 8, mmxext)
|
||||
PRED8x8(horizontal, 8, ssse3)
|
||||
PRED8x8(plane, 8, mmx)
|
||||
PRED8x8(plane, 8, mmxext)
|
||||
PRED8x8(plane, 8, sse2)
|
||||
PRED8x8(plane, 8, ssse3)
|
||||
PRED8x8(tm_vp8, 8, mmx)
|
||||
PRED8x8(tm_vp8, 8, mmxext)
|
||||
PRED8x8(tm_vp8, 8, sse2)
|
||||
PRED8x8(tm_vp8, 8, ssse3)
|
||||
|
||||
PRED8x8L(top_dc, 8, mmxext)
|
||||
PRED8x8L(top_dc, 8, ssse3)
|
||||
PRED8x8L(dc, 8, mmxext)
|
||||
PRED8x8L(dc, 8, ssse3)
|
||||
PRED8x8L(horizontal, 8, mmxext)
|
||||
PRED8x8L(horizontal, 8, ssse3)
|
||||
PRED8x8L(vertical, 8, mmxext)
|
||||
PRED8x8L(vertical, 8, ssse3)
|
||||
PRED8x8L(down_left, 8, mmxext)
|
||||
PRED8x8L(down_left, 8, sse2)
|
||||
PRED8x8L(down_left, 8, ssse3)
|
||||
PRED8x8L(down_right, 8, mmxext)
|
||||
PRED8x8L(down_right, 8, sse2)
|
||||
PRED8x8L(down_right, 8, ssse3)
|
||||
PRED8x8L(vertical_right, 8, mmxext)
|
||||
PRED8x8L(vertical_right, 8, sse2)
|
||||
PRED8x8L(vertical_right, 8, ssse3)
|
||||
PRED8x8L(vertical_left, 8, sse2)
|
||||
PRED8x8L(vertical_left, 8, ssse3)
|
||||
PRED8x8L(horizontal_up, 8, mmxext)
|
||||
PRED8x8L(horizontal_up, 8, ssse3)
|
||||
PRED8x8L(horizontal_down, 8, mmxext)
|
||||
PRED8x8L(horizontal_down, 8, sse2)
|
||||
PRED8x8L(horizontal_down, 8, ssse3)
|
||||
|
||||
PRED4x4(dc, 8, mmxext)
|
||||
PRED4x4(down_left, 8, mmxext)
|
||||
PRED4x4(down_right, 8, mmxext)
|
||||
PRED4x4(vertical_left, 8, mmxext)
|
||||
PRED4x4(vertical_right, 8, mmxext)
|
||||
PRED4x4(horizontal_up, 8, mmxext)
|
||||
PRED4x4(horizontal_down, 8, mmxext)
|
||||
PRED4x4(tm_vp8, 8, mmx)
|
||||
PRED4x4(tm_vp8, 8, mmxext)
|
||||
PRED4x4(tm_vp8, 8, ssse3)
|
||||
PRED4x4(vertical_vp8, 8, mmxext)
|
||||
|
||||
av_cold void ff_h264_pred_init_x86(H264PredContext *h, int codec_id,
|
||||
const int bit_depth,
|
||||
const int chroma_format_idc)
|
||||
{
|
||||
int cpu_flags = av_get_cpu_flags();
|
||||
|
||||
if (bit_depth == 8) {
|
||||
if (EXTERNAL_MMX(cpu_flags)) {
|
||||
h->pred16x16[VERT_PRED8x8 ] = ff_pred16x16_vertical_8_mmx;
|
||||
h->pred16x16[HOR_PRED8x8 ] = ff_pred16x16_horizontal_8_mmx;
|
||||
if (chroma_format_idc <= 1) {
|
||||
h->pred8x8 [VERT_PRED8x8 ] = ff_pred8x8_vertical_8_mmx;
|
||||
h->pred8x8 [HOR_PRED8x8 ] = ff_pred8x8_horizontal_8_mmx;
|
||||
}
|
||||
if (codec_id == AV_CODEC_ID_VP7 || codec_id == AV_CODEC_ID_VP8) {
|
||||
h->pred16x16[PLANE_PRED8x8 ] = ff_pred16x16_tm_vp8_8_mmx;
|
||||
h->pred8x8 [PLANE_PRED8x8 ] = ff_pred8x8_tm_vp8_8_mmx;
|
||||
h->pred4x4 [TM_VP8_PRED ] = ff_pred4x4_tm_vp8_8_mmx;
|
||||
} else {
|
||||
if (chroma_format_idc <= 1)
|
||||
h->pred8x8 [PLANE_PRED8x8] = ff_pred8x8_plane_8_mmx;
|
||||
if (codec_id == AV_CODEC_ID_SVQ3) {
|
||||
if (cpu_flags & AV_CPU_FLAG_CMOV)
|
||||
h->pred16x16[PLANE_PRED8x8] = ff_pred16x16_plane_svq3_8_mmx;
|
||||
} else if (codec_id == AV_CODEC_ID_RV40) {
|
||||
h->pred16x16[PLANE_PRED8x8] = ff_pred16x16_plane_rv40_8_mmx;
|
||||
} else {
|
||||
h->pred16x16[PLANE_PRED8x8] = ff_pred16x16_plane_h264_8_mmx;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (EXTERNAL_MMXEXT(cpu_flags)) {
|
||||
h->pred16x16[HOR_PRED8x8 ] = ff_pred16x16_horizontal_8_mmxext;
|
||||
h->pred16x16[DC_PRED8x8 ] = ff_pred16x16_dc_8_mmxext;
|
||||
if (chroma_format_idc <= 1)
|
||||
h->pred8x8[HOR_PRED8x8 ] = ff_pred8x8_horizontal_8_mmxext;
|
||||
h->pred8x8l [TOP_DC_PRED ] = ff_pred8x8l_top_dc_8_mmxext;
|
||||
h->pred8x8l [DC_PRED ] = ff_pred8x8l_dc_8_mmxext;
|
||||
h->pred8x8l [HOR_PRED ] = ff_pred8x8l_horizontal_8_mmxext;
|
||||
h->pred8x8l [VERT_PRED ] = ff_pred8x8l_vertical_8_mmxext;
|
||||
h->pred8x8l [DIAG_DOWN_RIGHT_PRED ] = ff_pred8x8l_down_right_8_mmxext;
|
||||
h->pred8x8l [VERT_RIGHT_PRED ] = ff_pred8x8l_vertical_right_8_mmxext;
|
||||
h->pred8x8l [HOR_UP_PRED ] = ff_pred8x8l_horizontal_up_8_mmxext;
|
||||
h->pred8x8l [DIAG_DOWN_LEFT_PRED ] = ff_pred8x8l_down_left_8_mmxext;
|
||||
h->pred8x8l [HOR_DOWN_PRED ] = ff_pred8x8l_horizontal_down_8_mmxext;
|
||||
h->pred4x4 [DIAG_DOWN_RIGHT_PRED ] = ff_pred4x4_down_right_8_mmxext;
|
||||
h->pred4x4 [VERT_RIGHT_PRED ] = ff_pred4x4_vertical_right_8_mmxext;
|
||||
h->pred4x4 [HOR_DOWN_PRED ] = ff_pred4x4_horizontal_down_8_mmxext;
|
||||
h->pred4x4 [DC_PRED ] = ff_pred4x4_dc_8_mmxext;
|
||||
if (codec_id == AV_CODEC_ID_VP7 || codec_id == AV_CODEC_ID_VP8 ||
|
||||
codec_id == AV_CODEC_ID_H264) {
|
||||
h->pred4x4 [DIAG_DOWN_LEFT_PRED] = ff_pred4x4_down_left_8_mmxext;
|
||||
}
|
||||
if (codec_id == AV_CODEC_ID_SVQ3 || codec_id == AV_CODEC_ID_H264) {
|
||||
h->pred4x4 [VERT_LEFT_PRED ] = ff_pred4x4_vertical_left_8_mmxext;
|
||||
}
|
||||
if (codec_id != AV_CODEC_ID_RV40) {
|
||||
h->pred4x4 [HOR_UP_PRED ] = ff_pred4x4_horizontal_up_8_mmxext;
|
||||
}
|
||||
if (codec_id == AV_CODEC_ID_SVQ3 || codec_id == AV_CODEC_ID_H264) {
|
||||
if (chroma_format_idc <= 1) {
|
||||
h->pred8x8[TOP_DC_PRED8x8 ] = ff_pred8x8_top_dc_8_mmxext;
|
||||
h->pred8x8[DC_PRED8x8 ] = ff_pred8x8_dc_8_mmxext;
|
||||
}
|
||||
}
|
||||
if (codec_id == AV_CODEC_ID_VP7 || codec_id == AV_CODEC_ID_VP8) {
|
||||
h->pred16x16[PLANE_PRED8x8 ] = ff_pred16x16_tm_vp8_8_mmxext;
|
||||
h->pred8x8 [DC_PRED8x8 ] = ff_pred8x8_dc_rv40_8_mmxext;
|
||||
h->pred8x8 [PLANE_PRED8x8 ] = ff_pred8x8_tm_vp8_8_mmxext;
|
||||
h->pred4x4 [TM_VP8_PRED ] = ff_pred4x4_tm_vp8_8_mmxext;
|
||||
h->pred4x4 [VERT_PRED ] = ff_pred4x4_vertical_vp8_8_mmxext;
|
||||
} else {
|
||||
if (chroma_format_idc <= 1)
|
||||
h->pred8x8 [PLANE_PRED8x8] = ff_pred8x8_plane_8_mmxext;
|
||||
if (codec_id == AV_CODEC_ID_SVQ3) {
|
||||
h->pred16x16[PLANE_PRED8x8 ] = ff_pred16x16_plane_svq3_8_mmxext;
|
||||
} else if (codec_id == AV_CODEC_ID_RV40) {
|
||||
h->pred16x16[PLANE_PRED8x8 ] = ff_pred16x16_plane_rv40_8_mmxext;
|
||||
} else {
|
||||
h->pred16x16[PLANE_PRED8x8 ] = ff_pred16x16_plane_h264_8_mmxext;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE(cpu_flags)) {
|
||||
h->pred16x16[VERT_PRED8x8] = ff_pred16x16_vertical_8_sse;
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE2(cpu_flags)) {
|
||||
h->pred16x16[DC_PRED8x8 ] = ff_pred16x16_dc_8_sse2;
|
||||
h->pred8x8l [DIAG_DOWN_LEFT_PRED ] = ff_pred8x8l_down_left_8_sse2;
|
||||
h->pred8x8l [DIAG_DOWN_RIGHT_PRED ] = ff_pred8x8l_down_right_8_sse2;
|
||||
h->pred8x8l [VERT_RIGHT_PRED ] = ff_pred8x8l_vertical_right_8_sse2;
|
||||
h->pred8x8l [VERT_LEFT_PRED ] = ff_pred8x8l_vertical_left_8_sse2;
|
||||
h->pred8x8l [HOR_DOWN_PRED ] = ff_pred8x8l_horizontal_down_8_sse2;
|
||||
if (codec_id == AV_CODEC_ID_VP7 || codec_id == AV_CODEC_ID_VP8) {
|
||||
h->pred16x16[PLANE_PRED8x8 ] = ff_pred16x16_tm_vp8_8_sse2;
|
||||
h->pred8x8 [PLANE_PRED8x8 ] = ff_pred8x8_tm_vp8_8_sse2;
|
||||
} else {
|
||||
if (chroma_format_idc <= 1)
|
||||
h->pred8x8 [PLANE_PRED8x8] = ff_pred8x8_plane_8_sse2;
|
||||
if (codec_id == AV_CODEC_ID_SVQ3) {
|
||||
h->pred16x16[PLANE_PRED8x8] = ff_pred16x16_plane_svq3_8_sse2;
|
||||
} else if (codec_id == AV_CODEC_ID_RV40) {
|
||||
h->pred16x16[PLANE_PRED8x8] = ff_pred16x16_plane_rv40_8_sse2;
|
||||
} else {
|
||||
h->pred16x16[PLANE_PRED8x8] = ff_pred16x16_plane_h264_8_sse2;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSSE3(cpu_flags)) {
|
||||
h->pred16x16[HOR_PRED8x8 ] = ff_pred16x16_horizontal_8_ssse3;
|
||||
h->pred16x16[DC_PRED8x8 ] = ff_pred16x16_dc_8_ssse3;
|
||||
if (chroma_format_idc <= 1)
|
||||
h->pred8x8 [HOR_PRED8x8 ] = ff_pred8x8_horizontal_8_ssse3;
|
||||
h->pred8x8l [TOP_DC_PRED ] = ff_pred8x8l_top_dc_8_ssse3;
|
||||
h->pred8x8l [DC_PRED ] = ff_pred8x8l_dc_8_ssse3;
|
||||
h->pred8x8l [HOR_PRED ] = ff_pred8x8l_horizontal_8_ssse3;
|
||||
h->pred8x8l [VERT_PRED ] = ff_pred8x8l_vertical_8_ssse3;
|
||||
h->pred8x8l [DIAG_DOWN_LEFT_PRED ] = ff_pred8x8l_down_left_8_ssse3;
|
||||
h->pred8x8l [DIAG_DOWN_RIGHT_PRED ] = ff_pred8x8l_down_right_8_ssse3;
|
||||
h->pred8x8l [VERT_RIGHT_PRED ] = ff_pred8x8l_vertical_right_8_ssse3;
|
||||
h->pred8x8l [VERT_LEFT_PRED ] = ff_pred8x8l_vertical_left_8_ssse3;
|
||||
h->pred8x8l [HOR_UP_PRED ] = ff_pred8x8l_horizontal_up_8_ssse3;
|
||||
h->pred8x8l [HOR_DOWN_PRED ] = ff_pred8x8l_horizontal_down_8_ssse3;
|
||||
if (codec_id == AV_CODEC_ID_VP7 || codec_id == AV_CODEC_ID_VP8) {
|
||||
h->pred8x8 [PLANE_PRED8x8 ] = ff_pred8x8_tm_vp8_8_ssse3;
|
||||
h->pred4x4 [TM_VP8_PRED ] = ff_pred4x4_tm_vp8_8_ssse3;
|
||||
} else {
|
||||
if (chroma_format_idc <= 1)
|
||||
h->pred8x8 [PLANE_PRED8x8] = ff_pred8x8_plane_8_ssse3;
|
||||
if (codec_id == AV_CODEC_ID_SVQ3) {
|
||||
h->pred16x16[PLANE_PRED8x8] = ff_pred16x16_plane_svq3_8_ssse3;
|
||||
} else if (codec_id == AV_CODEC_ID_RV40) {
|
||||
h->pred16x16[PLANE_PRED8x8] = ff_pred16x16_plane_rv40_8_ssse3;
|
||||
} else {
|
||||
h->pred16x16[PLANE_PRED8x8] = ff_pred16x16_plane_h264_8_ssse3;
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if (bit_depth == 10) {
|
||||
if (EXTERNAL_MMXEXT(cpu_flags)) {
|
||||
h->pred4x4[DC_PRED ] = ff_pred4x4_dc_10_mmxext;
|
||||
h->pred4x4[HOR_UP_PRED ] = ff_pred4x4_horizontal_up_10_mmxext;
|
||||
|
||||
if (chroma_format_idc <= 1)
|
||||
h->pred8x8[DC_PRED8x8 ] = ff_pred8x8_dc_10_mmxext;
|
||||
|
||||
h->pred8x8l[DC_128_PRED ] = ff_pred8x8l_128_dc_10_mmxext;
|
||||
|
||||
h->pred16x16[DC_PRED8x8 ] = ff_pred16x16_dc_10_mmxext;
|
||||
h->pred16x16[TOP_DC_PRED8x8 ] = ff_pred16x16_top_dc_10_mmxext;
|
||||
h->pred16x16[DC_128_PRED8x8 ] = ff_pred16x16_128_dc_10_mmxext;
|
||||
h->pred16x16[LEFT_DC_PRED8x8 ] = ff_pred16x16_left_dc_10_mmxext;
|
||||
h->pred16x16[VERT_PRED8x8 ] = ff_pred16x16_vertical_10_mmxext;
|
||||
h->pred16x16[HOR_PRED8x8 ] = ff_pred16x16_horizontal_10_mmxext;
|
||||
}
|
||||
if (EXTERNAL_SSE2(cpu_flags)) {
|
||||
h->pred4x4[DIAG_DOWN_LEFT_PRED ] = ff_pred4x4_down_left_10_sse2;
|
||||
h->pred4x4[DIAG_DOWN_RIGHT_PRED] = ff_pred4x4_down_right_10_sse2;
|
||||
h->pred4x4[VERT_LEFT_PRED ] = ff_pred4x4_vertical_left_10_sse2;
|
||||
h->pred4x4[VERT_RIGHT_PRED ] = ff_pred4x4_vertical_right_10_sse2;
|
||||
h->pred4x4[HOR_DOWN_PRED ] = ff_pred4x4_horizontal_down_10_sse2;
|
||||
|
||||
if (chroma_format_idc <= 1) {
|
||||
h->pred8x8[DC_PRED8x8 ] = ff_pred8x8_dc_10_sse2;
|
||||
h->pred8x8[TOP_DC_PRED8x8 ] = ff_pred8x8_top_dc_10_sse2;
|
||||
h->pred8x8[PLANE_PRED8x8 ] = ff_pred8x8_plane_10_sse2;
|
||||
h->pred8x8[VERT_PRED8x8 ] = ff_pred8x8_vertical_10_sse2;
|
||||
h->pred8x8[HOR_PRED8x8 ] = ff_pred8x8_horizontal_10_sse2;
|
||||
}
|
||||
|
||||
h->pred8x8l[VERT_PRED ] = ff_pred8x8l_vertical_10_sse2;
|
||||
h->pred8x8l[HOR_PRED ] = ff_pred8x8l_horizontal_10_sse2;
|
||||
h->pred8x8l[DC_PRED ] = ff_pred8x8l_dc_10_sse2;
|
||||
h->pred8x8l[DC_128_PRED ] = ff_pred8x8l_128_dc_10_sse2;
|
||||
h->pred8x8l[TOP_DC_PRED ] = ff_pred8x8l_top_dc_10_sse2;
|
||||
h->pred8x8l[DIAG_DOWN_LEFT_PRED ] = ff_pred8x8l_down_left_10_sse2;
|
||||
h->pred8x8l[DIAG_DOWN_RIGHT_PRED] = ff_pred8x8l_down_right_10_sse2;
|
||||
h->pred8x8l[VERT_RIGHT_PRED ] = ff_pred8x8l_vertical_right_10_sse2;
|
||||
h->pred8x8l[HOR_UP_PRED ] = ff_pred8x8l_horizontal_up_10_sse2;
|
||||
|
||||
h->pred16x16[DC_PRED8x8 ] = ff_pred16x16_dc_10_sse2;
|
||||
h->pred16x16[TOP_DC_PRED8x8 ] = ff_pred16x16_top_dc_10_sse2;
|
||||
h->pred16x16[DC_128_PRED8x8 ] = ff_pred16x16_128_dc_10_sse2;
|
||||
h->pred16x16[LEFT_DC_PRED8x8 ] = ff_pred16x16_left_dc_10_sse2;
|
||||
h->pred16x16[VERT_PRED8x8 ] = ff_pred16x16_vertical_10_sse2;
|
||||
h->pred16x16[HOR_PRED8x8 ] = ff_pred16x16_horizontal_10_sse2;
|
||||
}
|
||||
if (EXTERNAL_SSSE3(cpu_flags)) {
|
||||
h->pred4x4[DIAG_DOWN_RIGHT_PRED] = ff_pred4x4_down_right_10_ssse3;
|
||||
h->pred4x4[VERT_RIGHT_PRED ] = ff_pred4x4_vertical_right_10_ssse3;
|
||||
h->pred4x4[HOR_DOWN_PRED ] = ff_pred4x4_horizontal_down_10_ssse3;
|
||||
|
||||
h->pred8x8l[HOR_PRED ] = ff_pred8x8l_horizontal_10_ssse3;
|
||||
h->pred8x8l[DIAG_DOWN_LEFT_PRED ] = ff_pred8x8l_down_left_10_ssse3;
|
||||
h->pred8x8l[DIAG_DOWN_RIGHT_PRED] = ff_pred8x8l_down_right_10_ssse3;
|
||||
h->pred8x8l[VERT_RIGHT_PRED ] = ff_pred8x8l_vertical_right_10_ssse3;
|
||||
h->pred8x8l[HOR_UP_PRED ] = ff_pred8x8l_horizontal_up_10_ssse3;
|
||||
}
|
||||
if (EXTERNAL_AVX(cpu_flags)) {
|
||||
h->pred4x4[DIAG_DOWN_LEFT_PRED ] = ff_pred4x4_down_left_10_avx;
|
||||
h->pred4x4[DIAG_DOWN_RIGHT_PRED] = ff_pred4x4_down_right_10_avx;
|
||||
h->pred4x4[VERT_LEFT_PRED ] = ff_pred4x4_vertical_left_10_avx;
|
||||
h->pred4x4[VERT_RIGHT_PRED ] = ff_pred4x4_vertical_right_10_avx;
|
||||
h->pred4x4[HOR_DOWN_PRED ] = ff_pred4x4_horizontal_down_10_avx;
|
||||
|
||||
h->pred8x8l[VERT_PRED ] = ff_pred8x8l_vertical_10_avx;
|
||||
h->pred8x8l[HOR_PRED ] = ff_pred8x8l_horizontal_10_avx;
|
||||
h->pred8x8l[DC_PRED ] = ff_pred8x8l_dc_10_avx;
|
||||
h->pred8x8l[TOP_DC_PRED ] = ff_pred8x8l_top_dc_10_avx;
|
||||
h->pred8x8l[DIAG_DOWN_RIGHT_PRED] = ff_pred8x8l_down_right_10_avx;
|
||||
h->pred8x8l[DIAG_DOWN_LEFT_PRED ] = ff_pred8x8l_down_left_10_avx;
|
||||
h->pred8x8l[VERT_RIGHT_PRED ] = ff_pred8x8l_vertical_right_10_avx;
|
||||
h->pred8x8l[HOR_UP_PRED ] = ff_pred8x8l_horizontal_up_10_avx;
|
||||
}
|
||||
}
|
||||
}
|
||||
133
media/ffvpx/libavcodec/x86/mathops.h
Normal file
133
media/ffvpx/libavcodec/x86/mathops.h
Normal file
|
|
@ -0,0 +1,133 @@
|
|||
/*
|
||||
* simple math operations
|
||||
* Copyright (c) 2006 Michael Niedermayer <michaelni@gmx.at> et al
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#ifndef AVCODEC_X86_MATHOPS_H
|
||||
#define AVCODEC_X86_MATHOPS_H
|
||||
|
||||
#include "config.h"
|
||||
|
||||
#include "libavutil/common.h"
|
||||
#include "libavutil/x86/asm.h"
|
||||
|
||||
#if HAVE_INLINE_ASM
|
||||
|
||||
#if ARCH_X86_32
|
||||
|
||||
#define MULL MULL
|
||||
static av_always_inline av_const int MULL(int a, int b, unsigned shift)
|
||||
{
|
||||
int rt, dummy;
|
||||
__asm__ (
|
||||
"imull %3 \n\t"
|
||||
"shrdl %4, %%edx, %%eax \n\t"
|
||||
:"=a"(rt), "=d"(dummy)
|
||||
:"a"(a), "rm"(b), "ci"((uint8_t)shift)
|
||||
);
|
||||
return rt;
|
||||
}
|
||||
|
||||
#define MULH MULH
|
||||
static av_always_inline av_const int MULH(int a, int b)
|
||||
{
|
||||
int rt, dummy;
|
||||
__asm__ (
|
||||
"imull %3"
|
||||
:"=d"(rt), "=a"(dummy)
|
||||
:"a"(a), "rm"(b)
|
||||
);
|
||||
return rt;
|
||||
}
|
||||
|
||||
#define MUL64 MUL64
|
||||
static av_always_inline av_const int64_t MUL64(int a, int b)
|
||||
{
|
||||
int64_t rt;
|
||||
__asm__ (
|
||||
"imull %2"
|
||||
:"=A"(rt)
|
||||
:"a"(a), "rm"(b)
|
||||
);
|
||||
return rt;
|
||||
}
|
||||
|
||||
#endif /* ARCH_X86_32 */
|
||||
|
||||
#if HAVE_I686
|
||||
/* median of 3 */
|
||||
#define mid_pred mid_pred
|
||||
static inline av_const int mid_pred(int a, int b, int c)
|
||||
{
|
||||
int i=b;
|
||||
__asm__ (
|
||||
"cmp %2, %1 \n\t"
|
||||
"cmovg %1, %0 \n\t"
|
||||
"cmovg %2, %1 \n\t"
|
||||
"cmp %3, %1 \n\t"
|
||||
"cmovl %3, %1 \n\t"
|
||||
"cmp %1, %0 \n\t"
|
||||
"cmovg %1, %0 \n\t"
|
||||
:"+&r"(i), "+&r"(a)
|
||||
:"r"(b), "r"(c)
|
||||
);
|
||||
return i;
|
||||
}
|
||||
|
||||
#if HAVE_6REGS
|
||||
#define COPY3_IF_LT(x, y, a, b, c, d)\
|
||||
__asm__ volatile(\
|
||||
"cmpl %0, %3 \n\t"\
|
||||
"cmovl %3, %0 \n\t"\
|
||||
"cmovl %4, %1 \n\t"\
|
||||
"cmovl %5, %2 \n\t"\
|
||||
: "+&r" (x), "+&r" (a), "+r" (c)\
|
||||
: "r" (y), "r" (b), "r" (d)\
|
||||
);
|
||||
#endif /* HAVE_6REGS */
|
||||
|
||||
#endif /* HAVE_I686 */
|
||||
|
||||
#define MASK_ABS(mask, level) \
|
||||
__asm__ ("cdq \n\t" \
|
||||
"xorl %1, %0 \n\t" \
|
||||
"subl %1, %0 \n\t" \
|
||||
: "+a"(level), "=&d"(mask))
|
||||
|
||||
// avoid +32 for shift optimization (gcc should do that ...)
|
||||
#define NEG_SSR32 NEG_SSR32
|
||||
static inline int32_t NEG_SSR32( int32_t a, int8_t s){
|
||||
__asm__ ("sarl %1, %0\n\t"
|
||||
: "+r" (a)
|
||||
: "ic" ((uint8_t)(-s))
|
||||
);
|
||||
return a;
|
||||
}
|
||||
|
||||
#define NEG_USR32 NEG_USR32
|
||||
static inline uint32_t NEG_USR32(uint32_t a, int8_t s){
|
||||
__asm__ ("shrl %1, %0\n\t"
|
||||
: "+r" (a)
|
||||
: "ic" ((uint8_t)(-s))
|
||||
);
|
||||
return a;
|
||||
}
|
||||
|
||||
#endif /* HAVE_INLINE_ASM */
|
||||
#endif /* AVCODEC_X86_MATHOPS_H */
|
||||
35
media/ffvpx/libavcodec/x86/moz.build
Normal file
35
media/ffvpx/libavcodec/x86/moz.build
Normal file
|
|
@ -0,0 +1,35 @@
|
|||
# -*- Mode: python; indent-tabs-mode: nil; tab-width: 40 -*-
|
||||
# vim: set filetype=python:
|
||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
SOURCES += [
|
||||
'constants.c',
|
||||
'flacdsp.asm',
|
||||
'flacdsp_init.c',
|
||||
'h264_intrapred.asm',
|
||||
'h264_intrapred_10bit.asm',
|
||||
'h264_intrapred_init.c',
|
||||
'videodsp.asm',
|
||||
'videodsp_init.c',
|
||||
'vp8dsp.asm',
|
||||
'vp8dsp_init.c',
|
||||
'vp8dsp_loopfilter.asm',
|
||||
'vp9dsp_init.c',
|
||||
'vp9dsp_init_10bpp.c',
|
||||
'vp9dsp_init_12bpp.c',
|
||||
'vp9dsp_init_16bpp.c',
|
||||
'vp9intrapred.asm',
|
||||
'vp9intrapred_16bpp.asm',
|
||||
'vp9itxfm.asm',
|
||||
'vp9itxfm_16bpp.asm',
|
||||
'vp9lpf.asm',
|
||||
'vp9lpf_16bpp.asm',
|
||||
'vp9mc.asm',
|
||||
'vp9mc_16bpp.asm',
|
||||
]
|
||||
|
||||
FINAL_LIBRARY = 'mozavcodec'
|
||||
|
||||
include('/media/ffvpx/ffvpxcommon.mozbuild')
|
||||
468
media/ffvpx/libavcodec/x86/videodsp.asm
Normal file
468
media/ffvpx/libavcodec/x86/videodsp.asm
Normal file
|
|
@ -0,0 +1,468 @@
|
|||
;******************************************************************************
|
||||
;* Core video DSP functions
|
||||
;* Copyright (c) 2012 Ronald S. Bultje <rsbultje@gmail.com>
|
||||
;*
|
||||
;* This file is part of FFmpeg.
|
||||
;*
|
||||
;* FFmpeg is free software; you can redistribute it and/or
|
||||
;* modify it under the terms of the GNU Lesser General Public
|
||||
;* License as published by the Free Software Foundation; either
|
||||
;* version 2.1 of the License, or (at your option) any later version.
|
||||
;*
|
||||
;* FFmpeg is distributed in the hope that it will be useful,
|
||||
;* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
;* Lesser General Public License for more details.
|
||||
;*
|
||||
;* You should have received a copy of the GNU Lesser General Public
|
||||
;* License along with FFmpeg; if not, write to the Free Software
|
||||
;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
;******************************************************************************
|
||||
|
||||
%include "libavutil/x86/x86util.asm"
|
||||
|
||||
SECTION .text
|
||||
|
||||
; slow vertical extension loop function. Works with variable-width, and
|
||||
; does per-line reading/writing of source data
|
||||
|
||||
%macro V_COPY_ROW 2 ; type (top/body/bottom), h
|
||||
.%1_y_loop: ; do {
|
||||
mov wq, r7mp ; initialize w (r7mp = wmp)
|
||||
.%1_x_loop: ; do {
|
||||
movu m0, [srcq+wq] ; m0 = read($mmsize)
|
||||
movu [dstq+wq], m0 ; write(m0, $mmsize)
|
||||
add wq, mmsize ; w -= $mmsize
|
||||
cmp wq, -mmsize ; } while (w > $mmsize);
|
||||
jl .%1_x_loop
|
||||
movu m0, [srcq-mmsize] ; m0 = read($mmsize)
|
||||
movu [dstq-mmsize], m0 ; write(m0, $mmsize)
|
||||
%ifidn %1, body ; if ($type == body) {
|
||||
add srcq, src_strideq ; src += src_stride
|
||||
%endif ; }
|
||||
add dstq, dst_strideq ; dst += dst_stride
|
||||
dec %2 ; } while (--$h);
|
||||
jnz .%1_y_loop
|
||||
%endmacro
|
||||
|
||||
%macro vvar_fn 0
|
||||
; .----. <- zero
|
||||
; | | <- top is copied from first line in body of source
|
||||
; |----| <- start_y
|
||||
; | | <- body is copied verbatim (line-by-line) from source
|
||||
; |----| <- end_y
|
||||
; | | <- bottom is copied from last line in body of source
|
||||
; '----' <- bh
|
||||
%if ARCH_X86_64
|
||||
cglobal emu_edge_vvar, 7, 8, 1, dst, dst_stride, src, src_stride, \
|
||||
start_y, end_y, bh, w
|
||||
%else ; x86-32
|
||||
cglobal emu_edge_vvar, 1, 6, 1, dst, src, start_y, end_y, bh, w
|
||||
%define src_strideq r3mp
|
||||
%define dst_strideq r1mp
|
||||
mov srcq, r2mp
|
||||
mov start_yq, r4mp
|
||||
mov end_yq, r5mp
|
||||
mov bhq, r6mp
|
||||
%endif
|
||||
sub bhq, end_yq ; bh -= end_q
|
||||
sub end_yq, start_yq ; end_q -= start_q
|
||||
add srcq, r7mp ; (r7mp = wmp)
|
||||
add dstq, r7mp ; (r7mp = wmp)
|
||||
neg r7mp ; (r7mp = wmp)
|
||||
test start_yq, start_yq ; if (start_q) {
|
||||
jz .body
|
||||
V_COPY_ROW top, start_yq ; v_copy_row(top, start_yq)
|
||||
.body: ; }
|
||||
V_COPY_ROW body, end_yq ; v_copy_row(body, end_yq)
|
||||
test bhq, bhq ; if (bh) {
|
||||
jz .end
|
||||
sub srcq, src_strideq ; src -= src_stride
|
||||
V_COPY_ROW bottom, bhq ; v_copy_row(bottom, bh)
|
||||
.end: ; }
|
||||
RET
|
||||
%endmacro
|
||||
|
||||
%if ARCH_X86_32
|
||||
INIT_MMX mmx
|
||||
vvar_fn
|
||||
%endif
|
||||
|
||||
INIT_XMM sse
|
||||
vvar_fn
|
||||
|
||||
%macro hvar_fn 0
|
||||
cglobal emu_edge_hvar, 5, 6, 1, dst, dst_stride, start_x, n_words, h, w
|
||||
lea dstq, [dstq+n_wordsq*2]
|
||||
neg n_wordsq
|
||||
lea start_xq, [start_xq+n_wordsq*2]
|
||||
.y_loop: ; do {
|
||||
%if cpuflag(avx2)
|
||||
vpbroadcastb m0, [dstq+start_xq]
|
||||
mov wq, n_wordsq ; initialize w
|
||||
%else
|
||||
movzx wd, byte [dstq+start_xq] ; w = read(1)
|
||||
imul wd, 0x01010101 ; w *= 0x01010101
|
||||
movd m0, wd
|
||||
mov wq, n_wordsq ; initialize w
|
||||
%if cpuflag(sse2)
|
||||
pshufd m0, m0, q0000 ; splat
|
||||
%else ; mmx
|
||||
punpckldq m0, m0 ; splat
|
||||
%endif ; mmx/sse
|
||||
%endif ; avx2
|
||||
.x_loop: ; do {
|
||||
movu [dstq+wq*2], m0 ; write($reg, $mmsize)
|
||||
add wq, mmsize/2 ; w -= $mmsize/2
|
||||
cmp wq, -mmsize/2 ; } while (w > $mmsize/2)
|
||||
jl .x_loop
|
||||
movu [dstq-mmsize], m0 ; write($reg, $mmsize)
|
||||
add dstq, dst_strideq ; dst += dst_stride
|
||||
dec hq ; } while (h--)
|
||||
jnz .y_loop
|
||||
RET
|
||||
%endmacro
|
||||
|
||||
%if ARCH_X86_32
|
||||
INIT_MMX mmx
|
||||
hvar_fn
|
||||
%endif
|
||||
|
||||
INIT_XMM sse2
|
||||
hvar_fn
|
||||
|
||||
%if HAVE_AVX2_EXTERNAL
|
||||
INIT_XMM avx2
|
||||
hvar_fn
|
||||
%endif
|
||||
|
||||
; macro to read/write a horizontal number of pixels (%2) to/from registers
|
||||
; on sse, - fills xmm0-15 for consecutive sets of 16 pixels
|
||||
; - if (%2 & 8) fills 8 bytes into xmm$next
|
||||
; - if (%2 & 4) fills 4 bytes into xmm$next
|
||||
; - if (%2 & 3) fills 1, 2 or 4 bytes in eax
|
||||
; on mmx, - fills mm0-7 for consecutive sets of 8 pixels
|
||||
; - if (%2 & 4) fills 4 bytes into mm$next
|
||||
; - if (%2 & 3) fills 1, 2 or 4 bytes in eax
|
||||
; writing data out is in the same way
|
||||
%macro READ_NUM_BYTES 2
|
||||
%assign %%off 0 ; offset in source buffer
|
||||
%assign %%mmx_idx 0 ; mmx register index
|
||||
%assign %%xmm_idx 0 ; xmm register index
|
||||
|
||||
%rep %2/mmsize
|
||||
%if mmsize == 16
|
||||
movu xmm %+ %%xmm_idx, [srcq+%%off]
|
||||
%assign %%xmm_idx %%xmm_idx+1
|
||||
%else ; mmx
|
||||
movu mm %+ %%mmx_idx, [srcq+%%off]
|
||||
%assign %%mmx_idx %%mmx_idx+1
|
||||
%endif
|
||||
%assign %%off %%off+mmsize
|
||||
%endrep ; %2/mmsize
|
||||
|
||||
%if mmsize == 16
|
||||
%if (%2-%%off) >= 8
|
||||
%if %2 > 16 && (%2-%%off) > 8
|
||||
movu xmm %+ %%xmm_idx, [srcq+%2-16]
|
||||
%assign %%xmm_idx %%xmm_idx+1
|
||||
%assign %%off %2
|
||||
%else
|
||||
movq mm %+ %%mmx_idx, [srcq+%%off]
|
||||
%assign %%mmx_idx %%mmx_idx+1
|
||||
%assign %%off %%off+8
|
||||
%endif
|
||||
%endif ; (%2-%%off) >= 8
|
||||
%endif
|
||||
|
||||
%if (%2-%%off) >= 4
|
||||
%if %2 > 8 && (%2-%%off) > 4
|
||||
movq mm %+ %%mmx_idx, [srcq+%2-8]
|
||||
%assign %%off %2
|
||||
%else
|
||||
movd mm %+ %%mmx_idx, [srcq+%%off]
|
||||
%assign %%off %%off+4
|
||||
%endif
|
||||
%assign %%mmx_idx %%mmx_idx+1
|
||||
%endif ; (%2-%%off) >= 4
|
||||
|
||||
%if (%2-%%off) >= 1
|
||||
%if %2 >= 4
|
||||
movd mm %+ %%mmx_idx, [srcq+%2-4]
|
||||
%elif (%2-%%off) == 1
|
||||
mov valb, [srcq+%2-1]
|
||||
%elif (%2-%%off) == 2
|
||||
mov valw, [srcq+%2-2]
|
||||
%else
|
||||
mov valb, [srcq+%2-1]
|
||||
ror vald, 16
|
||||
mov valw, [srcq+%2-3]
|
||||
%endif
|
||||
%endif ; (%2-%%off) >= 1
|
||||
%endmacro ; READ_NUM_BYTES
|
||||
|
||||
%macro WRITE_NUM_BYTES 2
|
||||
%assign %%off 0 ; offset in destination buffer
|
||||
%assign %%mmx_idx 0 ; mmx register index
|
||||
%assign %%xmm_idx 0 ; xmm register index
|
||||
|
||||
%rep %2/mmsize
|
||||
%if mmsize == 16
|
||||
movu [dstq+%%off], xmm %+ %%xmm_idx
|
||||
%assign %%xmm_idx %%xmm_idx+1
|
||||
%else ; mmx
|
||||
movu [dstq+%%off], mm %+ %%mmx_idx
|
||||
%assign %%mmx_idx %%mmx_idx+1
|
||||
%endif
|
||||
%assign %%off %%off+mmsize
|
||||
%endrep ; %2/mmsize
|
||||
|
||||
%if mmsize == 16
|
||||
%if (%2-%%off) >= 8
|
||||
%if %2 > 16 && (%2-%%off) > 8
|
||||
movu [dstq+%2-16], xmm %+ %%xmm_idx
|
||||
%assign %%xmm_idx %%xmm_idx+1
|
||||
%assign %%off %2
|
||||
%else
|
||||
movq [dstq+%%off], mm %+ %%mmx_idx
|
||||
%assign %%mmx_idx %%mmx_idx+1
|
||||
%assign %%off %%off+8
|
||||
%endif
|
||||
%endif ; (%2-%%off) >= 8
|
||||
%endif
|
||||
|
||||
%if (%2-%%off) >= 4
|
||||
%if %2 > 8 && (%2-%%off) > 4
|
||||
movq [dstq+%2-8], mm %+ %%mmx_idx
|
||||
%assign %%off %2
|
||||
%else
|
||||
movd [dstq+%%off], mm %+ %%mmx_idx
|
||||
%assign %%off %%off+4
|
||||
%endif
|
||||
%assign %%mmx_idx %%mmx_idx+1
|
||||
%endif ; (%2-%%off) >= 4
|
||||
|
||||
%if (%2-%%off) >= 1
|
||||
%if %2 >= 4
|
||||
movd [dstq+%2-4], mm %+ %%mmx_idx
|
||||
%elif (%2-%%off) == 1
|
||||
mov [dstq+%2-1], valb
|
||||
%elif (%2-%%off) == 2
|
||||
mov [dstq+%2-2], valw
|
||||
%else
|
||||
mov [dstq+%2-3], valw
|
||||
ror vald, 16
|
||||
mov [dstq+%2-1], valb
|
||||
%ifnidn %1, body
|
||||
ror vald, 16
|
||||
%endif
|
||||
%endif
|
||||
%endif ; (%2-%%off) >= 1
|
||||
%endmacro ; WRITE_NUM_BYTES
|
||||
|
||||
; vertical top/bottom extend and body copy fast loops
|
||||
; these are function pointers to set-width line copy functions, i.e.
|
||||
; they read a fixed number of pixels into set registers, and write
|
||||
; those out into the destination buffer
|
||||
%macro VERTICAL_EXTEND 2
|
||||
%assign %%n %1
|
||||
%rep 1+%2-%1
|
||||
%if %%n <= 3
|
||||
%if ARCH_X86_64
|
||||
cglobal emu_edge_vfix %+ %%n, 6, 8, 0, dst, dst_stride, src, src_stride, \
|
||||
start_y, end_y, val, bh
|
||||
mov bhq, r6mp ; r6mp = bhmp
|
||||
%else ; x86-32
|
||||
cglobal emu_edge_vfix %+ %%n, 0, 6, 0, val, dst, src, start_y, end_y, bh
|
||||
mov dstq, r0mp
|
||||
mov srcq, r2mp
|
||||
mov start_yq, r4mp
|
||||
mov end_yq, r5mp
|
||||
mov bhq, r6mp
|
||||
%define dst_strideq r1mp
|
||||
%define src_strideq r3mp
|
||||
%endif ; x86-64/32
|
||||
%else
|
||||
%if ARCH_X86_64
|
||||
cglobal emu_edge_vfix %+ %%n, 7, 7, 1, dst, dst_stride, src, src_stride, \
|
||||
start_y, end_y, bh
|
||||
%else ; x86-32
|
||||
cglobal emu_edge_vfix %+ %%n, 1, 5, 1, dst, src, start_y, end_y, bh
|
||||
mov srcq, r2mp
|
||||
mov start_yq, r4mp
|
||||
mov end_yq, r5mp
|
||||
mov bhq, r6mp
|
||||
%define dst_strideq r1mp
|
||||
%define src_strideq r3mp
|
||||
%endif ; x86-64/32
|
||||
%endif
|
||||
; FIXME move this to c wrapper?
|
||||
sub bhq, end_yq ; bh -= end_y
|
||||
sub end_yq, start_yq ; end_y -= start_y
|
||||
|
||||
; extend pixels above body
|
||||
test start_yq, start_yq ; if (start_y) {
|
||||
jz .body_loop
|
||||
READ_NUM_BYTES top, %%n ; $variable_regs = read($n)
|
||||
.top_loop: ; do {
|
||||
WRITE_NUM_BYTES top, %%n ; write($variable_regs, $n)
|
||||
add dstq, dst_strideq ; dst += linesize
|
||||
dec start_yq ; } while (--start_y)
|
||||
jnz .top_loop ; }
|
||||
|
||||
; copy body pixels
|
||||
.body_loop: ; do {
|
||||
READ_NUM_BYTES body, %%n ; $variable_regs = read($n)
|
||||
WRITE_NUM_BYTES body, %%n ; write($variable_regs, $n)
|
||||
add dstq, dst_strideq ; dst += dst_stride
|
||||
add srcq, src_strideq ; src += src_stride
|
||||
dec end_yq ; } while (--end_y)
|
||||
jnz .body_loop
|
||||
|
||||
; copy bottom pixels
|
||||
test bhq, bhq ; if (block_h) {
|
||||
jz .end
|
||||
sub srcq, src_strideq ; src -= linesize
|
||||
READ_NUM_BYTES bottom, %%n ; $variable_regs = read($n)
|
||||
.bottom_loop: ; do {
|
||||
WRITE_NUM_BYTES bottom, %%n ; write($variable_regs, $n)
|
||||
add dstq, dst_strideq ; dst += linesize
|
||||
dec bhq ; } while (--bh)
|
||||
jnz .bottom_loop ; }
|
||||
|
||||
.end:
|
||||
RET
|
||||
%assign %%n %%n+1
|
||||
%endrep ; 1+%2-%1
|
||||
%endmacro ; VERTICAL_EXTEND
|
||||
|
||||
INIT_MMX mmx
|
||||
VERTICAL_EXTEND 1, 15
|
||||
%if ARCH_X86_32
|
||||
VERTICAL_EXTEND 16, 22
|
||||
%endif
|
||||
|
||||
INIT_XMM sse
|
||||
VERTICAL_EXTEND 16, 22
|
||||
|
||||
; left/right (horizontal) fast extend functions
|
||||
; these are essentially identical to the vertical extend ones above,
|
||||
; just left/right separated because number of pixels to extend is
|
||||
; obviously not the same on both sides.
|
||||
|
||||
%macro READ_V_PIXEL 2
|
||||
%if cpuflag(avx2)
|
||||
vpbroadcastb m0, %2
|
||||
%else
|
||||
movzx vald, byte %2
|
||||
imul vald, 0x01010101
|
||||
%if %1 >= 8
|
||||
movd m0, vald
|
||||
%if mmsize == 16
|
||||
pshufd m0, m0, q0000
|
||||
%else
|
||||
punpckldq m0, m0
|
||||
%endif ; mmsize == 16
|
||||
%endif ; %1 > 16
|
||||
%endif ; avx2
|
||||
%endmacro ; READ_V_PIXEL
|
||||
|
||||
%macro WRITE_V_PIXEL 2
|
||||
%assign %%off 0
|
||||
|
||||
%if %1 >= 8
|
||||
|
||||
%rep %1/mmsize
|
||||
movu [%2+%%off], m0
|
||||
%assign %%off %%off+mmsize
|
||||
%endrep ; %1/mmsize
|
||||
|
||||
%if mmsize == 16
|
||||
%if %1-%%off >= 8
|
||||
%if %1 > 16 && %1-%%off > 8
|
||||
movu [%2+%1-16], m0
|
||||
%assign %%off %1
|
||||
%else
|
||||
movq [%2+%%off], m0
|
||||
%assign %%off %%off+8
|
||||
%endif
|
||||
%endif ; %1-%%off >= 8
|
||||
%endif ; mmsize == 16
|
||||
|
||||
%if %1-%%off >= 4
|
||||
%if %1 > 8 && %1-%%off > 4
|
||||
movq [%2+%1-8], m0
|
||||
%assign %%off %1
|
||||
%else
|
||||
movd [%2+%%off], m0
|
||||
%assign %%off %%off+4
|
||||
%endif
|
||||
%endif ; %1-%%off >= 4
|
||||
|
||||
%else ; %1 < 8
|
||||
|
||||
%rep %1/4
|
||||
mov [%2+%%off], vald
|
||||
%assign %%off %%off+4
|
||||
%endrep ; %1/4
|
||||
|
||||
%endif ; %1 >=/< 8
|
||||
|
||||
%if %1-%%off == 2
|
||||
%if cpuflag(avx2)
|
||||
movd [%2+%%off-2], m0
|
||||
%else
|
||||
mov [%2+%%off], valw
|
||||
%endif ; avx2
|
||||
%endif ; (%1-%%off)/2
|
||||
%endmacro ; WRITE_V_PIXEL
|
||||
|
||||
%macro H_EXTEND 2
|
||||
%assign %%n %1
|
||||
%rep 1+(%2-%1)/2
|
||||
%if cpuflag(avx2)
|
||||
cglobal emu_edge_hfix %+ %%n, 4, 4, 1, dst, dst_stride, start_x, bh
|
||||
%else
|
||||
cglobal emu_edge_hfix %+ %%n, 4, 5, 1, dst, dst_stride, start_x, bh, val
|
||||
%endif
|
||||
.loop_y: ; do {
|
||||
READ_V_PIXEL %%n, [dstq+start_xq] ; $variable_regs = read($n)
|
||||
WRITE_V_PIXEL %%n, dstq ; write($variable_regs, $n)
|
||||
add dstq, dst_strideq ; dst += dst_stride
|
||||
dec bhq ; } while (--bh)
|
||||
jnz .loop_y
|
||||
RET
|
||||
%assign %%n %%n+2
|
||||
%endrep ; 1+(%2-%1)/2
|
||||
%endmacro ; H_EXTEND
|
||||
|
||||
INIT_MMX mmx
|
||||
H_EXTEND 2, 14
|
||||
%if ARCH_X86_32
|
||||
H_EXTEND 16, 22
|
||||
%endif
|
||||
|
||||
INIT_XMM sse2
|
||||
H_EXTEND 16, 22
|
||||
|
||||
%if HAVE_AVX2_EXTERNAL
|
||||
INIT_XMM avx2
|
||||
H_EXTEND 8, 22
|
||||
%endif
|
||||
|
||||
%macro PREFETCH_FN 1
|
||||
cglobal prefetch, 3, 3, 0, buf, stride, h
|
||||
.loop:
|
||||
%1 [bufq]
|
||||
add bufq, strideq
|
||||
dec hd
|
||||
jg .loop
|
||||
REP_RET
|
||||
%endmacro
|
||||
|
||||
INIT_MMX mmxext
|
||||
PREFETCH_FN prefetcht0
|
||||
%if ARCH_X86_32
|
||||
INIT_MMX 3dnow
|
||||
PREFETCH_FN prefetch
|
||||
%endif
|
||||
309
media/ffvpx/libavcodec/x86/videodsp_init.c
Normal file
309
media/ffvpx/libavcodec/x86/videodsp_init.c
Normal file
|
|
@ -0,0 +1,309 @@
|
|||
/*
|
||||
* Copyright (C) 2002-2012 Michael Niedermayer
|
||||
* Copyright (C) 2012 Ronald S. Bultje
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#include "config.h"
|
||||
#include "libavutil/attributes.h"
|
||||
#include "libavutil/avassert.h"
|
||||
#include "libavutil/common.h"
|
||||
#include "libavutil/cpu.h"
|
||||
#include "libavutil/mem.h"
|
||||
#include "libavutil/x86/asm.h"
|
||||
#include "libavutil/x86/cpu.h"
|
||||
#include "libavcodec/videodsp.h"
|
||||
|
||||
#if HAVE_YASM
|
||||
typedef void emu_edge_vfix_func(uint8_t *dst, x86_reg dst_stride,
|
||||
const uint8_t *src, x86_reg src_stride,
|
||||
x86_reg start_y, x86_reg end_y, x86_reg bh);
|
||||
typedef void emu_edge_vvar_func(uint8_t *dst, x86_reg dst_stride,
|
||||
const uint8_t *src, x86_reg src_stride,
|
||||
x86_reg start_y, x86_reg end_y, x86_reg bh,
|
||||
x86_reg w);
|
||||
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix1_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix2_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix3_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix4_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix5_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix6_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix7_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix8_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix9_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix10_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix11_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix12_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix13_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix14_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix15_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix16_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix17_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix18_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix19_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix20_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix21_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix22_mmx;
|
||||
#if ARCH_X86_32
|
||||
static emu_edge_vfix_func * const vfixtbl_mmx[22] = {
|
||||
&ff_emu_edge_vfix1_mmx, &ff_emu_edge_vfix2_mmx, &ff_emu_edge_vfix3_mmx,
|
||||
&ff_emu_edge_vfix4_mmx, &ff_emu_edge_vfix5_mmx, &ff_emu_edge_vfix6_mmx,
|
||||
&ff_emu_edge_vfix7_mmx, &ff_emu_edge_vfix8_mmx, &ff_emu_edge_vfix9_mmx,
|
||||
&ff_emu_edge_vfix10_mmx, &ff_emu_edge_vfix11_mmx, &ff_emu_edge_vfix12_mmx,
|
||||
&ff_emu_edge_vfix13_mmx, &ff_emu_edge_vfix14_mmx, &ff_emu_edge_vfix15_mmx,
|
||||
&ff_emu_edge_vfix16_mmx, &ff_emu_edge_vfix17_mmx, &ff_emu_edge_vfix18_mmx,
|
||||
&ff_emu_edge_vfix19_mmx, &ff_emu_edge_vfix20_mmx, &ff_emu_edge_vfix21_mmx,
|
||||
&ff_emu_edge_vfix22_mmx
|
||||
};
|
||||
#endif
|
||||
extern emu_edge_vvar_func ff_emu_edge_vvar_mmx;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix16_sse;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix17_sse;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix18_sse;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix19_sse;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix20_sse;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix21_sse;
|
||||
extern emu_edge_vfix_func ff_emu_edge_vfix22_sse;
|
||||
static emu_edge_vfix_func * const vfixtbl_sse[22] = {
|
||||
ff_emu_edge_vfix1_mmx, ff_emu_edge_vfix2_mmx, ff_emu_edge_vfix3_mmx,
|
||||
ff_emu_edge_vfix4_mmx, ff_emu_edge_vfix5_mmx, ff_emu_edge_vfix6_mmx,
|
||||
ff_emu_edge_vfix7_mmx, ff_emu_edge_vfix8_mmx, ff_emu_edge_vfix9_mmx,
|
||||
ff_emu_edge_vfix10_mmx, ff_emu_edge_vfix11_mmx, ff_emu_edge_vfix12_mmx,
|
||||
ff_emu_edge_vfix13_mmx, ff_emu_edge_vfix14_mmx, ff_emu_edge_vfix15_mmx,
|
||||
ff_emu_edge_vfix16_sse, ff_emu_edge_vfix17_sse, ff_emu_edge_vfix18_sse,
|
||||
ff_emu_edge_vfix19_sse, ff_emu_edge_vfix20_sse, ff_emu_edge_vfix21_sse,
|
||||
ff_emu_edge_vfix22_sse
|
||||
};
|
||||
extern emu_edge_vvar_func ff_emu_edge_vvar_sse;
|
||||
|
||||
typedef void emu_edge_hfix_func(uint8_t *dst, x86_reg dst_stride,
|
||||
x86_reg start_x, x86_reg bh);
|
||||
typedef void emu_edge_hvar_func(uint8_t *dst, x86_reg dst_stride,
|
||||
x86_reg start_x, x86_reg n_words, x86_reg bh);
|
||||
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix2_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix4_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix6_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix8_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix10_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix12_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix14_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix16_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix18_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix20_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix22_mmx;
|
||||
#if ARCH_X86_32
|
||||
static emu_edge_hfix_func * const hfixtbl_mmx[11] = {
|
||||
ff_emu_edge_hfix2_mmx, ff_emu_edge_hfix4_mmx, ff_emu_edge_hfix6_mmx,
|
||||
ff_emu_edge_hfix8_mmx, ff_emu_edge_hfix10_mmx, ff_emu_edge_hfix12_mmx,
|
||||
ff_emu_edge_hfix14_mmx, ff_emu_edge_hfix16_mmx, ff_emu_edge_hfix18_mmx,
|
||||
ff_emu_edge_hfix20_mmx, ff_emu_edge_hfix22_mmx
|
||||
};
|
||||
#endif
|
||||
extern emu_edge_hvar_func ff_emu_edge_hvar_mmx;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix16_sse2;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix18_sse2;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix20_sse2;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix22_sse2;
|
||||
static emu_edge_hfix_func * const hfixtbl_sse2[11] = {
|
||||
ff_emu_edge_hfix2_mmx, ff_emu_edge_hfix4_mmx, ff_emu_edge_hfix6_mmx,
|
||||
ff_emu_edge_hfix8_mmx, ff_emu_edge_hfix10_mmx, ff_emu_edge_hfix12_mmx,
|
||||
ff_emu_edge_hfix14_mmx, ff_emu_edge_hfix16_sse2, ff_emu_edge_hfix18_sse2,
|
||||
ff_emu_edge_hfix20_sse2, ff_emu_edge_hfix22_sse2
|
||||
};
|
||||
extern emu_edge_hvar_func ff_emu_edge_hvar_sse2;
|
||||
#if HAVE_AVX2_EXTERNAL
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix8_avx2;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix10_avx2;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix12_avx2;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix14_avx2;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix16_avx2;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix18_avx2;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix20_avx2;
|
||||
extern emu_edge_hfix_func ff_emu_edge_hfix22_avx2;
|
||||
static emu_edge_hfix_func * const hfixtbl_avx2[11] = {
|
||||
ff_emu_edge_hfix2_mmx, ff_emu_edge_hfix4_mmx, ff_emu_edge_hfix6_mmx,
|
||||
ff_emu_edge_hfix8_avx2, ff_emu_edge_hfix10_avx2, ff_emu_edge_hfix12_avx2,
|
||||
ff_emu_edge_hfix14_avx2, ff_emu_edge_hfix16_avx2, ff_emu_edge_hfix18_avx2,
|
||||
ff_emu_edge_hfix20_avx2, ff_emu_edge_hfix22_avx2
|
||||
};
|
||||
extern emu_edge_hvar_func ff_emu_edge_hvar_avx2;
|
||||
#endif
|
||||
|
||||
static av_always_inline void emulated_edge_mc(uint8_t *dst, const uint8_t *src,
|
||||
ptrdiff_t dst_stride,
|
||||
ptrdiff_t src_stride,
|
||||
x86_reg block_w, x86_reg block_h,
|
||||
x86_reg src_x, x86_reg src_y,
|
||||
x86_reg w, x86_reg h,
|
||||
emu_edge_vfix_func * const *vfix_tbl,
|
||||
emu_edge_vvar_func *v_extend_var,
|
||||
emu_edge_hfix_func * const *hfix_tbl,
|
||||
emu_edge_hvar_func *h_extend_var)
|
||||
{
|
||||
x86_reg start_y, start_x, end_y, end_x, src_y_add = 0, p;
|
||||
|
||||
if (!w || !h)
|
||||
return;
|
||||
|
||||
av_assert2(block_w <= FFABS(dst_stride));
|
||||
|
||||
if (src_y >= h) {
|
||||
src -= src_y*src_stride;
|
||||
src_y_add = h - 1;
|
||||
src_y = h - 1;
|
||||
} else if (src_y <= -block_h) {
|
||||
src -= src_y*src_stride;
|
||||
src_y_add = 1 - block_h;
|
||||
src_y = 1 - block_h;
|
||||
}
|
||||
if (src_x >= w) {
|
||||
src += w - 1 - src_x;
|
||||
src_x = w - 1;
|
||||
} else if (src_x <= -block_w) {
|
||||
src += 1 - block_w - src_x;
|
||||
src_x = 1 - block_w;
|
||||
}
|
||||
|
||||
start_y = FFMAX(0, -src_y);
|
||||
start_x = FFMAX(0, -src_x);
|
||||
end_y = FFMIN(block_h, h-src_y);
|
||||
end_x = FFMIN(block_w, w-src_x);
|
||||
av_assert2(start_x < end_x && block_w > 0);
|
||||
av_assert2(start_y < end_y && block_h > 0);
|
||||
|
||||
// fill in the to-be-copied part plus all above/below
|
||||
src += (src_y_add + start_y) * src_stride + start_x;
|
||||
w = end_x - start_x;
|
||||
if (w <= 22) {
|
||||
vfix_tbl[w - 1](dst + start_x, dst_stride, src, src_stride,
|
||||
start_y, end_y, block_h);
|
||||
} else {
|
||||
v_extend_var(dst + start_x, dst_stride, src, src_stride,
|
||||
start_y, end_y, block_h, w);
|
||||
}
|
||||
|
||||
// fill left
|
||||
if (start_x) {
|
||||
if (start_x <= 22) {
|
||||
hfix_tbl[(start_x - 1) >> 1](dst, dst_stride, start_x, block_h);
|
||||
} else {
|
||||
h_extend_var(dst, dst_stride,
|
||||
start_x, (start_x + 1) >> 1, block_h);
|
||||
}
|
||||
}
|
||||
|
||||
// fill right
|
||||
p = block_w - end_x;
|
||||
if (p) {
|
||||
if (p <= 22) {
|
||||
hfix_tbl[(p - 1) >> 1](dst + end_x - (p & 1), dst_stride,
|
||||
-!(p & 1), block_h);
|
||||
} else {
|
||||
h_extend_var(dst + end_x - (p & 1), dst_stride,
|
||||
-!(p & 1), (p + 1) >> 1, block_h);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#if ARCH_X86_32
|
||||
static av_noinline void emulated_edge_mc_mmx(uint8_t *buf, const uint8_t *src,
|
||||
ptrdiff_t buf_stride,
|
||||
ptrdiff_t src_stride,
|
||||
int block_w, int block_h,
|
||||
int src_x, int src_y, int w, int h)
|
||||
{
|
||||
emulated_edge_mc(buf, src, buf_stride, src_stride, block_w, block_h,
|
||||
src_x, src_y, w, h, vfixtbl_mmx, &ff_emu_edge_vvar_mmx,
|
||||
hfixtbl_mmx, &ff_emu_edge_hvar_mmx);
|
||||
}
|
||||
|
||||
static av_noinline void emulated_edge_mc_sse(uint8_t *buf, const uint8_t *src,
|
||||
ptrdiff_t buf_stride,
|
||||
ptrdiff_t src_stride,
|
||||
int block_w, int block_h,
|
||||
int src_x, int src_y, int w, int h)
|
||||
{
|
||||
emulated_edge_mc(buf, src, buf_stride, src_stride, block_w, block_h,
|
||||
src_x, src_y, w, h, vfixtbl_sse, &ff_emu_edge_vvar_sse,
|
||||
hfixtbl_mmx, &ff_emu_edge_hvar_mmx);
|
||||
}
|
||||
#endif
|
||||
|
||||
static av_noinline void emulated_edge_mc_sse2(uint8_t *buf, const uint8_t *src,
|
||||
ptrdiff_t buf_stride,
|
||||
ptrdiff_t src_stride,
|
||||
int block_w, int block_h,
|
||||
int src_x, int src_y, int w,
|
||||
int h)
|
||||
{
|
||||
emulated_edge_mc(buf, src, buf_stride, src_stride, block_w, block_h,
|
||||
src_x, src_y, w, h, vfixtbl_sse, &ff_emu_edge_vvar_sse,
|
||||
hfixtbl_sse2, &ff_emu_edge_hvar_sse2);
|
||||
}
|
||||
|
||||
#if HAVE_AVX2_EXTERNAL
|
||||
static av_noinline void emulated_edge_mc_avx2(uint8_t *buf, const uint8_t *src,
|
||||
ptrdiff_t buf_stride,
|
||||
ptrdiff_t src_stride,
|
||||
int block_w, int block_h,
|
||||
int src_x, int src_y, int w,
|
||||
int h)
|
||||
{
|
||||
emulated_edge_mc(buf, src, buf_stride, src_stride, block_w, block_h,
|
||||
src_x, src_y, w, h, vfixtbl_sse, &ff_emu_edge_vvar_sse,
|
||||
hfixtbl_avx2, &ff_emu_edge_hvar_avx2);
|
||||
}
|
||||
#endif /* HAVE_AVX2_EXTERNAL */
|
||||
#endif /* HAVE_YASM */
|
||||
|
||||
void ff_prefetch_mmxext(uint8_t *buf, ptrdiff_t stride, int h);
|
||||
void ff_prefetch_3dnow(uint8_t *buf, ptrdiff_t stride, int h);
|
||||
|
||||
av_cold void ff_videodsp_init_x86(VideoDSPContext *ctx, int bpc)
|
||||
{
|
||||
#if HAVE_YASM
|
||||
int cpu_flags = av_get_cpu_flags();
|
||||
|
||||
#if ARCH_X86_32
|
||||
if (EXTERNAL_MMX(cpu_flags) && bpc <= 8) {
|
||||
ctx->emulated_edge_mc = emulated_edge_mc_mmx;
|
||||
}
|
||||
if (EXTERNAL_AMD3DNOW(cpu_flags)) {
|
||||
ctx->prefetch = ff_prefetch_3dnow;
|
||||
}
|
||||
#endif /* ARCH_X86_32 */
|
||||
if (EXTERNAL_MMXEXT(cpu_flags)) {
|
||||
ctx->prefetch = ff_prefetch_mmxext;
|
||||
}
|
||||
#if ARCH_X86_32
|
||||
if (EXTERNAL_SSE(cpu_flags) && bpc <= 8) {
|
||||
ctx->emulated_edge_mc = emulated_edge_mc_sse;
|
||||
}
|
||||
#endif /* ARCH_X86_32 */
|
||||
if (EXTERNAL_SSE2(cpu_flags) && bpc <= 8) {
|
||||
ctx->emulated_edge_mc = emulated_edge_mc_sse2;
|
||||
}
|
||||
#if HAVE_AVX2_EXTERNAL
|
||||
if (EXTERNAL_AVX2(cpu_flags) && bpc <= 8) {
|
||||
ctx->emulated_edge_mc = emulated_edge_mc_avx2;
|
||||
}
|
||||
#endif
|
||||
#endif /* HAVE_YASM */
|
||||
}
|
||||
51
media/ffvpx/libavcodec/x86/vp56_arith.h
Normal file
51
media/ffvpx/libavcodec/x86/vp56_arith.h
Normal file
|
|
@ -0,0 +1,51 @@
|
|||
/**
|
||||
* VP5 and VP6 compatible video decoder (arith decoder)
|
||||
*
|
||||
* Copyright (C) 2006 Aurelien Jacobs <aurel@gnuage.org>
|
||||
* Copyright (C) 2010 Eli Friedman
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#ifndef AVCODEC_X86_VP56_ARITH_H
|
||||
#define AVCODEC_X86_VP56_ARITH_H
|
||||
|
||||
#if HAVE_INLINE_ASM && HAVE_FAST_CMOV && HAVE_6REGS
|
||||
#define vp56_rac_get_prob vp56_rac_get_prob
|
||||
static av_always_inline int vp56_rac_get_prob(VP56RangeCoder *c, uint8_t prob)
|
||||
{
|
||||
unsigned int code_word = vp56_rac_renorm(c);
|
||||
unsigned int low = 1 + (((c->high - 1) * prob) >> 8);
|
||||
unsigned int low_shift = low << 16;
|
||||
int bit = 0;
|
||||
c->code_word = code_word;
|
||||
|
||||
__asm__(
|
||||
"subl %4, %1 \n\t"
|
||||
"subl %3, %2 \n\t"
|
||||
"setae %b0 \n\t"
|
||||
"cmovb %4, %1 \n\t"
|
||||
"cmovb %5, %2 \n\t"
|
||||
: "+q"(bit), "+&r"(c->high), "+&r"(c->code_word)
|
||||
: "r"(low_shift), "r"(low), "r"(code_word)
|
||||
);
|
||||
|
||||
return bit;
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif /* AVCODEC_X86_VP56_ARITH_H */
|
||||
1225
media/ffvpx/libavcodec/x86/vp8dsp.asm
Normal file
1225
media/ffvpx/libavcodec/x86/vp8dsp.asm
Normal file
File diff suppressed because it is too large
Load diff
464
media/ffvpx/libavcodec/x86/vp8dsp_init.c
Normal file
464
media/ffvpx/libavcodec/x86/vp8dsp_init.c
Normal file
|
|
@ -0,0 +1,464 @@
|
|||
/*
|
||||
* VP8 DSP functions x86-optimized
|
||||
* Copyright (c) 2010 Ronald S. Bultje <rsbultje@gmail.com>
|
||||
* Copyright (c) 2010 Fiona Glaser <fiona@x264.com>
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#include "libavutil/attributes.h"
|
||||
#include "libavutil/cpu.h"
|
||||
#include "libavutil/mem.h"
|
||||
#include "libavutil/x86/cpu.h"
|
||||
#include "libavcodec/vp8dsp.h"
|
||||
|
||||
#if HAVE_YASM
|
||||
|
||||
/*
|
||||
* MC functions
|
||||
*/
|
||||
void ff_put_vp8_epel4_h4_mmxext(uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel4_h6_mmxext(uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel4_v4_mmxext(uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel4_v6_mmxext(uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
|
||||
void ff_put_vp8_epel8_h4_sse2 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel8_h6_sse2 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel8_v4_sse2 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel8_v6_sse2 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
|
||||
void ff_put_vp8_epel4_h4_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel4_h6_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel4_v4_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel4_v6_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel8_h4_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel8_h6_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel8_v4_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_epel8_v6_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
|
||||
void ff_put_vp8_bilinear4_h_mmxext(uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_bilinear8_h_sse2 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_bilinear4_h_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_bilinear8_h_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
|
||||
void ff_put_vp8_bilinear4_v_mmxext(uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_bilinear8_v_sse2 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_bilinear4_v_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_bilinear8_v_ssse3 (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
|
||||
|
||||
void ff_put_vp8_pixels8_mmx (uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_pixels16_mmx(uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
void ff_put_vp8_pixels16_sse(uint8_t *dst, ptrdiff_t dststride,
|
||||
uint8_t *src, ptrdiff_t srcstride,
|
||||
int height, int mx, int my);
|
||||
|
||||
#define TAP_W16(OPT, FILTERTYPE, TAPTYPE) \
|
||||
static void ff_put_vp8_ ## FILTERTYPE ## 16_ ## TAPTYPE ## _ ## OPT( \
|
||||
uint8_t *dst, ptrdiff_t dststride, uint8_t *src, \
|
||||
ptrdiff_t srcstride, int height, int mx, int my) \
|
||||
{ \
|
||||
ff_put_vp8_ ## FILTERTYPE ## 8_ ## TAPTYPE ## _ ## OPT( \
|
||||
dst, dststride, src, srcstride, height, mx, my); \
|
||||
ff_put_vp8_ ## FILTERTYPE ## 8_ ## TAPTYPE ## _ ## OPT( \
|
||||
dst + 8, dststride, src + 8, srcstride, height, mx, my); \
|
||||
}
|
||||
#define TAP_W8(OPT, FILTERTYPE, TAPTYPE) \
|
||||
static void ff_put_vp8_ ## FILTERTYPE ## 8_ ## TAPTYPE ## _ ## OPT( \
|
||||
uint8_t *dst, ptrdiff_t dststride, uint8_t *src, \
|
||||
ptrdiff_t srcstride, int height, int mx, int my) \
|
||||
{ \
|
||||
ff_put_vp8_ ## FILTERTYPE ## 4_ ## TAPTYPE ## _ ## OPT( \
|
||||
dst, dststride, src, srcstride, height, mx, my); \
|
||||
ff_put_vp8_ ## FILTERTYPE ## 4_ ## TAPTYPE ## _ ## OPT( \
|
||||
dst + 4, dststride, src + 4, srcstride, height, mx, my); \
|
||||
}
|
||||
|
||||
#if ARCH_X86_32
|
||||
TAP_W8 (mmxext, epel, h4)
|
||||
TAP_W8 (mmxext, epel, h6)
|
||||
TAP_W16(mmxext, epel, h6)
|
||||
TAP_W8 (mmxext, epel, v4)
|
||||
TAP_W8 (mmxext, epel, v6)
|
||||
TAP_W16(mmxext, epel, v6)
|
||||
TAP_W8 (mmxext, bilinear, h)
|
||||
TAP_W16(mmxext, bilinear, h)
|
||||
TAP_W8 (mmxext, bilinear, v)
|
||||
TAP_W16(mmxext, bilinear, v)
|
||||
#endif
|
||||
|
||||
TAP_W16(sse2, epel, h6)
|
||||
TAP_W16(sse2, epel, v6)
|
||||
TAP_W16(sse2, bilinear, h)
|
||||
TAP_W16(sse2, bilinear, v)
|
||||
|
||||
TAP_W16(ssse3, epel, h6)
|
||||
TAP_W16(ssse3, epel, v6)
|
||||
TAP_W16(ssse3, bilinear, h)
|
||||
TAP_W16(ssse3, bilinear, v)
|
||||
|
||||
#define HVTAP(OPT, ALIGN, TAPNUMX, TAPNUMY, SIZE, MAXHEIGHT) \
|
||||
static void ff_put_vp8_epel ## SIZE ## _h ## TAPNUMX ## v ## TAPNUMY ## _ ## OPT( \
|
||||
uint8_t *dst, ptrdiff_t dststride, uint8_t *src, \
|
||||
ptrdiff_t srcstride, int height, int mx, int my) \
|
||||
{ \
|
||||
LOCAL_ALIGNED(ALIGN, uint8_t, tmp, [SIZE * (MAXHEIGHT + TAPNUMY - 1)]); \
|
||||
uint8_t *tmpptr = tmp + SIZE * (TAPNUMY / 2 - 1); \
|
||||
src -= srcstride * (TAPNUMY / 2 - 1); \
|
||||
ff_put_vp8_epel ## SIZE ## _h ## TAPNUMX ## _ ## OPT( \
|
||||
tmp, SIZE, src, srcstride, height + TAPNUMY - 1, mx, my); \
|
||||
ff_put_vp8_epel ## SIZE ## _v ## TAPNUMY ## _ ## OPT( \
|
||||
dst, dststride, tmpptr, SIZE, height, mx, my); \
|
||||
}
|
||||
|
||||
#if ARCH_X86_32
|
||||
#define HVTAPMMX(x, y) \
|
||||
HVTAP(mmxext, 8, x, y, 4, 8) \
|
||||
HVTAP(mmxext, 8, x, y, 8, 16)
|
||||
|
||||
HVTAP(mmxext, 8, 6, 6, 16, 16)
|
||||
#else
|
||||
#define HVTAPMMX(x, y) \
|
||||
HVTAP(mmxext, 8, x, y, 4, 8)
|
||||
#endif
|
||||
|
||||
HVTAPMMX(4, 4)
|
||||
HVTAPMMX(4, 6)
|
||||
HVTAPMMX(6, 4)
|
||||
HVTAPMMX(6, 6)
|
||||
|
||||
#define HVTAPSSE2(x, y, w) \
|
||||
HVTAP(sse2, 16, x, y, w, 16) \
|
||||
HVTAP(ssse3, 16, x, y, w, 16)
|
||||
|
||||
HVTAPSSE2(4, 4, 8)
|
||||
HVTAPSSE2(4, 6, 8)
|
||||
HVTAPSSE2(6, 4, 8)
|
||||
HVTAPSSE2(6, 6, 8)
|
||||
HVTAPSSE2(6, 6, 16)
|
||||
|
||||
HVTAP(ssse3, 16, 4, 4, 4, 8)
|
||||
HVTAP(ssse3, 16, 4, 6, 4, 8)
|
||||
HVTAP(ssse3, 16, 6, 4, 4, 8)
|
||||
HVTAP(ssse3, 16, 6, 6, 4, 8)
|
||||
|
||||
#define HVBILIN(OPT, ALIGN, SIZE, MAXHEIGHT) \
|
||||
static void ff_put_vp8_bilinear ## SIZE ## _hv_ ## OPT( \
|
||||
uint8_t *dst, ptrdiff_t dststride, uint8_t *src, \
|
||||
ptrdiff_t srcstride, int height, int mx, int my) \
|
||||
{ \
|
||||
LOCAL_ALIGNED(ALIGN, uint8_t, tmp, [SIZE * (MAXHEIGHT + 2)]); \
|
||||
ff_put_vp8_bilinear ## SIZE ## _h_ ## OPT( \
|
||||
tmp, SIZE, src, srcstride, height + 1, mx, my); \
|
||||
ff_put_vp8_bilinear ## SIZE ## _v_ ## OPT( \
|
||||
dst, dststride, tmp, SIZE, height, mx, my); \
|
||||
}
|
||||
|
||||
HVBILIN(mmxext, 8, 4, 8)
|
||||
#if ARCH_X86_32
|
||||
HVBILIN(mmxext, 8, 8, 16)
|
||||
HVBILIN(mmxext, 8, 16, 16)
|
||||
#endif
|
||||
HVBILIN(sse2, 8, 8, 16)
|
||||
HVBILIN(sse2, 8, 16, 16)
|
||||
HVBILIN(ssse3, 8, 4, 8)
|
||||
HVBILIN(ssse3, 8, 8, 16)
|
||||
HVBILIN(ssse3, 8, 16, 16)
|
||||
|
||||
void ff_vp8_idct_dc_add_mmx(uint8_t *dst, int16_t block[16],
|
||||
ptrdiff_t stride);
|
||||
void ff_vp8_idct_dc_add_sse4(uint8_t *dst, int16_t block[16],
|
||||
ptrdiff_t stride);
|
||||
void ff_vp8_idct_dc_add4y_mmx(uint8_t *dst, int16_t block[4][16],
|
||||
ptrdiff_t stride);
|
||||
void ff_vp8_idct_dc_add4y_sse2(uint8_t *dst, int16_t block[4][16],
|
||||
ptrdiff_t stride);
|
||||
void ff_vp8_idct_dc_add4uv_mmx(uint8_t *dst, int16_t block[2][16],
|
||||
ptrdiff_t stride);
|
||||
void ff_vp8_luma_dc_wht_mmx(int16_t block[4][4][16], int16_t dc[16]);
|
||||
void ff_vp8_luma_dc_wht_sse(int16_t block[4][4][16], int16_t dc[16]);
|
||||
void ff_vp8_idct_add_mmx(uint8_t *dst, int16_t block[16], ptrdiff_t stride);
|
||||
void ff_vp8_idct_add_sse(uint8_t *dst, int16_t block[16], ptrdiff_t stride);
|
||||
|
||||
#define DECLARE_LOOP_FILTER(NAME) \
|
||||
void ff_vp8_v_loop_filter_simple_ ## NAME(uint8_t *dst, \
|
||||
ptrdiff_t stride, \
|
||||
int flim); \
|
||||
void ff_vp8_h_loop_filter_simple_ ## NAME(uint8_t *dst, \
|
||||
ptrdiff_t stride, \
|
||||
int flim); \
|
||||
void ff_vp8_v_loop_filter16y_inner_ ## NAME (uint8_t *dst, \
|
||||
ptrdiff_t stride, \
|
||||
int e, int i, int hvt); \
|
||||
void ff_vp8_h_loop_filter16y_inner_ ## NAME (uint8_t *dst, \
|
||||
ptrdiff_t stride, \
|
||||
int e, int i, int hvt); \
|
||||
void ff_vp8_v_loop_filter8uv_inner_ ## NAME (uint8_t *dstU, \
|
||||
uint8_t *dstV, \
|
||||
ptrdiff_t s, \
|
||||
int e, int i, int hvt); \
|
||||
void ff_vp8_h_loop_filter8uv_inner_ ## NAME (uint8_t *dstU, \
|
||||
uint8_t *dstV, \
|
||||
ptrdiff_t s, \
|
||||
int e, int i, int hvt); \
|
||||
void ff_vp8_v_loop_filter16y_mbedge_ ## NAME(uint8_t *dst, \
|
||||
ptrdiff_t stride, \
|
||||
int e, int i, int hvt); \
|
||||
void ff_vp8_h_loop_filter16y_mbedge_ ## NAME(uint8_t *dst, \
|
||||
ptrdiff_t stride, \
|
||||
int e, int i, int hvt); \
|
||||
void ff_vp8_v_loop_filter8uv_mbedge_ ## NAME(uint8_t *dstU, \
|
||||
uint8_t *dstV, \
|
||||
ptrdiff_t s, \
|
||||
int e, int i, int hvt); \
|
||||
void ff_vp8_h_loop_filter8uv_mbedge_ ## NAME(uint8_t *dstU, \
|
||||
uint8_t *dstV, \
|
||||
ptrdiff_t s, \
|
||||
int e, int i, int hvt);
|
||||
|
||||
DECLARE_LOOP_FILTER(mmx)
|
||||
DECLARE_LOOP_FILTER(mmxext)
|
||||
DECLARE_LOOP_FILTER(sse2)
|
||||
DECLARE_LOOP_FILTER(ssse3)
|
||||
DECLARE_LOOP_FILTER(sse4)
|
||||
|
||||
#endif /* HAVE_YASM */
|
||||
|
||||
#define VP8_LUMA_MC_FUNC(IDX, SIZE, OPT) \
|
||||
c->put_vp8_epel_pixels_tab[IDX][0][2] = ff_put_vp8_epel ## SIZE ## _h6_ ## OPT; \
|
||||
c->put_vp8_epel_pixels_tab[IDX][2][0] = ff_put_vp8_epel ## SIZE ## _v6_ ## OPT; \
|
||||
c->put_vp8_epel_pixels_tab[IDX][2][2] = ff_put_vp8_epel ## SIZE ## _h6v6_ ## OPT
|
||||
|
||||
#define VP8_MC_FUNC(IDX, SIZE, OPT) \
|
||||
c->put_vp8_epel_pixels_tab[IDX][0][1] = ff_put_vp8_epel ## SIZE ## _h4_ ## OPT; \
|
||||
c->put_vp8_epel_pixels_tab[IDX][1][0] = ff_put_vp8_epel ## SIZE ## _v4_ ## OPT; \
|
||||
c->put_vp8_epel_pixels_tab[IDX][1][1] = ff_put_vp8_epel ## SIZE ## _h4v4_ ## OPT; \
|
||||
c->put_vp8_epel_pixels_tab[IDX][1][2] = ff_put_vp8_epel ## SIZE ## _h6v4_ ## OPT; \
|
||||
c->put_vp8_epel_pixels_tab[IDX][2][1] = ff_put_vp8_epel ## SIZE ## _h4v6_ ## OPT; \
|
||||
VP8_LUMA_MC_FUNC(IDX, SIZE, OPT)
|
||||
|
||||
#define VP8_BILINEAR_MC_FUNC(IDX, SIZE, OPT) \
|
||||
c->put_vp8_bilinear_pixels_tab[IDX][0][1] = ff_put_vp8_bilinear ## SIZE ## _h_ ## OPT; \
|
||||
c->put_vp8_bilinear_pixels_tab[IDX][0][2] = ff_put_vp8_bilinear ## SIZE ## _h_ ## OPT; \
|
||||
c->put_vp8_bilinear_pixels_tab[IDX][1][0] = ff_put_vp8_bilinear ## SIZE ## _v_ ## OPT; \
|
||||
c->put_vp8_bilinear_pixels_tab[IDX][1][1] = ff_put_vp8_bilinear ## SIZE ## _hv_ ## OPT; \
|
||||
c->put_vp8_bilinear_pixels_tab[IDX][1][2] = ff_put_vp8_bilinear ## SIZE ## _hv_ ## OPT; \
|
||||
c->put_vp8_bilinear_pixels_tab[IDX][2][0] = ff_put_vp8_bilinear ## SIZE ## _v_ ## OPT; \
|
||||
c->put_vp8_bilinear_pixels_tab[IDX][2][1] = ff_put_vp8_bilinear ## SIZE ## _hv_ ## OPT; \
|
||||
c->put_vp8_bilinear_pixels_tab[IDX][2][2] = ff_put_vp8_bilinear ## SIZE ## _hv_ ## OPT
|
||||
|
||||
|
||||
av_cold void ff_vp78dsp_init_x86(VP8DSPContext *c)
|
||||
{
|
||||
#if HAVE_YASM
|
||||
int cpu_flags = av_get_cpu_flags();
|
||||
|
||||
if (EXTERNAL_MMX(cpu_flags)) {
|
||||
#if ARCH_X86_32
|
||||
c->put_vp8_epel_pixels_tab[0][0][0] =
|
||||
c->put_vp8_bilinear_pixels_tab[0][0][0] = ff_put_vp8_pixels16_mmx;
|
||||
#endif
|
||||
c->put_vp8_epel_pixels_tab[1][0][0] =
|
||||
c->put_vp8_bilinear_pixels_tab[1][0][0] = ff_put_vp8_pixels8_mmx;
|
||||
}
|
||||
|
||||
/* note that 4-tap width=16 functions are missing because w=16
|
||||
* is only used for luma, and luma is always a copy or sixtap. */
|
||||
if (EXTERNAL_MMXEXT(cpu_flags)) {
|
||||
VP8_MC_FUNC(2, 4, mmxext);
|
||||
VP8_BILINEAR_MC_FUNC(2, 4, mmxext);
|
||||
#if ARCH_X86_32
|
||||
VP8_LUMA_MC_FUNC(0, 16, mmxext);
|
||||
VP8_MC_FUNC(1, 8, mmxext);
|
||||
VP8_BILINEAR_MC_FUNC(0, 16, mmxext);
|
||||
VP8_BILINEAR_MC_FUNC(1, 8, mmxext);
|
||||
#endif
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE(cpu_flags)) {
|
||||
c->put_vp8_epel_pixels_tab[0][0][0] =
|
||||
c->put_vp8_bilinear_pixels_tab[0][0][0] = ff_put_vp8_pixels16_sse;
|
||||
}
|
||||
|
||||
if (HAVE_SSE2_EXTERNAL && cpu_flags & (AV_CPU_FLAG_SSE2 | AV_CPU_FLAG_SSE2SLOW)) {
|
||||
VP8_LUMA_MC_FUNC(0, 16, sse2);
|
||||
VP8_MC_FUNC(1, 8, sse2);
|
||||
VP8_BILINEAR_MC_FUNC(0, 16, sse2);
|
||||
VP8_BILINEAR_MC_FUNC(1, 8, sse2);
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSSE3(cpu_flags)) {
|
||||
VP8_LUMA_MC_FUNC(0, 16, ssse3);
|
||||
VP8_MC_FUNC(1, 8, ssse3);
|
||||
VP8_MC_FUNC(2, 4, ssse3);
|
||||
VP8_BILINEAR_MC_FUNC(0, 16, ssse3);
|
||||
VP8_BILINEAR_MC_FUNC(1, 8, ssse3);
|
||||
VP8_BILINEAR_MC_FUNC(2, 4, ssse3);
|
||||
}
|
||||
#endif /* HAVE_YASM */
|
||||
}
|
||||
|
||||
av_cold void ff_vp8dsp_init_x86(VP8DSPContext *c)
|
||||
{
|
||||
#if HAVE_YASM
|
||||
int cpu_flags = av_get_cpu_flags();
|
||||
|
||||
if (EXTERNAL_MMX(cpu_flags)) {
|
||||
c->vp8_idct_dc_add = ff_vp8_idct_dc_add_mmx;
|
||||
c->vp8_idct_dc_add4uv = ff_vp8_idct_dc_add4uv_mmx;
|
||||
#if ARCH_X86_32
|
||||
c->vp8_idct_dc_add4y = ff_vp8_idct_dc_add4y_mmx;
|
||||
c->vp8_idct_add = ff_vp8_idct_add_mmx;
|
||||
c->vp8_luma_dc_wht = ff_vp8_luma_dc_wht_mmx;
|
||||
|
||||
c->vp8_v_loop_filter_simple = ff_vp8_v_loop_filter_simple_mmx;
|
||||
c->vp8_h_loop_filter_simple = ff_vp8_h_loop_filter_simple_mmx;
|
||||
|
||||
c->vp8_v_loop_filter16y_inner = ff_vp8_v_loop_filter16y_inner_mmx;
|
||||
c->vp8_h_loop_filter16y_inner = ff_vp8_h_loop_filter16y_inner_mmx;
|
||||
c->vp8_v_loop_filter8uv_inner = ff_vp8_v_loop_filter8uv_inner_mmx;
|
||||
c->vp8_h_loop_filter8uv_inner = ff_vp8_h_loop_filter8uv_inner_mmx;
|
||||
|
||||
c->vp8_v_loop_filter16y = ff_vp8_v_loop_filter16y_mbedge_mmx;
|
||||
c->vp8_h_loop_filter16y = ff_vp8_h_loop_filter16y_mbedge_mmx;
|
||||
c->vp8_v_loop_filter8uv = ff_vp8_v_loop_filter8uv_mbedge_mmx;
|
||||
c->vp8_h_loop_filter8uv = ff_vp8_h_loop_filter8uv_mbedge_mmx;
|
||||
#endif
|
||||
}
|
||||
|
||||
/* note that 4-tap width=16 functions are missing because w=16
|
||||
* is only used for luma, and luma is always a copy or sixtap. */
|
||||
if (EXTERNAL_MMXEXT(cpu_flags)) {
|
||||
#if ARCH_X86_32
|
||||
c->vp8_v_loop_filter_simple = ff_vp8_v_loop_filter_simple_mmxext;
|
||||
c->vp8_h_loop_filter_simple = ff_vp8_h_loop_filter_simple_mmxext;
|
||||
|
||||
c->vp8_v_loop_filter16y_inner = ff_vp8_v_loop_filter16y_inner_mmxext;
|
||||
c->vp8_h_loop_filter16y_inner = ff_vp8_h_loop_filter16y_inner_mmxext;
|
||||
c->vp8_v_loop_filter8uv_inner = ff_vp8_v_loop_filter8uv_inner_mmxext;
|
||||
c->vp8_h_loop_filter8uv_inner = ff_vp8_h_loop_filter8uv_inner_mmxext;
|
||||
|
||||
c->vp8_v_loop_filter16y = ff_vp8_v_loop_filter16y_mbedge_mmxext;
|
||||
c->vp8_h_loop_filter16y = ff_vp8_h_loop_filter16y_mbedge_mmxext;
|
||||
c->vp8_v_loop_filter8uv = ff_vp8_v_loop_filter8uv_mbedge_mmxext;
|
||||
c->vp8_h_loop_filter8uv = ff_vp8_h_loop_filter8uv_mbedge_mmxext;
|
||||
#endif
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE(cpu_flags)) {
|
||||
c->vp8_idct_add = ff_vp8_idct_add_sse;
|
||||
c->vp8_luma_dc_wht = ff_vp8_luma_dc_wht_sse;
|
||||
}
|
||||
|
||||
if (HAVE_SSE2_EXTERNAL && cpu_flags & (AV_CPU_FLAG_SSE2 | AV_CPU_FLAG_SSE2SLOW)) {
|
||||
c->vp8_v_loop_filter_simple = ff_vp8_v_loop_filter_simple_sse2;
|
||||
|
||||
c->vp8_v_loop_filter16y_inner = ff_vp8_v_loop_filter16y_inner_sse2;
|
||||
c->vp8_v_loop_filter8uv_inner = ff_vp8_v_loop_filter8uv_inner_sse2;
|
||||
|
||||
c->vp8_v_loop_filter16y = ff_vp8_v_loop_filter16y_mbedge_sse2;
|
||||
c->vp8_v_loop_filter8uv = ff_vp8_v_loop_filter8uv_mbedge_sse2;
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE2(cpu_flags)) {
|
||||
c->vp8_idct_dc_add4y = ff_vp8_idct_dc_add4y_sse2;
|
||||
|
||||
c->vp8_h_loop_filter_simple = ff_vp8_h_loop_filter_simple_sse2;
|
||||
|
||||
c->vp8_h_loop_filter16y_inner = ff_vp8_h_loop_filter16y_inner_sse2;
|
||||
c->vp8_h_loop_filter8uv_inner = ff_vp8_h_loop_filter8uv_inner_sse2;
|
||||
|
||||
c->vp8_h_loop_filter16y = ff_vp8_h_loop_filter16y_mbedge_sse2;
|
||||
c->vp8_h_loop_filter8uv = ff_vp8_h_loop_filter8uv_mbedge_sse2;
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSSE3(cpu_flags)) {
|
||||
c->vp8_v_loop_filter_simple = ff_vp8_v_loop_filter_simple_ssse3;
|
||||
c->vp8_h_loop_filter_simple = ff_vp8_h_loop_filter_simple_ssse3;
|
||||
|
||||
c->vp8_v_loop_filter16y_inner = ff_vp8_v_loop_filter16y_inner_ssse3;
|
||||
c->vp8_h_loop_filter16y_inner = ff_vp8_h_loop_filter16y_inner_ssse3;
|
||||
c->vp8_v_loop_filter8uv_inner = ff_vp8_v_loop_filter8uv_inner_ssse3;
|
||||
c->vp8_h_loop_filter8uv_inner = ff_vp8_h_loop_filter8uv_inner_ssse3;
|
||||
|
||||
c->vp8_v_loop_filter16y = ff_vp8_v_loop_filter16y_mbedge_ssse3;
|
||||
c->vp8_h_loop_filter16y = ff_vp8_h_loop_filter16y_mbedge_ssse3;
|
||||
c->vp8_v_loop_filter8uv = ff_vp8_v_loop_filter8uv_mbedge_ssse3;
|
||||
c->vp8_h_loop_filter8uv = ff_vp8_h_loop_filter8uv_mbedge_ssse3;
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE4(cpu_flags)) {
|
||||
c->vp8_idct_dc_add = ff_vp8_idct_dc_add_sse4;
|
||||
|
||||
c->vp8_h_loop_filter_simple = ff_vp8_h_loop_filter_simple_sse4;
|
||||
c->vp8_h_loop_filter16y = ff_vp8_h_loop_filter16y_mbedge_sse4;
|
||||
c->vp8_h_loop_filter8uv = ff_vp8_h_loop_filter8uv_mbedge_sse4;
|
||||
}
|
||||
#endif /* HAVE_YASM */
|
||||
}
|
||||
1584
media/ffvpx/libavcodec/x86/vp8dsp_loopfilter.asm
Normal file
1584
media/ffvpx/libavcodec/x86/vp8dsp_loopfilter.asm
Normal file
File diff suppressed because it is too large
Load diff
400
media/ffvpx/libavcodec/x86/vp9dsp_init.c
Normal file
400
media/ffvpx/libavcodec/x86/vp9dsp_init.c
Normal file
|
|
@ -0,0 +1,400 @@
|
|||
/*
|
||||
* VP9 SIMD optimizations
|
||||
*
|
||||
* Copyright (c) 2013 Ronald S. Bultje <rsbultje gmail com>
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#include "libavutil/attributes.h"
|
||||
#include "libavutil/cpu.h"
|
||||
#include "libavutil/mem.h"
|
||||
#include "libavutil/x86/cpu.h"
|
||||
#include "libavcodec/vp9dsp.h"
|
||||
#include "libavcodec/x86/vp9dsp_init.h"
|
||||
|
||||
#if HAVE_YASM
|
||||
|
||||
decl_fpel_func(put, 4, , mmx);
|
||||
decl_fpel_func(put, 8, , mmx);
|
||||
decl_fpel_func(put, 16, , sse);
|
||||
decl_fpel_func(put, 32, , sse);
|
||||
decl_fpel_func(put, 64, , sse);
|
||||
decl_fpel_func(avg, 4, _8, mmxext);
|
||||
decl_fpel_func(avg, 8, _8, mmxext);
|
||||
decl_fpel_func(avg, 16, _8, sse2);
|
||||
decl_fpel_func(avg, 32, _8, sse2);
|
||||
decl_fpel_func(avg, 64, _8, sse2);
|
||||
decl_fpel_func(put, 32, , avx);
|
||||
decl_fpel_func(put, 64, , avx);
|
||||
decl_fpel_func(avg, 32, _8, avx2);
|
||||
decl_fpel_func(avg, 64, _8, avx2);
|
||||
|
||||
decl_mc_funcs(4, mmxext, int16_t, 8, 8);
|
||||
decl_mc_funcs(8, sse2, int16_t, 8, 8);
|
||||
decl_mc_funcs(4, ssse3, int8_t, 32, 8);
|
||||
decl_mc_funcs(8, ssse3, int8_t, 32, 8);
|
||||
#if ARCH_X86_64
|
||||
decl_mc_funcs(16, ssse3, int8_t, 32, 8);
|
||||
decl_mc_funcs(32, avx2, int8_t, 32, 8);
|
||||
#endif
|
||||
|
||||
mc_rep_funcs(16, 8, 8, sse2, int16_t, 8, 8)
|
||||
#if ARCH_X86_32
|
||||
mc_rep_funcs(16, 8, 8, ssse3, int8_t, 32, 8)
|
||||
#endif
|
||||
mc_rep_funcs(32, 16, 16, sse2, int16_t, 8, 8)
|
||||
mc_rep_funcs(32, 16, 16, ssse3, int8_t, 32, 8)
|
||||
mc_rep_funcs(64, 32, 32, sse2, int16_t, 8, 8)
|
||||
mc_rep_funcs(64, 32, 32, ssse3, int8_t, 32, 8)
|
||||
#if ARCH_X86_64 && HAVE_AVX2_EXTERNAL
|
||||
mc_rep_funcs(64, 32, 32, avx2, int8_t, 32, 8)
|
||||
#endif
|
||||
|
||||
extern const int8_t ff_filters_ssse3[3][15][4][32];
|
||||
extern const int16_t ff_filters_sse2[3][15][8][8];
|
||||
|
||||
filters_8tap_2d_fn2(put, 16, 8, 1, mmxext, sse2, sse2)
|
||||
filters_8tap_2d_fn2(avg, 16, 8, 1, mmxext, sse2, sse2)
|
||||
filters_8tap_2d_fn2(put, 16, 8, 1, ssse3, ssse3, ssse3)
|
||||
filters_8tap_2d_fn2(avg, 16, 8, 1, ssse3, ssse3, ssse3)
|
||||
#if ARCH_X86_64 && HAVE_AVX2_EXTERNAL
|
||||
filters_8tap_2d_fn(put, 64, 32, 8, 1, avx2, ssse3)
|
||||
filters_8tap_2d_fn(put, 32, 32, 8, 1, avx2, ssse3)
|
||||
filters_8tap_2d_fn(avg, 64, 32, 8, 1, avx2, ssse3)
|
||||
filters_8tap_2d_fn(avg, 32, 32, 8, 1, avx2, ssse3)
|
||||
#endif
|
||||
|
||||
filters_8tap_1d_fn3(put, 8, mmxext, sse2, sse2)
|
||||
filters_8tap_1d_fn3(avg, 8, mmxext, sse2, sse2)
|
||||
filters_8tap_1d_fn3(put, 8, ssse3, ssse3, ssse3)
|
||||
filters_8tap_1d_fn3(avg, 8, ssse3, ssse3, ssse3)
|
||||
#if ARCH_X86_64 && HAVE_AVX2_EXTERNAL
|
||||
filters_8tap_1d_fn2(put, 64, 8, avx2, ssse3)
|
||||
filters_8tap_1d_fn2(put, 32, 8, avx2, ssse3)
|
||||
filters_8tap_1d_fn2(avg, 64, 8, avx2, ssse3)
|
||||
filters_8tap_1d_fn2(avg, 32, 8, avx2, ssse3)
|
||||
#endif
|
||||
|
||||
#define itxfm_func(typea, typeb, size, opt) \
|
||||
void ff_vp9_##typea##_##typeb##_##size##x##size##_add_##opt(uint8_t *dst, ptrdiff_t stride, \
|
||||
int16_t *block, int eob)
|
||||
#define itxfm_funcs(size, opt) \
|
||||
itxfm_func(idct, idct, size, opt); \
|
||||
itxfm_func(iadst, idct, size, opt); \
|
||||
itxfm_func(idct, iadst, size, opt); \
|
||||
itxfm_func(iadst, iadst, size, opt)
|
||||
|
||||
itxfm_func(idct, idct, 4, mmxext);
|
||||
itxfm_func(idct, iadst, 4, sse2);
|
||||
itxfm_func(iadst, idct, 4, sse2);
|
||||
itxfm_func(iadst, iadst, 4, sse2);
|
||||
itxfm_funcs(4, ssse3);
|
||||
itxfm_funcs(8, sse2);
|
||||
itxfm_funcs(8, ssse3);
|
||||
itxfm_funcs(8, avx);
|
||||
itxfm_funcs(16, sse2);
|
||||
itxfm_funcs(16, ssse3);
|
||||
itxfm_funcs(16, avx);
|
||||
itxfm_func(idct, idct, 32, sse2);
|
||||
itxfm_func(idct, idct, 32, ssse3);
|
||||
itxfm_func(idct, idct, 32, avx);
|
||||
itxfm_func(iwht, iwht, 4, mmx);
|
||||
|
||||
#undef itxfm_func
|
||||
#undef itxfm_funcs
|
||||
|
||||
#define lpf_funcs(size1, size2, opt) \
|
||||
void ff_vp9_loop_filter_v_##size1##_##size2##_##opt(uint8_t *dst, ptrdiff_t stride, \
|
||||
int E, int I, int H); \
|
||||
void ff_vp9_loop_filter_h_##size1##_##size2##_##opt(uint8_t *dst, ptrdiff_t stride, \
|
||||
int E, int I, int H)
|
||||
|
||||
lpf_funcs(16, 16, sse2);
|
||||
lpf_funcs(16, 16, ssse3);
|
||||
lpf_funcs(16, 16, avx);
|
||||
lpf_funcs(44, 16, sse2);
|
||||
lpf_funcs(44, 16, ssse3);
|
||||
lpf_funcs(44, 16, avx);
|
||||
lpf_funcs(84, 16, sse2);
|
||||
lpf_funcs(84, 16, ssse3);
|
||||
lpf_funcs(84, 16, avx);
|
||||
lpf_funcs(48, 16, sse2);
|
||||
lpf_funcs(48, 16, ssse3);
|
||||
lpf_funcs(48, 16, avx);
|
||||
lpf_funcs(88, 16, sse2);
|
||||
lpf_funcs(88, 16, ssse3);
|
||||
lpf_funcs(88, 16, avx);
|
||||
|
||||
#undef lpf_funcs
|
||||
|
||||
#define ipred_func(size, type, opt) \
|
||||
void ff_vp9_ipred_##type##_##size##x##size##_##opt(uint8_t *dst, ptrdiff_t stride, \
|
||||
const uint8_t *l, const uint8_t *a)
|
||||
|
||||
ipred_func(8, v, mmx);
|
||||
|
||||
#define ipred_dc_funcs(size, opt) \
|
||||
ipred_func(size, dc, opt); \
|
||||
ipred_func(size, dc_left, opt); \
|
||||
ipred_func(size, dc_top, opt)
|
||||
|
||||
ipred_dc_funcs(4, mmxext);
|
||||
ipred_dc_funcs(8, mmxext);
|
||||
|
||||
#define ipred_dir_tm_funcs(size, opt) \
|
||||
ipred_func(size, tm, opt); \
|
||||
ipred_func(size, dl, opt); \
|
||||
ipred_func(size, dr, opt); \
|
||||
ipred_func(size, hd, opt); \
|
||||
ipred_func(size, hu, opt); \
|
||||
ipred_func(size, vl, opt); \
|
||||
ipred_func(size, vr, opt)
|
||||
|
||||
ipred_dir_tm_funcs(4, mmxext);
|
||||
|
||||
ipred_func(16, v, sse);
|
||||
ipred_func(32, v, sse);
|
||||
|
||||
ipred_dc_funcs(16, sse2);
|
||||
ipred_dc_funcs(32, sse2);
|
||||
|
||||
#define ipred_dir_tm_h_funcs(size, opt) \
|
||||
ipred_dir_tm_funcs(size, opt); \
|
||||
ipred_func(size, h, opt)
|
||||
|
||||
ipred_dir_tm_h_funcs(8, sse2);
|
||||
ipred_dir_tm_h_funcs(16, sse2);
|
||||
ipred_dir_tm_h_funcs(32, sse2);
|
||||
|
||||
ipred_func(4, h, sse2);
|
||||
|
||||
#define ipred_all_funcs(size, opt) \
|
||||
ipred_dc_funcs(size, opt); \
|
||||
ipred_dir_tm_h_funcs(size, opt)
|
||||
|
||||
// FIXME hd/vl_4x4_ssse3 does not exist
|
||||
ipred_all_funcs(4, ssse3);
|
||||
ipred_all_funcs(8, ssse3);
|
||||
ipred_all_funcs(16, ssse3);
|
||||
ipred_all_funcs(32, ssse3);
|
||||
|
||||
ipred_dir_tm_h_funcs(8, avx);
|
||||
ipred_dir_tm_h_funcs(16, avx);
|
||||
ipred_dir_tm_h_funcs(32, avx);
|
||||
|
||||
ipred_func(32, v, avx);
|
||||
|
||||
ipred_dc_funcs(32, avx2);
|
||||
ipred_func(32, h, avx2);
|
||||
ipred_func(32, tm, avx2);
|
||||
|
||||
#undef ipred_func
|
||||
#undef ipred_dir_tm_h_funcs
|
||||
#undef ipred_dir_tm_funcs
|
||||
#undef ipred_dc_funcs
|
||||
|
||||
#endif /* HAVE_YASM */
|
||||
|
||||
av_cold void ff_vp9dsp_init_x86(VP9DSPContext *dsp, int bpp, int bitexact)
|
||||
{
|
||||
#if HAVE_YASM
|
||||
int cpu_flags;
|
||||
|
||||
if (bpp == 10) {
|
||||
ff_vp9dsp_init_10bpp_x86(dsp, bitexact);
|
||||
return;
|
||||
} else if (bpp == 12) {
|
||||
ff_vp9dsp_init_12bpp_x86(dsp, bitexact);
|
||||
return;
|
||||
}
|
||||
|
||||
cpu_flags = av_get_cpu_flags();
|
||||
|
||||
#define init_lpf(opt) do { \
|
||||
dsp->loop_filter_16[0] = ff_vp9_loop_filter_h_16_16_##opt; \
|
||||
dsp->loop_filter_16[1] = ff_vp9_loop_filter_v_16_16_##opt; \
|
||||
dsp->loop_filter_mix2[0][0][0] = ff_vp9_loop_filter_h_44_16_##opt; \
|
||||
dsp->loop_filter_mix2[0][0][1] = ff_vp9_loop_filter_v_44_16_##opt; \
|
||||
dsp->loop_filter_mix2[0][1][0] = ff_vp9_loop_filter_h_48_16_##opt; \
|
||||
dsp->loop_filter_mix2[0][1][1] = ff_vp9_loop_filter_v_48_16_##opt; \
|
||||
dsp->loop_filter_mix2[1][0][0] = ff_vp9_loop_filter_h_84_16_##opt; \
|
||||
dsp->loop_filter_mix2[1][0][1] = ff_vp9_loop_filter_v_84_16_##opt; \
|
||||
dsp->loop_filter_mix2[1][1][0] = ff_vp9_loop_filter_h_88_16_##opt; \
|
||||
dsp->loop_filter_mix2[1][1][1] = ff_vp9_loop_filter_v_88_16_##opt; \
|
||||
} while (0)
|
||||
|
||||
#define init_ipred(sz, opt, t, e) \
|
||||
dsp->intra_pred[TX_##sz##X##sz][e##_PRED] = ff_vp9_ipred_##t##_##sz##x##sz##_##opt
|
||||
|
||||
#define ff_vp9_ipred_hd_4x4_ssse3 ff_vp9_ipred_hd_4x4_mmxext
|
||||
#define ff_vp9_ipred_vl_4x4_ssse3 ff_vp9_ipred_vl_4x4_mmxext
|
||||
#define init_dir_tm_ipred(sz, opt) do { \
|
||||
init_ipred(sz, opt, dl, DIAG_DOWN_LEFT); \
|
||||
init_ipred(sz, opt, dr, DIAG_DOWN_RIGHT); \
|
||||
init_ipred(sz, opt, hd, HOR_DOWN); \
|
||||
init_ipred(sz, opt, vl, VERT_LEFT); \
|
||||
init_ipred(sz, opt, hu, HOR_UP); \
|
||||
init_ipred(sz, opt, tm, TM_VP8); \
|
||||
init_ipred(sz, opt, vr, VERT_RIGHT); \
|
||||
} while (0)
|
||||
#define init_dir_tm_h_ipred(sz, opt) do { \
|
||||
init_dir_tm_ipred(sz, opt); \
|
||||
init_ipred(sz, opt, h, HOR); \
|
||||
} while (0)
|
||||
#define init_dc_ipred(sz, opt) do { \
|
||||
init_ipred(sz, opt, dc, DC); \
|
||||
init_ipred(sz, opt, dc_left, LEFT_DC); \
|
||||
init_ipred(sz, opt, dc_top, TOP_DC); \
|
||||
} while (0)
|
||||
#define init_all_ipred(sz, opt) do { \
|
||||
init_dc_ipred(sz, opt); \
|
||||
init_dir_tm_h_ipred(sz, opt); \
|
||||
} while (0)
|
||||
|
||||
if (EXTERNAL_MMX(cpu_flags)) {
|
||||
init_fpel_func(4, 0, 4, put, , mmx);
|
||||
init_fpel_func(3, 0, 8, put, , mmx);
|
||||
if (!bitexact) {
|
||||
dsp->itxfm_add[4 /* lossless */][DCT_DCT] =
|
||||
dsp->itxfm_add[4 /* lossless */][ADST_DCT] =
|
||||
dsp->itxfm_add[4 /* lossless */][DCT_ADST] =
|
||||
dsp->itxfm_add[4 /* lossless */][ADST_ADST] = ff_vp9_iwht_iwht_4x4_add_mmx;
|
||||
}
|
||||
init_ipred(8, mmx, v, VERT);
|
||||
}
|
||||
|
||||
if (EXTERNAL_MMXEXT(cpu_flags)) {
|
||||
init_subpel2(4, 0, 4, put, 8, mmxext);
|
||||
init_subpel2(4, 1, 4, avg, 8, mmxext);
|
||||
init_fpel_func(4, 1, 4, avg, _8, mmxext);
|
||||
init_fpel_func(3, 1, 8, avg, _8, mmxext);
|
||||
dsp->itxfm_add[TX_4X4][DCT_DCT] = ff_vp9_idct_idct_4x4_add_mmxext;
|
||||
init_dc_ipred(4, mmxext);
|
||||
init_dc_ipred(8, mmxext);
|
||||
init_dir_tm_ipred(4, mmxext);
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE(cpu_flags)) {
|
||||
init_fpel_func(2, 0, 16, put, , sse);
|
||||
init_fpel_func(1, 0, 32, put, , sse);
|
||||
init_fpel_func(0, 0, 64, put, , sse);
|
||||
init_ipred(16, sse, v, VERT);
|
||||
init_ipred(32, sse, v, VERT);
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE2(cpu_flags)) {
|
||||
init_subpel3_8to64(0, put, 8, sse2);
|
||||
init_subpel3_8to64(1, avg, 8, sse2);
|
||||
init_fpel_func(2, 1, 16, avg, _8, sse2);
|
||||
init_fpel_func(1, 1, 32, avg, _8, sse2);
|
||||
init_fpel_func(0, 1, 64, avg, _8, sse2);
|
||||
init_lpf(sse2);
|
||||
dsp->itxfm_add[TX_4X4][ADST_DCT] = ff_vp9_idct_iadst_4x4_add_sse2;
|
||||
dsp->itxfm_add[TX_4X4][DCT_ADST] = ff_vp9_iadst_idct_4x4_add_sse2;
|
||||
dsp->itxfm_add[TX_4X4][ADST_ADST] = ff_vp9_iadst_iadst_4x4_add_sse2;
|
||||
dsp->itxfm_add[TX_8X8][DCT_DCT] = ff_vp9_idct_idct_8x8_add_sse2;
|
||||
dsp->itxfm_add[TX_8X8][ADST_DCT] = ff_vp9_idct_iadst_8x8_add_sse2;
|
||||
dsp->itxfm_add[TX_8X8][DCT_ADST] = ff_vp9_iadst_idct_8x8_add_sse2;
|
||||
dsp->itxfm_add[TX_8X8][ADST_ADST] = ff_vp9_iadst_iadst_8x8_add_sse2;
|
||||
dsp->itxfm_add[TX_16X16][DCT_DCT] = ff_vp9_idct_idct_16x16_add_sse2;
|
||||
dsp->itxfm_add[TX_16X16][ADST_DCT] = ff_vp9_idct_iadst_16x16_add_sse2;
|
||||
dsp->itxfm_add[TX_16X16][DCT_ADST] = ff_vp9_iadst_idct_16x16_add_sse2;
|
||||
dsp->itxfm_add[TX_16X16][ADST_ADST] = ff_vp9_iadst_iadst_16x16_add_sse2;
|
||||
dsp->itxfm_add[TX_32X32][ADST_ADST] =
|
||||
dsp->itxfm_add[TX_32X32][ADST_DCT] =
|
||||
dsp->itxfm_add[TX_32X32][DCT_ADST] =
|
||||
dsp->itxfm_add[TX_32X32][DCT_DCT] = ff_vp9_idct_idct_32x32_add_sse2;
|
||||
init_dc_ipred(16, sse2);
|
||||
init_dc_ipred(32, sse2);
|
||||
init_dir_tm_h_ipred(8, sse2);
|
||||
init_dir_tm_h_ipred(16, sse2);
|
||||
init_dir_tm_h_ipred(32, sse2);
|
||||
init_ipred(4, sse2, h, HOR);
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSSE3(cpu_flags)) {
|
||||
init_subpel3(0, put, 8, ssse3);
|
||||
init_subpel3(1, avg, 8, ssse3);
|
||||
dsp->itxfm_add[TX_4X4][DCT_DCT] = ff_vp9_idct_idct_4x4_add_ssse3;
|
||||
dsp->itxfm_add[TX_4X4][ADST_DCT] = ff_vp9_idct_iadst_4x4_add_ssse3;
|
||||
dsp->itxfm_add[TX_4X4][DCT_ADST] = ff_vp9_iadst_idct_4x4_add_ssse3;
|
||||
dsp->itxfm_add[TX_4X4][ADST_ADST] = ff_vp9_iadst_iadst_4x4_add_ssse3;
|
||||
dsp->itxfm_add[TX_8X8][DCT_DCT] = ff_vp9_idct_idct_8x8_add_ssse3;
|
||||
dsp->itxfm_add[TX_8X8][ADST_DCT] = ff_vp9_idct_iadst_8x8_add_ssse3;
|
||||
dsp->itxfm_add[TX_8X8][DCT_ADST] = ff_vp9_iadst_idct_8x8_add_ssse3;
|
||||
dsp->itxfm_add[TX_8X8][ADST_ADST] = ff_vp9_iadst_iadst_8x8_add_ssse3;
|
||||
dsp->itxfm_add[TX_16X16][DCT_DCT] = ff_vp9_idct_idct_16x16_add_ssse3;
|
||||
dsp->itxfm_add[TX_16X16][ADST_DCT] = ff_vp9_idct_iadst_16x16_add_ssse3;
|
||||
dsp->itxfm_add[TX_16X16][DCT_ADST] = ff_vp9_iadst_idct_16x16_add_ssse3;
|
||||
dsp->itxfm_add[TX_16X16][ADST_ADST] = ff_vp9_iadst_iadst_16x16_add_ssse3;
|
||||
dsp->itxfm_add[TX_32X32][ADST_ADST] =
|
||||
dsp->itxfm_add[TX_32X32][ADST_DCT] =
|
||||
dsp->itxfm_add[TX_32X32][DCT_ADST] =
|
||||
dsp->itxfm_add[TX_32X32][DCT_DCT] = ff_vp9_idct_idct_32x32_add_ssse3;
|
||||
init_lpf(ssse3);
|
||||
init_all_ipred(4, ssse3);
|
||||
init_all_ipred(8, ssse3);
|
||||
init_all_ipred(16, ssse3);
|
||||
init_all_ipred(32, ssse3);
|
||||
}
|
||||
|
||||
if (EXTERNAL_AVX(cpu_flags)) {
|
||||
dsp->itxfm_add[TX_8X8][DCT_DCT] = ff_vp9_idct_idct_8x8_add_avx;
|
||||
dsp->itxfm_add[TX_8X8][ADST_DCT] = ff_vp9_idct_iadst_8x8_add_avx;
|
||||
dsp->itxfm_add[TX_8X8][DCT_ADST] = ff_vp9_iadst_idct_8x8_add_avx;
|
||||
dsp->itxfm_add[TX_8X8][ADST_ADST] = ff_vp9_iadst_iadst_8x8_add_avx;
|
||||
dsp->itxfm_add[TX_16X16][DCT_DCT] = ff_vp9_idct_idct_16x16_add_avx;
|
||||
dsp->itxfm_add[TX_16X16][ADST_DCT] = ff_vp9_idct_iadst_16x16_add_avx;
|
||||
dsp->itxfm_add[TX_16X16][DCT_ADST] = ff_vp9_iadst_idct_16x16_add_avx;
|
||||
dsp->itxfm_add[TX_16X16][ADST_ADST] = ff_vp9_iadst_iadst_16x16_add_avx;
|
||||
dsp->itxfm_add[TX_32X32][ADST_ADST] =
|
||||
dsp->itxfm_add[TX_32X32][ADST_DCT] =
|
||||
dsp->itxfm_add[TX_32X32][DCT_ADST] =
|
||||
dsp->itxfm_add[TX_32X32][DCT_DCT] = ff_vp9_idct_idct_32x32_add_avx;
|
||||
init_lpf(avx);
|
||||
init_dir_tm_h_ipred(8, avx);
|
||||
init_dir_tm_h_ipred(16, avx);
|
||||
init_dir_tm_h_ipred(32, avx);
|
||||
}
|
||||
if (EXTERNAL_AVX_FAST(cpu_flags)) {
|
||||
init_fpel_func(1, 0, 32, put, , avx);
|
||||
init_fpel_func(0, 0, 64, put, , avx);
|
||||
init_ipred(32, avx, v, VERT);
|
||||
}
|
||||
|
||||
if (EXTERNAL_AVX2_FAST(cpu_flags)) {
|
||||
init_fpel_func(1, 1, 32, avg, _8, avx2);
|
||||
init_fpel_func(0, 1, 64, avg, _8, avx2);
|
||||
if (ARCH_X86_64) {
|
||||
#if ARCH_X86_64 && HAVE_AVX2_EXTERNAL
|
||||
init_subpel3_32_64(0, put, 8, avx2);
|
||||
init_subpel3_32_64(1, avg, 8, avx2);
|
||||
#endif
|
||||
}
|
||||
init_dc_ipred(32, avx2);
|
||||
init_ipred(32, avx2, h, HOR);
|
||||
init_ipred(32, avx2, tm, TM_VP8);
|
||||
}
|
||||
|
||||
#undef init_fpel
|
||||
#undef init_subpel1
|
||||
#undef init_subpel2
|
||||
#undef init_subpel3
|
||||
|
||||
#endif /* HAVE_YASM */
|
||||
}
|
||||
189
media/ffvpx/libavcodec/x86/vp9dsp_init.h
Normal file
189
media/ffvpx/libavcodec/x86/vp9dsp_init.h
Normal file
|
|
@ -0,0 +1,189 @@
|
|||
/*
|
||||
* VP9 SIMD optimizations
|
||||
*
|
||||
* Copyright (c) 2013 Ronald S. Bultje <rsbultje gmail com>
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#ifndef AVCODEC_X86_VP9DSP_INIT_H
|
||||
#define AVCODEC_X86_VP9DSP_INIT_H
|
||||
|
||||
#include "libavcodec/vp9dsp.h"
|
||||
|
||||
// hack to force-expand BPC
|
||||
#define cat(a, bpp, b) a##bpp##b
|
||||
|
||||
#define decl_fpel_func(avg, sz, bpp, opt) \
|
||||
void ff_vp9_##avg##sz##bpp##_##opt(uint8_t *dst, ptrdiff_t dst_stride, \
|
||||
const uint8_t *src, ptrdiff_t src_stride, \
|
||||
int h, int mx, int my)
|
||||
|
||||
#define decl_mc_func(avg, sz, dir, opt, type, f_sz, bpp) \
|
||||
void ff_vp9_##avg##_8tap_1d_##dir##_##sz##_##bpp##_##opt(uint8_t *dst, ptrdiff_t dst_stride, \
|
||||
const uint8_t *src, ptrdiff_t src_stride, \
|
||||
int h, const type (*filter)[f_sz])
|
||||
|
||||
#define decl_mc_funcs(sz, opt, type, fsz, bpp) \
|
||||
decl_mc_func(put, sz, h, opt, type, fsz, bpp); \
|
||||
decl_mc_func(avg, sz, h, opt, type, fsz, bpp); \
|
||||
decl_mc_func(put, sz, v, opt, type, fsz, bpp); \
|
||||
decl_mc_func(avg, sz, v, opt, type, fsz, bpp)
|
||||
|
||||
#define decl_ipred_fn(type, sz, bpp, opt) \
|
||||
void ff_vp9_ipred_##type##_##sz##x##sz##_##bpp##_##opt(uint8_t *dst, \
|
||||
ptrdiff_t stride, \
|
||||
const uint8_t *l, \
|
||||
const uint8_t *a)
|
||||
|
||||
#define decl_ipred_fns(type, bpp, opt4, opt8_16_32) \
|
||||
decl_ipred_fn(type, 4, bpp, opt4); \
|
||||
decl_ipred_fn(type, 8, bpp, opt8_16_32); \
|
||||
decl_ipred_fn(type, 16, bpp, opt8_16_32); \
|
||||
decl_ipred_fn(type, 32, bpp, opt8_16_32)
|
||||
|
||||
#define decl_itxfm_func(typea, typeb, size, bpp, opt) \
|
||||
void cat(ff_vp9_##typea##_##typeb##_##size##x##size##_add_, bpp, _##opt)(uint8_t *dst, \
|
||||
ptrdiff_t stride, \
|
||||
int16_t *block, \
|
||||
int eob)
|
||||
|
||||
#define decl_itxfm_funcs(size, bpp, opt) \
|
||||
decl_itxfm_func(idct, idct, size, bpp, opt); \
|
||||
decl_itxfm_func(iadst, idct, size, bpp, opt); \
|
||||
decl_itxfm_func(idct, iadst, size, bpp, opt); \
|
||||
decl_itxfm_func(iadst, iadst, size, bpp, opt)
|
||||
|
||||
#define mc_rep_func(avg, sz, hsz, hszb, dir, opt, type, f_sz, bpp) \
|
||||
static av_always_inline void \
|
||||
ff_vp9_##avg##_8tap_1d_##dir##_##sz##_##bpp##_##opt(uint8_t *dst, ptrdiff_t dst_stride, \
|
||||
const uint8_t *src, ptrdiff_t src_stride, \
|
||||
int h, const type (*filter)[f_sz]) \
|
||||
{ \
|
||||
ff_vp9_##avg##_8tap_1d_##dir##_##hsz##_##bpp##_##opt(dst, dst_stride, src, \
|
||||
src_stride, h, filter); \
|
||||
ff_vp9_##avg##_8tap_1d_##dir##_##hsz##_##bpp##_##opt(dst + hszb, dst_stride, src + hszb, \
|
||||
src_stride, h, filter); \
|
||||
}
|
||||
|
||||
#define mc_rep_funcs(sz, hsz, hszb, opt, type, fsz, bpp) \
|
||||
mc_rep_func(put, sz, hsz, hszb, h, opt, type, fsz, bpp) \
|
||||
mc_rep_func(avg, sz, hsz, hszb, h, opt, type, fsz, bpp) \
|
||||
mc_rep_func(put, sz, hsz, hszb, v, opt, type, fsz, bpp) \
|
||||
mc_rep_func(avg, sz, hsz, hszb, v, opt, type, fsz, bpp)
|
||||
|
||||
#define filter_8tap_1d_fn(op, sz, f, f_opt, fname, dir, dvar, bpp, opt) \
|
||||
static void op##_8tap_##fname##_##sz##dir##_##bpp##_##opt(uint8_t *dst, ptrdiff_t dst_stride, \
|
||||
const uint8_t *src, ptrdiff_t src_stride, \
|
||||
int h, int mx, int my) \
|
||||
{ \
|
||||
ff_vp9_##op##_8tap_1d_##dir##_##sz##_##bpp##_##opt(dst, dst_stride, src, src_stride, \
|
||||
h, ff_filters_##f_opt[f][dvar - 1]); \
|
||||
}
|
||||
|
||||
#define filters_8tap_1d_fn(op, sz, dir, dvar, bpp, opt, f_opt) \
|
||||
filter_8tap_1d_fn(op, sz, FILTER_8TAP_REGULAR, f_opt, regular, dir, dvar, bpp, opt) \
|
||||
filter_8tap_1d_fn(op, sz, FILTER_8TAP_SHARP, f_opt, sharp, dir, dvar, bpp, opt) \
|
||||
filter_8tap_1d_fn(op, sz, FILTER_8TAP_SMOOTH, f_opt, smooth, dir, dvar, bpp, opt)
|
||||
|
||||
#define filters_8tap_1d_fn2(op, sz, bpp, opt, f_opt) \
|
||||
filters_8tap_1d_fn(op, sz, h, mx, bpp, opt, f_opt) \
|
||||
filters_8tap_1d_fn(op, sz, v, my, bpp, opt, f_opt)
|
||||
|
||||
#define filters_8tap_1d_fn3(op, bpp, opt4, opt8, f_opt) \
|
||||
filters_8tap_1d_fn2(op, 64, bpp, opt8, f_opt) \
|
||||
filters_8tap_1d_fn2(op, 32, bpp, opt8, f_opt) \
|
||||
filters_8tap_1d_fn2(op, 16, bpp, opt8, f_opt) \
|
||||
filters_8tap_1d_fn2(op, 8, bpp, opt8, f_opt) \
|
||||
filters_8tap_1d_fn2(op, 4, bpp, opt4, f_opt)
|
||||
|
||||
#define filter_8tap_2d_fn(op, sz, f, f_opt, fname, align, bpp, bytes, opt) \
|
||||
static void op##_8tap_##fname##_##sz##hv_##bpp##_##opt(uint8_t *dst, ptrdiff_t dst_stride, \
|
||||
const uint8_t *src, ptrdiff_t src_stride, \
|
||||
int h, int mx, int my) \
|
||||
{ \
|
||||
LOCAL_ALIGNED_##align(uint8_t, temp, [71 * 64 * bytes]); \
|
||||
ff_vp9_put_8tap_1d_h_##sz##_##bpp##_##opt(temp, 64 * bytes, src - 3 * src_stride, \
|
||||
src_stride, h + 7, \
|
||||
ff_filters_##f_opt[f][mx - 1]); \
|
||||
ff_vp9_##op##_8tap_1d_v_##sz##_##bpp##_##opt(dst, dst_stride, temp + 3 * bytes * 64, \
|
||||
64 * bytes, h, \
|
||||
ff_filters_##f_opt[f][my - 1]); \
|
||||
}
|
||||
|
||||
#define filters_8tap_2d_fn(op, sz, align, bpp, bytes, opt, f_opt) \
|
||||
filter_8tap_2d_fn(op, sz, FILTER_8TAP_REGULAR, f_opt, regular, align, bpp, bytes, opt) \
|
||||
filter_8tap_2d_fn(op, sz, FILTER_8TAP_SHARP, f_opt, sharp, align, bpp, bytes, opt) \
|
||||
filter_8tap_2d_fn(op, sz, FILTER_8TAP_SMOOTH, f_opt, smooth, align, bpp, bytes, opt)
|
||||
|
||||
#define filters_8tap_2d_fn2(op, align, bpp, bytes, opt4, opt8, f_opt) \
|
||||
filters_8tap_2d_fn(op, 64, align, bpp, bytes, opt8, f_opt) \
|
||||
filters_8tap_2d_fn(op, 32, align, bpp, bytes, opt8, f_opt) \
|
||||
filters_8tap_2d_fn(op, 16, align, bpp, bytes, opt8, f_opt) \
|
||||
filters_8tap_2d_fn(op, 8, align, bpp, bytes, opt8, f_opt) \
|
||||
filters_8tap_2d_fn(op, 4, align, bpp, bytes, opt4, f_opt)
|
||||
|
||||
#define init_fpel_func(idx1, idx2, sz, type, bpp, opt) \
|
||||
dsp->mc[idx1][FILTER_8TAP_SMOOTH ][idx2][0][0] = \
|
||||
dsp->mc[idx1][FILTER_8TAP_REGULAR][idx2][0][0] = \
|
||||
dsp->mc[idx1][FILTER_8TAP_SHARP ][idx2][0][0] = \
|
||||
dsp->mc[idx1][FILTER_BILINEAR ][idx2][0][0] = ff_vp9_##type##sz##bpp##_##opt
|
||||
|
||||
#define init_subpel1(idx1, idx2, idxh, idxv, sz, dir, type, bpp, opt) \
|
||||
dsp->mc[idx1][FILTER_8TAP_SMOOTH ][idx2][idxh][idxv] = \
|
||||
type##_8tap_smooth_##sz##dir##_##bpp##_##opt; \
|
||||
dsp->mc[idx1][FILTER_8TAP_REGULAR][idx2][idxh][idxv] = \
|
||||
type##_8tap_regular_##sz##dir##_##bpp##_##opt; \
|
||||
dsp->mc[idx1][FILTER_8TAP_SHARP ][idx2][idxh][idxv] = \
|
||||
type##_8tap_sharp_##sz##dir##_##bpp##_##opt
|
||||
|
||||
#define init_subpel2(idx1, idx2, sz, type, bpp, opt) \
|
||||
init_subpel1(idx1, idx2, 1, 1, sz, hv, type, bpp, opt); \
|
||||
init_subpel1(idx1, idx2, 0, 1, sz, v, type, bpp, opt); \
|
||||
init_subpel1(idx1, idx2, 1, 0, sz, h, type, bpp, opt)
|
||||
|
||||
#define init_subpel3_32_64(idx, type, bpp, opt) \
|
||||
init_subpel2(0, idx, 64, type, bpp, opt); \
|
||||
init_subpel2(1, idx, 32, type, bpp, opt)
|
||||
|
||||
#define init_subpel3_8to64(idx, type, bpp, opt) \
|
||||
init_subpel3_32_64(idx, type, bpp, opt); \
|
||||
init_subpel2(2, idx, 16, type, bpp, opt); \
|
||||
init_subpel2(3, idx, 8, type, bpp, opt)
|
||||
|
||||
#define init_subpel3(idx, type, bpp, opt) \
|
||||
init_subpel3_8to64(idx, type, bpp, opt); \
|
||||
init_subpel2(4, idx, 4, type, bpp, opt)
|
||||
|
||||
#define init_ipred_func(type, enum, sz, bpp, opt) \
|
||||
dsp->intra_pred[TX_##sz##X##sz][enum##_PRED] = \
|
||||
cat(ff_vp9_ipred_##type##_##sz##x##sz##_, bpp, _##opt)
|
||||
|
||||
#define init_8_16_32_ipred_funcs(type, enum, bpp, opt) \
|
||||
init_ipred_func(type, enum, 8, bpp, opt); \
|
||||
init_ipred_func(type, enum, 16, bpp, opt); \
|
||||
init_ipred_func(type, enum, 32, bpp, opt)
|
||||
|
||||
#define init_ipred_funcs(type, enum, bpp, opt) \
|
||||
init_ipred_func(type, enum, 4, bpp, opt); \
|
||||
init_8_16_32_ipred_funcs(type, enum, bpp, opt)
|
||||
|
||||
void ff_vp9dsp_init_10bpp_x86(VP9DSPContext *dsp, int bitexact);
|
||||
void ff_vp9dsp_init_12bpp_x86(VP9DSPContext *dsp, int bitexact);
|
||||
void ff_vp9dsp_init_16bpp_x86(VP9DSPContext *dsp);
|
||||
|
||||
#endif /* AVCODEC_X86_VP9DSP_INIT_H */
|
||||
25
media/ffvpx/libavcodec/x86/vp9dsp_init_10bpp.c
Normal file
25
media/ffvpx/libavcodec/x86/vp9dsp_init_10bpp.c
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
/*
|
||||
* VP9 SIMD optimizations
|
||||
*
|
||||
* Copyright (c) 2013 Ronald S. Bultje <rsbultje gmail com>
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#define BPC 10
|
||||
#define INIT_FUNC ff_vp9dsp_init_10bpp_x86
|
||||
#include "vp9dsp_init_16bpp_template.c"
|
||||
25
media/ffvpx/libavcodec/x86/vp9dsp_init_12bpp.c
Normal file
25
media/ffvpx/libavcodec/x86/vp9dsp_init_12bpp.c
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
/*
|
||||
* VP9 SIMD optimizations
|
||||
*
|
||||
* Copyright (c) 2013 Ronald S. Bultje <rsbultje gmail com>
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#define BPC 12
|
||||
#define INIT_FUNC ff_vp9dsp_init_12bpp_x86
|
||||
#include "vp9dsp_init_16bpp_template.c"
|
||||
139
media/ffvpx/libavcodec/x86/vp9dsp_init_16bpp.c
Normal file
139
media/ffvpx/libavcodec/x86/vp9dsp_init_16bpp.c
Normal file
|
|
@ -0,0 +1,139 @@
|
|||
/*
|
||||
* VP9 SIMD optimizations
|
||||
*
|
||||
* Copyright (c) 2013 Ronald S. Bultje <rsbultje gmail com>
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#include "libavutil/attributes.h"
|
||||
#include "libavutil/cpu.h"
|
||||
#include "libavutil/mem.h"
|
||||
#include "libavutil/x86/cpu.h"
|
||||
#include "libavcodec/vp9dsp.h"
|
||||
#include "libavcodec/x86/vp9dsp_init.h"
|
||||
|
||||
#if HAVE_YASM
|
||||
|
||||
decl_fpel_func(put, 8, , mmx);
|
||||
decl_fpel_func(avg, 8, _16, mmxext);
|
||||
decl_fpel_func(put, 16, , sse);
|
||||
decl_fpel_func(put, 32, , sse);
|
||||
decl_fpel_func(put, 64, , sse);
|
||||
decl_fpel_func(put, 128, , sse);
|
||||
decl_fpel_func(avg, 16, _16, sse2);
|
||||
decl_fpel_func(avg, 32, _16, sse2);
|
||||
decl_fpel_func(avg, 64, _16, sse2);
|
||||
decl_fpel_func(avg, 128, _16, sse2);
|
||||
decl_fpel_func(put, 32, , avx);
|
||||
decl_fpel_func(put, 64, , avx);
|
||||
decl_fpel_func(put, 128, , avx);
|
||||
decl_fpel_func(avg, 32, _16, avx2);
|
||||
decl_fpel_func(avg, 64, _16, avx2);
|
||||
decl_fpel_func(avg, 128, _16, avx2);
|
||||
|
||||
decl_ipred_fns(v, 16, mmx, sse);
|
||||
decl_ipred_fns(h, 16, mmxext, sse2);
|
||||
decl_ipred_fns(dc, 16, mmxext, sse2);
|
||||
decl_ipred_fns(dc_top, 16, mmxext, sse2);
|
||||
decl_ipred_fns(dc_left, 16, mmxext, sse2);
|
||||
|
||||
#define decl_ipred_dir_funcs(type) \
|
||||
decl_ipred_fns(type, 16, sse2, sse2); \
|
||||
decl_ipred_fns(type, 16, ssse3, ssse3); \
|
||||
decl_ipred_fns(type, 16, avx, avx)
|
||||
|
||||
decl_ipred_dir_funcs(dl);
|
||||
decl_ipred_dir_funcs(dr);
|
||||
decl_ipred_dir_funcs(vl);
|
||||
decl_ipred_dir_funcs(vr);
|
||||
decl_ipred_dir_funcs(hu);
|
||||
decl_ipred_dir_funcs(hd);
|
||||
#endif /* HAVE_YASM */
|
||||
|
||||
av_cold void ff_vp9dsp_init_16bpp_x86(VP9DSPContext *dsp)
|
||||
{
|
||||
#if HAVE_YASM
|
||||
int cpu_flags = av_get_cpu_flags();
|
||||
|
||||
if (EXTERNAL_MMX(cpu_flags)) {
|
||||
init_fpel_func(4, 0, 8, put, , mmx);
|
||||
init_ipred_func(v, VERT, 4, 16, mmx);
|
||||
}
|
||||
|
||||
if (EXTERNAL_MMXEXT(cpu_flags)) {
|
||||
init_fpel_func(4, 1, 8, avg, _16, mmxext);
|
||||
init_ipred_func(h, HOR, 4, 16, mmxext);
|
||||
init_ipred_func(dc, DC, 4, 16, mmxext);
|
||||
init_ipred_func(dc_top, TOP_DC, 4, 16, mmxext);
|
||||
init_ipred_func(dc_left, LEFT_DC, 4, 16, mmxext);
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE(cpu_flags)) {
|
||||
init_fpel_func(3, 0, 16, put, , sse);
|
||||
init_fpel_func(2, 0, 32, put, , sse);
|
||||
init_fpel_func(1, 0, 64, put, , sse);
|
||||
init_fpel_func(0, 0, 128, put, , sse);
|
||||
init_8_16_32_ipred_funcs(v, VERT, 16, sse);
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE2(cpu_flags)) {
|
||||
init_fpel_func(3, 1, 16, avg, _16, sse2);
|
||||
init_fpel_func(2, 1, 32, avg, _16, sse2);
|
||||
init_fpel_func(1, 1, 64, avg, _16, sse2);
|
||||
init_fpel_func(0, 1, 128, avg, _16, sse2);
|
||||
init_8_16_32_ipred_funcs(h, HOR, 16, sse2);
|
||||
init_8_16_32_ipred_funcs(dc, DC, 16, sse2);
|
||||
init_8_16_32_ipred_funcs(dc_top, TOP_DC, 16, sse2);
|
||||
init_8_16_32_ipred_funcs(dc_left, LEFT_DC, 16, sse2);
|
||||
init_ipred_funcs(dl, DIAG_DOWN_LEFT, 16, sse2);
|
||||
init_ipred_funcs(dr, DIAG_DOWN_RIGHT, 16, sse2);
|
||||
init_ipred_funcs(vl, VERT_LEFT, 16, sse2);
|
||||
init_ipred_funcs(vr, VERT_RIGHT, 16, sse2);
|
||||
init_ipred_funcs(hu, HOR_UP, 16, sse2);
|
||||
init_ipred_funcs(hd, HOR_DOWN, 16, sse2);
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSSE3(cpu_flags)) {
|
||||
init_ipred_funcs(dl, DIAG_DOWN_LEFT, 16, ssse3);
|
||||
init_ipred_funcs(dr, DIAG_DOWN_RIGHT, 16, ssse3);
|
||||
init_ipred_funcs(vl, VERT_LEFT, 16, ssse3);
|
||||
init_ipred_funcs(vr, VERT_RIGHT, 16, ssse3);
|
||||
init_ipred_funcs(hu, HOR_UP, 16, ssse3);
|
||||
init_ipred_funcs(hd, HOR_DOWN, 16, ssse3);
|
||||
}
|
||||
|
||||
if (EXTERNAL_AVX_FAST(cpu_flags)) {
|
||||
init_fpel_func(2, 0, 32, put, , avx);
|
||||
init_fpel_func(1, 0, 64, put, , avx);
|
||||
init_fpel_func(0, 0, 128, put, , avx);
|
||||
init_ipred_funcs(dl, DIAG_DOWN_LEFT, 16, avx);
|
||||
init_ipred_funcs(dr, DIAG_DOWN_RIGHT, 16, avx);
|
||||
init_ipred_funcs(vl, VERT_LEFT, 16, avx);
|
||||
init_ipred_funcs(vr, VERT_RIGHT, 16, avx);
|
||||
init_ipred_funcs(hu, HOR_UP, 16, avx);
|
||||
init_ipred_funcs(hd, HOR_DOWN, 16, avx);
|
||||
}
|
||||
|
||||
if (EXTERNAL_AVX2_FAST(cpu_flags)) {
|
||||
init_fpel_func(2, 1, 32, avg, _16, avx2);
|
||||
init_fpel_func(1, 1, 64, avg, _16, avx2);
|
||||
init_fpel_func(0, 1, 128, avg, _16, avx2);
|
||||
}
|
||||
|
||||
#endif /* HAVE_YASM */
|
||||
}
|
||||
240
media/ffvpx/libavcodec/x86/vp9dsp_init_16bpp_template.c
Normal file
240
media/ffvpx/libavcodec/x86/vp9dsp_init_16bpp_template.c
Normal file
|
|
@ -0,0 +1,240 @@
|
|||
/*
|
||||
* VP9 SIMD optimizations
|
||||
*
|
||||
* Copyright (c) 2013 Ronald S. Bultje <rsbultje gmail com>
|
||||
*
|
||||
* This file is part of FFmpeg.
|
||||
*
|
||||
* FFmpeg is free software; you can redistribute it and/or
|
||||
* modify it under the terms of the GNU Lesser General Public
|
||||
* License as published by the Free Software Foundation; either
|
||||
* version 2.1 of the License, or (at your option) any later version.
|
||||
*
|
||||
* FFmpeg is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
* Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public
|
||||
* License along with FFmpeg; if not, write to the Free Software
|
||||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
*/
|
||||
|
||||
#include "libavutil/attributes.h"
|
||||
#include "libavutil/cpu.h"
|
||||
#include "libavutil/mem.h"
|
||||
#include "libavutil/x86/cpu.h"
|
||||
#include "libavcodec/vp9dsp.h"
|
||||
#include "libavcodec/x86/vp9dsp_init.h"
|
||||
|
||||
#if HAVE_YASM
|
||||
|
||||
extern const int16_t ff_filters_16bpp[3][15][4][16];
|
||||
|
||||
decl_mc_funcs(4, sse2, int16_t, 16, BPC);
|
||||
decl_mc_funcs(8, sse2, int16_t, 16, BPC);
|
||||
decl_mc_funcs(16, avx2, int16_t, 16, BPC);
|
||||
|
||||
mc_rep_funcs(16, 8, 16, sse2, int16_t, 16, BPC)
|
||||
mc_rep_funcs(32, 16, 32, sse2, int16_t, 16, BPC)
|
||||
mc_rep_funcs(64, 32, 64, sse2, int16_t, 16, BPC)
|
||||
#if HAVE_AVX2_EXTERNAL
|
||||
mc_rep_funcs(32, 16, 32, avx2, int16_t, 16, BPC)
|
||||
mc_rep_funcs(64, 32, 64, avx2, int16_t, 16, BPC)
|
||||
#endif
|
||||
|
||||
filters_8tap_2d_fn2(put, 16, BPC, 2, sse2, sse2, 16bpp)
|
||||
filters_8tap_2d_fn2(avg, 16, BPC, 2, sse2, sse2, 16bpp)
|
||||
#if HAVE_AVX2_EXTERNAL
|
||||
filters_8tap_2d_fn(put, 64, 32, BPC, 2, avx2, 16bpp)
|
||||
filters_8tap_2d_fn(avg, 64, 32, BPC, 2, avx2, 16bpp)
|
||||
filters_8tap_2d_fn(put, 32, 32, BPC, 2, avx2, 16bpp)
|
||||
filters_8tap_2d_fn(avg, 32, 32, BPC, 2, avx2, 16bpp)
|
||||
filters_8tap_2d_fn(put, 16, 32, BPC, 2, avx2, 16bpp)
|
||||
filters_8tap_2d_fn(avg, 16, 32, BPC, 2, avx2, 16bpp)
|
||||
#endif
|
||||
|
||||
filters_8tap_1d_fn3(put, BPC, sse2, sse2, 16bpp)
|
||||
filters_8tap_1d_fn3(avg, BPC, sse2, sse2, 16bpp)
|
||||
#if HAVE_AVX2_EXTERNAL
|
||||
filters_8tap_1d_fn2(put, 64, BPC, avx2, 16bpp)
|
||||
filters_8tap_1d_fn2(avg, 64, BPC, avx2, 16bpp)
|
||||
filters_8tap_1d_fn2(put, 32, BPC, avx2, 16bpp)
|
||||
filters_8tap_1d_fn2(avg, 32, BPC, avx2, 16bpp)
|
||||
filters_8tap_1d_fn2(put, 16, BPC, avx2, 16bpp)
|
||||
filters_8tap_1d_fn2(avg, 16, BPC, avx2, 16bpp)
|
||||
#endif
|
||||
|
||||
#define decl_lpf_func(dir, wd, bpp, opt) \
|
||||
void ff_vp9_loop_filter_##dir##_##wd##_##bpp##_##opt(uint8_t *dst, ptrdiff_t stride, \
|
||||
int E, int I, int H)
|
||||
|
||||
#define decl_lpf_funcs(dir, wd, bpp) \
|
||||
decl_lpf_func(dir, wd, bpp, sse2); \
|
||||
decl_lpf_func(dir, wd, bpp, ssse3); \
|
||||
decl_lpf_func(dir, wd, bpp, avx)
|
||||
|
||||
#define decl_lpf_funcs_wd(dir) \
|
||||
decl_lpf_funcs(dir, 4, BPC); \
|
||||
decl_lpf_funcs(dir, 8, BPC); \
|
||||
decl_lpf_funcs(dir, 16, BPC)
|
||||
|
||||
decl_lpf_funcs_wd(h);
|
||||
decl_lpf_funcs_wd(v);
|
||||
|
||||
#define lpf_16_wrapper(dir, off, bpp, opt) \
|
||||
static void loop_filter_##dir##_16_##bpp##_##opt(uint8_t *dst, ptrdiff_t stride, \
|
||||
int E, int I, int H) \
|
||||
{ \
|
||||
ff_vp9_loop_filter_##dir##_16_##bpp##_##opt(dst, stride, E, I, H); \
|
||||
ff_vp9_loop_filter_##dir##_16_##bpp##_##opt(dst + off, stride, E, I, H); \
|
||||
}
|
||||
|
||||
#define lpf_16_wrappers(bpp, opt) \
|
||||
lpf_16_wrapper(h, 8 * stride, bpp, opt) \
|
||||
lpf_16_wrapper(v, 16, bpp, opt)
|
||||
|
||||
lpf_16_wrappers(BPC, sse2)
|
||||
lpf_16_wrappers(BPC, ssse3)
|
||||
lpf_16_wrappers(BPC, avx)
|
||||
|
||||
#define lpf_mix2_wrapper(dir, off, wd1, wd2, bpp, opt) \
|
||||
static void loop_filter_##dir##_##wd1##wd2##_##bpp##_##opt(uint8_t *dst, ptrdiff_t stride, \
|
||||
int E, int I, int H) \
|
||||
{ \
|
||||
ff_vp9_loop_filter_##dir##_##wd1##_##bpp##_##opt(dst, stride, \
|
||||
E & 0xff, I & 0xff, H & 0xff); \
|
||||
ff_vp9_loop_filter_##dir##_##wd2##_##bpp##_##opt(dst + off, stride, \
|
||||
E >> 8, I >> 8, H >> 8); \
|
||||
}
|
||||
|
||||
#define lpf_mix2_wrappers(wd1, wd2, bpp, opt) \
|
||||
lpf_mix2_wrapper(h, 8 * stride, wd1, wd2, bpp, opt) \
|
||||
lpf_mix2_wrapper(v, 16, wd1, wd2, bpp, opt)
|
||||
|
||||
#define lpf_mix2_wrappers_set(bpp, opt) \
|
||||
lpf_mix2_wrappers(4, 4, bpp, opt) \
|
||||
lpf_mix2_wrappers(4, 8, bpp, opt) \
|
||||
lpf_mix2_wrappers(8, 4, bpp, opt) \
|
||||
lpf_mix2_wrappers(8, 8, bpp, opt) \
|
||||
|
||||
lpf_mix2_wrappers_set(BPC, sse2)
|
||||
lpf_mix2_wrappers_set(BPC, ssse3)
|
||||
lpf_mix2_wrappers_set(BPC, avx)
|
||||
|
||||
decl_ipred_fns(tm, BPC, mmxext, sse2);
|
||||
|
||||
decl_itxfm_func(iwht, iwht, 4, BPC, mmxext);
|
||||
#if BPC == 10
|
||||
decl_itxfm_func(idct, idct, 4, BPC, mmxext);
|
||||
decl_itxfm_funcs(4, BPC, ssse3);
|
||||
#else
|
||||
decl_itxfm_func(idct, idct, 4, BPC, sse2);
|
||||
#endif
|
||||
decl_itxfm_func(idct, iadst, 4, BPC, sse2);
|
||||
decl_itxfm_func(iadst, idct, 4, BPC, sse2);
|
||||
decl_itxfm_func(iadst, iadst, 4, BPC, sse2);
|
||||
decl_itxfm_funcs(8, BPC, sse2);
|
||||
decl_itxfm_funcs(16, BPC, sse2);
|
||||
decl_itxfm_func(idct, idct, 32, BPC, sse2);
|
||||
#endif /* HAVE_YASM */
|
||||
|
||||
av_cold void INIT_FUNC(VP9DSPContext *dsp, int bitexact)
|
||||
{
|
||||
#if HAVE_YASM
|
||||
int cpu_flags = av_get_cpu_flags();
|
||||
|
||||
#define init_lpf_8_func(idx1, idx2, dir, wd, bpp, opt) \
|
||||
dsp->loop_filter_8[idx1][idx2] = ff_vp9_loop_filter_##dir##_##wd##_##bpp##_##opt
|
||||
#define init_lpf_16_func(idx, dir, bpp, opt) \
|
||||
dsp->loop_filter_16[idx] = loop_filter_##dir##_16_##bpp##_##opt
|
||||
#define init_lpf_mix2_func(idx1, idx2, idx3, dir, wd1, wd2, bpp, opt) \
|
||||
dsp->loop_filter_mix2[idx1][idx2][idx3] = loop_filter_##dir##_##wd1##wd2##_##bpp##_##opt
|
||||
|
||||
#define init_lpf_funcs(bpp, opt) \
|
||||
init_lpf_8_func(0, 0, h, 4, bpp, opt); \
|
||||
init_lpf_8_func(0, 1, v, 4, bpp, opt); \
|
||||
init_lpf_8_func(1, 0, h, 8, bpp, opt); \
|
||||
init_lpf_8_func(1, 1, v, 8, bpp, opt); \
|
||||
init_lpf_8_func(2, 0, h, 16, bpp, opt); \
|
||||
init_lpf_8_func(2, 1, v, 16, bpp, opt); \
|
||||
init_lpf_16_func(0, h, bpp, opt); \
|
||||
init_lpf_16_func(1, v, bpp, opt); \
|
||||
init_lpf_mix2_func(0, 0, 0, h, 4, 4, bpp, opt); \
|
||||
init_lpf_mix2_func(0, 1, 0, h, 4, 8, bpp, opt); \
|
||||
init_lpf_mix2_func(1, 0, 0, h, 8, 4, bpp, opt); \
|
||||
init_lpf_mix2_func(1, 1, 0, h, 8, 8, bpp, opt); \
|
||||
init_lpf_mix2_func(0, 0, 1, v, 4, 4, bpp, opt); \
|
||||
init_lpf_mix2_func(0, 1, 1, v, 4, 8, bpp, opt); \
|
||||
init_lpf_mix2_func(1, 0, 1, v, 8, 4, bpp, opt); \
|
||||
init_lpf_mix2_func(1, 1, 1, v, 8, 8, bpp, opt)
|
||||
|
||||
#define init_itx_func(idxa, idxb, typea, typeb, size, bpp, opt) \
|
||||
dsp->itxfm_add[idxa][idxb] = \
|
||||
cat(ff_vp9_##typea##_##typeb##_##size##x##size##_add_, bpp, _##opt);
|
||||
#define init_itx_func_one(idx, typea, typeb, size, bpp, opt) \
|
||||
init_itx_func(idx, DCT_DCT, typea, typeb, size, bpp, opt); \
|
||||
init_itx_func(idx, ADST_DCT, typea, typeb, size, bpp, opt); \
|
||||
init_itx_func(idx, DCT_ADST, typea, typeb, size, bpp, opt); \
|
||||
init_itx_func(idx, ADST_ADST, typea, typeb, size, bpp, opt)
|
||||
#define init_itx_funcs(idx, size, bpp, opt) \
|
||||
init_itx_func(idx, DCT_DCT, idct, idct, size, bpp, opt); \
|
||||
init_itx_func(idx, ADST_DCT, idct, iadst, size, bpp, opt); \
|
||||
init_itx_func(idx, DCT_ADST, iadst, idct, size, bpp, opt); \
|
||||
init_itx_func(idx, ADST_ADST, iadst, iadst, size, bpp, opt); \
|
||||
|
||||
if (EXTERNAL_MMXEXT(cpu_flags)) {
|
||||
init_ipred_func(tm, TM_VP8, 4, BPC, mmxext);
|
||||
if (!bitexact) {
|
||||
init_itx_func_one(4 /* lossless */, iwht, iwht, 4, BPC, mmxext);
|
||||
#if BPC == 10
|
||||
init_itx_func(TX_4X4, DCT_DCT, idct, idct, 4, 10, mmxext);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSE2(cpu_flags)) {
|
||||
init_subpel3(0, put, BPC, sse2);
|
||||
init_subpel3(1, avg, BPC, sse2);
|
||||
init_lpf_funcs(BPC, sse2);
|
||||
init_8_16_32_ipred_funcs(tm, TM_VP8, BPC, sse2);
|
||||
#if BPC == 10
|
||||
if (!bitexact) {
|
||||
init_itx_func(TX_4X4, ADST_DCT, idct, iadst, 4, 10, sse2);
|
||||
init_itx_func(TX_4X4, DCT_ADST, iadst, idct, 4, 10, sse2);
|
||||
init_itx_func(TX_4X4, ADST_ADST, iadst, iadst, 4, 10, sse2);
|
||||
}
|
||||
#else
|
||||
init_itx_funcs(TX_4X4, 4, 12, sse2);
|
||||
#endif
|
||||
init_itx_funcs(TX_8X8, 8, BPC, sse2);
|
||||
init_itx_funcs(TX_16X16, 16, BPC, sse2);
|
||||
init_itx_func_one(TX_32X32, idct, idct, 32, BPC, sse2);
|
||||
}
|
||||
|
||||
if (EXTERNAL_SSSE3(cpu_flags)) {
|
||||
init_lpf_funcs(BPC, ssse3);
|
||||
#if BPC == 10
|
||||
if (!bitexact) {
|
||||
init_itx_funcs(TX_4X4, 4, BPC, ssse3);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
if (EXTERNAL_AVX(cpu_flags)) {
|
||||
init_lpf_funcs(BPC, avx);
|
||||
}
|
||||
|
||||
if (EXTERNAL_AVX2_FAST(cpu_flags)) {
|
||||
#if HAVE_AVX2_EXTERNAL
|
||||
init_subpel3_32_64(0, put, BPC, avx2);
|
||||
init_subpel3_32_64(1, avg, BPC, avx2);
|
||||
init_subpel2(2, 0, 16, put, BPC, avx2);
|
||||
init_subpel2(2, 1, 16, avg, BPC, avx2);
|
||||
#endif
|
||||
}
|
||||
|
||||
#endif /* HAVE_YASM */
|
||||
|
||||
ff_vp9dsp_init_16bpp_x86(dsp);
|
||||
}
|
||||
2044
media/ffvpx/libavcodec/x86/vp9intrapred.asm
Normal file
2044
media/ffvpx/libavcodec/x86/vp9intrapred.asm
Normal file
File diff suppressed because it is too large
Load diff
2135
media/ffvpx/libavcodec/x86/vp9intrapred_16bpp.asm
Normal file
2135
media/ffvpx/libavcodec/x86/vp9intrapred_16bpp.asm
Normal file
File diff suppressed because it is too large
Load diff
2625
media/ffvpx/libavcodec/x86/vp9itxfm.asm
Normal file
2625
media/ffvpx/libavcodec/x86/vp9itxfm.asm
Normal file
File diff suppressed because it is too large
Load diff
2044
media/ffvpx/libavcodec/x86/vp9itxfm_16bpp.asm
Normal file
2044
media/ffvpx/libavcodec/x86/vp9itxfm_16bpp.asm
Normal file
File diff suppressed because it is too large
Load diff
142
media/ffvpx/libavcodec/x86/vp9itxfm_template.asm
Normal file
142
media/ffvpx/libavcodec/x86/vp9itxfm_template.asm
Normal file
|
|
@ -0,0 +1,142 @@
|
|||
;******************************************************************************
|
||||
;* VP9 IDCT SIMD optimizations
|
||||
;*
|
||||
;* Copyright (C) 2013 Clément Bœsch <u pkh me>
|
||||
;* Copyright (C) 2013 Ronald S. Bultje <rsbultje gmail com>
|
||||
;*
|
||||
;* This file is part of FFmpeg.
|
||||
;*
|
||||
;* FFmpeg is free software; you can redistribute it and/or
|
||||
;* modify it under the terms of the GNU Lesser General Public
|
||||
;* License as published by the Free Software Foundation; either
|
||||
;* version 2.1 of the License, or (at your option) any later version.
|
||||
;*
|
||||
;* FFmpeg is distributed in the hope that it will be useful,
|
||||
;* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
;* Lesser General Public License for more details.
|
||||
;*
|
||||
;* You should have received a copy of the GNU Lesser General Public
|
||||
;* License along with FFmpeg; if not, write to the Free Software
|
||||
;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
;******************************************************************************
|
||||
|
||||
%macro VP9_IWHT4_1D 0
|
||||
SWAP 1, 2, 3
|
||||
paddw m0, m2
|
||||
psubw m3, m1
|
||||
psubw m4, m0, m3
|
||||
psraw m4, 1
|
||||
psubw m5, m4, m1
|
||||
SWAP 5, 1
|
||||
psubw m4, m2
|
||||
SWAP 4, 2
|
||||
psubw m0, m1
|
||||
paddw m3, m2
|
||||
SWAP 3, 2, 1
|
||||
%endmacro
|
||||
|
||||
; (a*x + b*y + round) >> shift
|
||||
%macro VP9_MULSUB_2W_2X 5 ; dst1, dst2/src, round, coefs1, coefs2
|
||||
pmaddwd m%1, m%2, %4
|
||||
pmaddwd m%2, %5
|
||||
paddd m%1, %3
|
||||
paddd m%2, %3
|
||||
psrad m%1, 14
|
||||
psrad m%2, 14
|
||||
%endmacro
|
||||
|
||||
%macro VP9_MULSUB_2W_4X 7 ; dst1, dst2, coef1, coef2, rnd, tmp1/src, tmp2
|
||||
VP9_MULSUB_2W_2X %7, %6, %5, [pw_m%3_%4], [pw_%4_%3]
|
||||
VP9_MULSUB_2W_2X %1, %2, %5, [pw_m%3_%4], [pw_%4_%3]
|
||||
packssdw m%1, m%7
|
||||
packssdw m%2, m%6
|
||||
%endmacro
|
||||
|
||||
%macro VP9_UNPACK_MULSUB_2W_4X 7-9 ; dst1, dst2, (src1, src2,) coef1, coef2, rnd, tmp1, tmp2
|
||||
%if %0 == 7
|
||||
punpckhwd m%6, m%2, m%1
|
||||
punpcklwd m%2, m%1
|
||||
VP9_MULSUB_2W_4X %1, %2, %3, %4, %5, %6, %7
|
||||
%else
|
||||
punpckhwd m%8, m%4, m%3
|
||||
punpcklwd m%2, m%4, m%3
|
||||
VP9_MULSUB_2W_4X %1, %2, %5, %6, %7, %8, %9
|
||||
%endif
|
||||
%endmacro
|
||||
|
||||
%macro VP9_IDCT4_1D_FINALIZE 0
|
||||
SUMSUB_BA w, 3, 2, 4 ; m3=t3+t0, m2=-t3+t0
|
||||
SUMSUB_BA w, 1, 0, 4 ; m1=t2+t1, m0=-t2+t1
|
||||
SWAP 0, 3, 2 ; 3102 -> 0123
|
||||
%endmacro
|
||||
|
||||
%macro VP9_IDCT4_1D 0
|
||||
%if cpuflag(ssse3)
|
||||
SUMSUB_BA w, 2, 0, 4 ; m2=IN(0)+IN(2) m0=IN(0)-IN(2)
|
||||
pmulhrsw m2, m6 ; m2=t0
|
||||
pmulhrsw m0, m6 ; m0=t1
|
||||
%else ; <= sse2
|
||||
VP9_UNPACK_MULSUB_2W_4X 0, 2, 11585, 11585, m7, 4, 5 ; m0=t1, m1=t0
|
||||
%endif
|
||||
VP9_UNPACK_MULSUB_2W_4X 1, 3, 15137, 6270, m7, 4, 5 ; m1=t2, m3=t3
|
||||
VP9_IDCT4_1D_FINALIZE
|
||||
%endmacro
|
||||
|
||||
%macro VP9_IADST4_1D 0
|
||||
movq2dq xmm0, m0
|
||||
movq2dq xmm1, m1
|
||||
movq2dq xmm2, m2
|
||||
movq2dq xmm3, m3
|
||||
%if cpuflag(ssse3)
|
||||
paddw m3, m0
|
||||
%endif
|
||||
punpcklwd xmm0, xmm1
|
||||
punpcklwd xmm2, xmm3
|
||||
pmaddwd xmm1, xmm0, [pw_5283_13377]
|
||||
pmaddwd xmm4, xmm0, [pw_9929_13377]
|
||||
%if notcpuflag(ssse3)
|
||||
pmaddwd xmm6, xmm0, [pw_13377_0]
|
||||
%endif
|
||||
pmaddwd xmm0, [pw_15212_m13377]
|
||||
pmaddwd xmm3, xmm2, [pw_15212_9929]
|
||||
%if notcpuflag(ssse3)
|
||||
pmaddwd xmm7, xmm2, [pw_m13377_13377]
|
||||
%endif
|
||||
pmaddwd xmm2, [pw_m5283_m15212]
|
||||
%if cpuflag(ssse3)
|
||||
psubw m3, m2
|
||||
%else
|
||||
paddd xmm6, xmm7
|
||||
%endif
|
||||
paddd xmm0, xmm2
|
||||
paddd xmm3, xmm5
|
||||
paddd xmm2, xmm5
|
||||
%if notcpuflag(ssse3)
|
||||
paddd xmm6, xmm5
|
||||
%endif
|
||||
paddd xmm1, xmm3
|
||||
paddd xmm0, xmm3
|
||||
paddd xmm4, xmm2
|
||||
psrad xmm1, 14
|
||||
psrad xmm0, 14
|
||||
psrad xmm4, 14
|
||||
%if cpuflag(ssse3)
|
||||
pmulhrsw m3, [pw_13377x2] ; out2
|
||||
%else
|
||||
psrad xmm6, 14
|
||||
%endif
|
||||
packssdw xmm0, xmm0
|
||||
packssdw xmm1, xmm1
|
||||
packssdw xmm4, xmm4
|
||||
%if notcpuflag(ssse3)
|
||||
packssdw xmm6, xmm6
|
||||
%endif
|
||||
movdq2q m0, xmm0 ; out3
|
||||
movdq2q m1, xmm1 ; out0
|
||||
movdq2q m2, xmm4 ; out1
|
||||
%if notcpuflag(ssse3)
|
||||
movdq2q m3, xmm6 ; out2
|
||||
%endif
|
||||
SWAP 0, 1, 2, 3
|
||||
%endmacro
|
||||
1139
media/ffvpx/libavcodec/x86/vp9lpf.asm
Normal file
1139
media/ffvpx/libavcodec/x86/vp9lpf.asm
Normal file
File diff suppressed because it is too large
Load diff
823
media/ffvpx/libavcodec/x86/vp9lpf_16bpp.asm
Normal file
823
media/ffvpx/libavcodec/x86/vp9lpf_16bpp.asm
Normal file
|
|
@ -0,0 +1,823 @@
|
|||
;******************************************************************************
|
||||
;* VP9 loop filter SIMD optimizations
|
||||
;*
|
||||
;* Copyright (C) 2015 Ronald S. Bultje <rsbultje@gmail.com>
|
||||
;*
|
||||
;* This file is part of FFmpeg.
|
||||
;*
|
||||
;* FFmpeg is free software; you can redistribute it and/or
|
||||
;* modify it under the terms of the GNU Lesser General Public
|
||||
;* License as published by the Free Software Foundation; either
|
||||
;* version 2.1 of the License, or (at your option) any later version.
|
||||
;*
|
||||
;* FFmpeg is distributed in the hope that it will be useful,
|
||||
;* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
;* Lesser General Public License for more details.
|
||||
;*
|
||||
;* You should have received a copy of the GNU Lesser General Public
|
||||
;* License along with FFmpeg; if not, write to the Free Software
|
||||
;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
;******************************************************************************
|
||||
|
||||
%include "libavutil/x86/x86util.asm"
|
||||
|
||||
SECTION_RODATA
|
||||
|
||||
pw_511: times 16 dw 511
|
||||
pw_2047: times 16 dw 2047
|
||||
pw_16384: times 16 dw 16384
|
||||
pw_m512: times 16 dw -512
|
||||
pw_m2048: times 16 dw -2048
|
||||
|
||||
cextern pw_1
|
||||
cextern pw_3
|
||||
cextern pw_4
|
||||
cextern pw_8
|
||||
cextern pw_16
|
||||
cextern pw_256
|
||||
cextern pw_1023
|
||||
cextern pw_4095
|
||||
cextern pw_m1
|
||||
|
||||
SECTION .text
|
||||
|
||||
%macro SCRATCH 3-4
|
||||
%if ARCH_X86_64
|
||||
SWAP %1, %2
|
||||
%if %0 == 4
|
||||
%define reg_%4 m%2
|
||||
%endif
|
||||
%else
|
||||
mova [%3], m%1
|
||||
%if %0 == 4
|
||||
%define reg_%4 [%3]
|
||||
%endif
|
||||
%endif
|
||||
%endmacro
|
||||
|
||||
%macro UNSCRATCH 3-4
|
||||
%if ARCH_X86_64
|
||||
SWAP %1, %2
|
||||
%else
|
||||
mova m%1, [%3]
|
||||
%endif
|
||||
%if %0 == 4
|
||||
%undef reg_%4
|
||||
%endif
|
||||
%endmacro
|
||||
|
||||
%macro PRELOAD 2-3
|
||||
%if ARCH_X86_64
|
||||
mova m%1, [%2]
|
||||
%if %0 == 3
|
||||
%define reg_%3 m%1
|
||||
%endif
|
||||
%elif %0 == 3
|
||||
%define reg_%3 [%2]
|
||||
%endif
|
||||
%endmacro
|
||||
|
||||
; calulate p or q portion of flat8out
|
||||
%macro FLAT8OUT_HALF 0
|
||||
psubw m4, m0 ; q4-q0
|
||||
psubw m5, m0 ; q5-q0
|
||||
psubw m6, m0 ; q6-q0
|
||||
psubw m7, m0 ; q7-q0
|
||||
ABS2 m4, m5, m2, m3 ; abs(q4-q0) | abs(q5-q0)
|
||||
ABS2 m6, m7, m2, m3 ; abs(q6-q0) | abs(q7-q0)
|
||||
pcmpgtw m4, reg_F ; abs(q4-q0) > F
|
||||
pcmpgtw m5, reg_F ; abs(q5-q0) > F
|
||||
pcmpgtw m6, reg_F ; abs(q6-q0) > F
|
||||
pcmpgtw m7, reg_F ; abs(q7-q0) > F
|
||||
por m5, m4
|
||||
por m7, m6
|
||||
por m7, m5 ; !flat8out, q portion
|
||||
%endmacro
|
||||
|
||||
; calculate p or q portion of flat8in/hev/fm (excluding mb_edge condition)
|
||||
%macro FLAT8IN_HALF 1
|
||||
%if %1 > 4
|
||||
psubw m4, m3, m0 ; q3-q0
|
||||
psubw m5, m2, m0 ; q2-q0
|
||||
ABS2 m4, m5, m6, m7 ; abs(q3-q0) | abs(q2-q0)
|
||||
pcmpgtw m4, reg_F ; abs(q3-q0) > F
|
||||
pcmpgtw m5, reg_F ; abs(q2-q0) > F
|
||||
%endif
|
||||
psubw m3, m2 ; q3-q2
|
||||
psubw m2, m1 ; q2-q1
|
||||
ABS2 m3, m2, m6, m7 ; abs(q3-q2) | abs(q2-q1)
|
||||
pcmpgtw m3, reg_I ; abs(q3-q2) > I
|
||||
pcmpgtw m2, reg_I ; abs(q2-q1) > I
|
||||
%if %1 > 4
|
||||
por m4, m5
|
||||
%endif
|
||||
por m2, m3
|
||||
psubw m3, m1, m0 ; q1-q0
|
||||
ABS1 m3, m5 ; abs(q1-q0)
|
||||
%if %1 > 4
|
||||
pcmpgtw m6, m3, reg_F ; abs(q1-q0) > F
|
||||
%endif
|
||||
pcmpgtw m7, m3, reg_H ; abs(q1-q0) > H
|
||||
pcmpgtw m3, reg_I ; abs(q1-q0) > I
|
||||
%if %1 > 4
|
||||
por m4, m6
|
||||
%endif
|
||||
por m2, m3
|
||||
%endmacro
|
||||
|
||||
; one step in filter_14/filter_6
|
||||
;
|
||||
; take sum $reg, downshift, apply mask and write into dst
|
||||
;
|
||||
; if sub2/add1-2 are present, add/sub as appropriate to prepare for the next
|
||||
; step's sum $reg. This is omitted for the last row in each filter.
|
||||
;
|
||||
; if dont_store is set, don't write the result into memory, instead keep the
|
||||
; values in register so we can write it out later
|
||||
%macro FILTER_STEP 6-10 "", "", "", 0 ; tmp, reg, mask, shift, dst, \
|
||||
; src/sub1, sub2, add1, add2, dont_store
|
||||
psrlw %1, %2, %4
|
||||
psubw %1, %6 ; abs->delta
|
||||
%ifnidn %7, ""
|
||||
psubw %2, %6
|
||||
psubw %2, %7
|
||||
paddw %2, %8
|
||||
paddw %2, %9
|
||||
%endif
|
||||
pand %1, reg_%3 ; apply mask
|
||||
%if %10 == 1
|
||||
paddw %6, %1 ; delta->abs
|
||||
%else
|
||||
paddw %1, %6 ; delta->abs
|
||||
mova [%5], %1
|
||||
%endif
|
||||
%endmacro
|
||||
|
||||
; FIXME avx2 versions for 16_16 and mix2_{4,8}{4,8}
|
||||
|
||||
%macro LOOP_FILTER 3 ; dir[h/v], wd[4/8/16], bpp[10/12]
|
||||
|
||||
%if ARCH_X86_64
|
||||
%if %2 == 16
|
||||
%assign %%num_xmm_regs 16
|
||||
%elif %2 == 8
|
||||
%assign %%num_xmm_regs 15
|
||||
%else ; %2 == 4
|
||||
%assign %%num_xmm_regs 14
|
||||
%endif ; %2
|
||||
%assign %%bak_mem 0
|
||||
%else ; ARCH_X86_32
|
||||
%assign %%num_xmm_regs 8
|
||||
%if %2 == 16
|
||||
%assign %%bak_mem 7
|
||||
%elif %2 == 8
|
||||
%assign %%bak_mem 6
|
||||
%else ; %2 == 4
|
||||
%assign %%bak_mem 5
|
||||
%endif ; %2
|
||||
%endif ; ARCH_X86_64/32
|
||||
|
||||
%if %2 == 16
|
||||
%ifidn %1, v
|
||||
%assign %%num_gpr_regs 6
|
||||
%else ; %1 == h
|
||||
%assign %%num_gpr_regs 5
|
||||
%endif ; %1
|
||||
%assign %%wd_mem 6
|
||||
%else ; %2 == 8/4
|
||||
%assign %%num_gpr_regs 5
|
||||
%if ARCH_X86_32 && %2 == 8
|
||||
%assign %%wd_mem 2
|
||||
%else ; ARCH_X86_64 || %2 == 4
|
||||
%assign %%wd_mem 0
|
||||
%endif ; ARCH_X86_64/32 etc.
|
||||
%endif ; %2
|
||||
|
||||
%ifidn %1, v
|
||||
%assign %%tsp_mem 0
|
||||
%elif %2 == 16 ; && %1 == h
|
||||
%assign %%tsp_mem 16
|
||||
%else ; %1 == h && %1 == 8/4
|
||||
%assign %%tsp_mem 8
|
||||
%endif ; %1/%2
|
||||
|
||||
%assign %%off %%wd_mem
|
||||
%assign %%tspoff %%bak_mem+%%wd_mem
|
||||
%assign %%stack_mem ((%%bak_mem+%%wd_mem+%%tsp_mem)*mmsize)
|
||||
|
||||
%if %3 == 10
|
||||
%define %%maxsgn 511
|
||||
%define %%minsgn m512
|
||||
%define %%maxusgn 1023
|
||||
%define %%maxf 4
|
||||
%else ; %3 == 12
|
||||
%define %%maxsgn 2047
|
||||
%define %%minsgn m2048
|
||||
%define %%maxusgn 4095
|
||||
%define %%maxf 16
|
||||
%endif ; %3
|
||||
|
||||
cglobal vp9_loop_filter_%1_%2_%3, 5, %%num_gpr_regs, %%num_xmm_regs, %%stack_mem, dst, stride, E, I, H
|
||||
; prepare E, I and H masks
|
||||
shl Ed, %3-8
|
||||
shl Id, %3-8
|
||||
shl Hd, %3-8
|
||||
%if cpuflag(ssse3)
|
||||
mova m0, [pw_256]
|
||||
%endif
|
||||
movd m1, Ed
|
||||
movd m2, Id
|
||||
movd m3, Hd
|
||||
%if cpuflag(ssse3)
|
||||
pshufb m1, m0 ; E << (bit_depth - 8)
|
||||
pshufb m2, m0 ; I << (bit_depth - 8)
|
||||
pshufb m3, m0 ; H << (bit_depth - 8)
|
||||
%else
|
||||
punpcklwd m1, m1
|
||||
punpcklwd m2, m2
|
||||
punpcklwd m3, m3
|
||||
pshufd m1, m1, q0000
|
||||
pshufd m2, m2, q0000
|
||||
pshufd m3, m3, q0000
|
||||
%endif
|
||||
SCRATCH 1, 8, rsp+(%%off+0)*mmsize, E
|
||||
SCRATCH 2, 9, rsp+(%%off+1)*mmsize, I
|
||||
SCRATCH 3, 10, rsp+(%%off+2)*mmsize, H
|
||||
%if %2 > 4
|
||||
PRELOAD 11, pw_ %+ %%maxf, F
|
||||
%endif
|
||||
|
||||
; set up variables to load data
|
||||
%ifidn %1, v
|
||||
DEFINE_ARGS dst8, stride, stride3, dst0, dst4, dst12
|
||||
lea stride3q, [strideq*3]
|
||||
neg strideq
|
||||
%if %2 == 16
|
||||
lea dst0q, [dst8q+strideq*8]
|
||||
%else
|
||||
lea dst4q, [dst8q+strideq*4]
|
||||
%endif
|
||||
neg strideq
|
||||
%if %2 == 16
|
||||
lea dst12q, [dst8q+strideq*4]
|
||||
lea dst4q, [dst0q+strideq*4]
|
||||
%endif
|
||||
|
||||
%if %2 == 16
|
||||
%define %%p7 dst0q
|
||||
%define %%p6 dst0q+strideq
|
||||
%define %%p5 dst0q+strideq*2
|
||||
%define %%p4 dst0q+stride3q
|
||||
%endif
|
||||
%define %%p3 dst4q
|
||||
%define %%p2 dst4q+strideq
|
||||
%define %%p1 dst4q+strideq*2
|
||||
%define %%p0 dst4q+stride3q
|
||||
%define %%q0 dst8q
|
||||
%define %%q1 dst8q+strideq
|
||||
%define %%q2 dst8q+strideq*2
|
||||
%define %%q3 dst8q+stride3q
|
||||
%if %2 == 16
|
||||
%define %%q4 dst12q
|
||||
%define %%q5 dst12q+strideq
|
||||
%define %%q6 dst12q+strideq*2
|
||||
%define %%q7 dst12q+stride3q
|
||||
%endif
|
||||
%else ; %1 == h
|
||||
DEFINE_ARGS dst0, stride, stride3, dst4
|
||||
lea stride3q, [strideq*3]
|
||||
lea dst4q, [dst0q+strideq*4]
|
||||
|
||||
%define %%p3 rsp+(%%tspoff+0)*mmsize
|
||||
%define %%p2 rsp+(%%tspoff+1)*mmsize
|
||||
%define %%p1 rsp+(%%tspoff+2)*mmsize
|
||||
%define %%p0 rsp+(%%tspoff+3)*mmsize
|
||||
%define %%q0 rsp+(%%tspoff+4)*mmsize
|
||||
%define %%q1 rsp+(%%tspoff+5)*mmsize
|
||||
%define %%q2 rsp+(%%tspoff+6)*mmsize
|
||||
%define %%q3 rsp+(%%tspoff+7)*mmsize
|
||||
|
||||
%if %2 < 16
|
||||
movu m0, [dst0q+strideq*0-8]
|
||||
movu m1, [dst0q+strideq*1-8]
|
||||
movu m2, [dst0q+strideq*2-8]
|
||||
movu m3, [dst0q+stride3q -8]
|
||||
movu m4, [dst4q+strideq*0-8]
|
||||
movu m5, [dst4q+strideq*1-8]
|
||||
movu m6, [dst4q+strideq*2-8]
|
||||
movu m7, [dst4q+stride3q -8]
|
||||
|
||||
%if ARCH_X86_64
|
||||
TRANSPOSE8x8W 0, 1, 2, 3, 4, 5, 6, 7, 12
|
||||
%else
|
||||
TRANSPOSE8x8W 0, 1, 2, 3, 4, 5, 6, 7, [%%p0], [%%q0]
|
||||
%endif
|
||||
|
||||
mova [%%p3], m0
|
||||
mova [%%p2], m1
|
||||
mova [%%p1], m2
|
||||
mova [%%p0], m3
|
||||
%if ARCH_X86_64
|
||||
mova [%%q0], m4
|
||||
%endif
|
||||
mova [%%q1], m5
|
||||
mova [%%q2], m6
|
||||
mova [%%q3], m7
|
||||
|
||||
; FIXME investigate if we can _not_ load q0-3 below if h, and adjust register
|
||||
; order here accordingly
|
||||
%else ; %2 == 16
|
||||
|
||||
%define %%p7 rsp+(%%tspoff+ 8)*mmsize
|
||||
%define %%p6 rsp+(%%tspoff+ 9)*mmsize
|
||||
%define %%p5 rsp+(%%tspoff+10)*mmsize
|
||||
%define %%p4 rsp+(%%tspoff+11)*mmsize
|
||||
%define %%q4 rsp+(%%tspoff+12)*mmsize
|
||||
%define %%q5 rsp+(%%tspoff+13)*mmsize
|
||||
%define %%q6 rsp+(%%tspoff+14)*mmsize
|
||||
%define %%q7 rsp+(%%tspoff+15)*mmsize
|
||||
|
||||
mova m0, [dst0q+strideq*0-16]
|
||||
mova m1, [dst0q+strideq*1-16]
|
||||
mova m2, [dst0q+strideq*2-16]
|
||||
mova m3, [dst0q+stride3q -16]
|
||||
mova m4, [dst4q+strideq*0-16]
|
||||
mova m5, [dst4q+strideq*1-16]
|
||||
%if ARCH_X86_64
|
||||
mova m6, [dst4q+strideq*2-16]
|
||||
%endif
|
||||
mova m7, [dst4q+stride3q -16]
|
||||
|
||||
%if ARCH_X86_64
|
||||
TRANSPOSE8x8W 0, 1, 2, 3, 4, 5, 6, 7, 12
|
||||
%else
|
||||
TRANSPOSE8x8W 0, 1, 2, 3, 4, 5, 6, 7, [dst4q+strideq*2-16], [%%p3], 1
|
||||
%endif
|
||||
|
||||
mova [%%p7], m0
|
||||
mova [%%p6], m1
|
||||
mova [%%p5], m2
|
||||
mova [%%p4], m3
|
||||
%if ARCH_X86_64
|
||||
mova [%%p3], m4
|
||||
%endif
|
||||
mova [%%p2], m5
|
||||
mova [%%p1], m6
|
||||
mova [%%p0], m7
|
||||
|
||||
mova m0, [dst0q+strideq*0]
|
||||
mova m1, [dst0q+strideq*1]
|
||||
mova m2, [dst0q+strideq*2]
|
||||
mova m3, [dst0q+stride3q ]
|
||||
mova m4, [dst4q+strideq*0]
|
||||
mova m5, [dst4q+strideq*1]
|
||||
%if ARCH_X86_64
|
||||
mova m6, [dst4q+strideq*2]
|
||||
%endif
|
||||
mova m7, [dst4q+stride3q ]
|
||||
|
||||
%if ARCH_X86_64
|
||||
TRANSPOSE8x8W 0, 1, 2, 3, 4, 5, 6, 7, 12
|
||||
%else
|
||||
TRANSPOSE8x8W 0, 1, 2, 3, 4, 5, 6, 7, [dst4q+strideq*2], [%%q4], 1
|
||||
%endif
|
||||
|
||||
mova [%%q0], m0
|
||||
mova [%%q1], m1
|
||||
mova [%%q2], m2
|
||||
mova [%%q3], m3
|
||||
%if ARCH_X86_64
|
||||
mova [%%q4], m4
|
||||
%endif
|
||||
mova [%%q5], m5
|
||||
mova [%%q6], m6
|
||||
mova [%%q7], m7
|
||||
|
||||
; FIXME investigate if we can _not_ load q0|q4-7 below if h, and adjust register
|
||||
; order here accordingly
|
||||
%endif ; %2
|
||||
%endif ; %1
|
||||
|
||||
; load q0|q4-7 data
|
||||
mova m0, [%%q0]
|
||||
%if %2 == 16
|
||||
mova m4, [%%q4]
|
||||
mova m5, [%%q5]
|
||||
mova m6, [%%q6]
|
||||
mova m7, [%%q7]
|
||||
|
||||
; flat8out q portion
|
||||
FLAT8OUT_HALF
|
||||
SCRATCH 7, 15, rsp+(%%off+6)*mmsize, F8O
|
||||
%endif
|
||||
|
||||
; load q1-3 data
|
||||
mova m1, [%%q1]
|
||||
mova m2, [%%q2]
|
||||
mova m3, [%%q3]
|
||||
|
||||
; r6-8|pw_4[m8-11]=reg_E/I/H/F
|
||||
; r9[m15]=!flatout[q]
|
||||
; m12-14=free
|
||||
; m0-3=q0-q3
|
||||
; m4-7=free
|
||||
|
||||
; flat8in|fm|hev q portion
|
||||
FLAT8IN_HALF %2
|
||||
SCRATCH 7, 13, rsp+(%%off+4)*mmsize, HEV
|
||||
%if %2 > 4
|
||||
SCRATCH 4, 14, rsp+(%%off+5)*mmsize, F8I
|
||||
%endif
|
||||
|
||||
; r6-8|pw_4[m8-11]=reg_E/I/H/F
|
||||
; r9[m15]=!flat8out[q]
|
||||
; r10[m13]=hev[q]
|
||||
; r11[m14]=!flat8in[q]
|
||||
; m2=!fm[q]
|
||||
; m0,1=q0-q1
|
||||
; m2-7=free
|
||||
; m12=free
|
||||
|
||||
; load p0-1
|
||||
mova m3, [%%p0]
|
||||
mova m4, [%%p1]
|
||||
|
||||
; fm mb_edge portion
|
||||
psubw m5, m3, m0 ; q0-p0
|
||||
psubw m6, m4, m1 ; q1-p1
|
||||
%if ARCH_X86_64
|
||||
ABS2 m5, m6, m7, m12 ; abs(q0-p0) | abs(q1-p1)
|
||||
%else
|
||||
ABS1 m5, m7 ; abs(q0-p0)
|
||||
ABS1 m6, m7 ; abs(q1-p1)
|
||||
%endif
|
||||
paddw m5, m5
|
||||
psraw m6, 1
|
||||
paddw m6, m5 ; abs(q0-p0)*2+(abs(q1-p1)>>1)
|
||||
pcmpgtw m6, reg_E
|
||||
por m2, m6
|
||||
SCRATCH 2, 12, rsp+(%%off+3)*mmsize, FM
|
||||
|
||||
; r6-8|pw_4[m8-11]=reg_E/I/H/F
|
||||
; r9[m15]=!flat8out[q]
|
||||
; r10[m13]=hev[q]
|
||||
; r11[m14]=!flat8in[q]
|
||||
; r12[m12]=!fm[q]
|
||||
; m3-4=q0-1
|
||||
; m0-2/5-7=free
|
||||
|
||||
; load p4-7 data
|
||||
SWAP 3, 0 ; p0
|
||||
SWAP 4, 1 ; p1
|
||||
%if %2 == 16
|
||||
mova m7, [%%p7]
|
||||
mova m6, [%%p6]
|
||||
mova m5, [%%p5]
|
||||
mova m4, [%%p4]
|
||||
|
||||
; flat8out p portion
|
||||
FLAT8OUT_HALF
|
||||
por m7, reg_F8O
|
||||
SCRATCH 7, 15, rsp+(%%off+6)*mmsize, F8O
|
||||
%endif
|
||||
|
||||
; r6-8|pw_4[m8-11]=reg_E/I/H/F
|
||||
; r9[m15]=!flat8out
|
||||
; r10[m13]=hev[q]
|
||||
; r11[m14]=!flat8in[q]
|
||||
; r12[m12]=!fm[q]
|
||||
; m0=p0
|
||||
; m1-7=free
|
||||
|
||||
; load p2-3 data
|
||||
mova m2, [%%p2]
|
||||
mova m3, [%%p3]
|
||||
|
||||
; flat8in|fm|hev p portion
|
||||
FLAT8IN_HALF %2
|
||||
por m7, reg_HEV
|
||||
%if %2 > 4
|
||||
por m4, reg_F8I
|
||||
%endif
|
||||
por m2, reg_FM
|
||||
%if %2 > 4
|
||||
por m4, m2 ; !flat8|!fm
|
||||
%if %2 == 16
|
||||
por m5, m4, reg_F8O ; !flat16|!fm
|
||||
pandn m2, m4 ; filter4_mask
|
||||
pandn m4, m5 ; filter8_mask
|
||||
pxor m5, [pw_m1] ; filter16_mask
|
||||
SCRATCH 5, 15, rsp+(%%off+6)*mmsize, F16M
|
||||
%else
|
||||
pandn m2, m4 ; filter4_mask
|
||||
pxor m4, [pw_m1] ; filter8_mask
|
||||
%endif
|
||||
SCRATCH 4, 14, rsp+(%%off+5)*mmsize, F8M
|
||||
%else
|
||||
pxor m2, [pw_m1] ; filter4_mask
|
||||
%endif
|
||||
SCRATCH 7, 13, rsp+(%%off+4)*mmsize, HEV
|
||||
SCRATCH 2, 12, rsp+(%%off+3)*mmsize, F4M
|
||||
|
||||
; r9[m15]=filter16_mask
|
||||
; r10[m13]=hev
|
||||
; r11[m14]=filter8_mask
|
||||
; r12[m12]=filter4_mask
|
||||
; m0,1=p0-p1
|
||||
; m2-7=free
|
||||
; m8-11=free
|
||||
|
||||
%if %2 > 4
|
||||
%if %2 == 16
|
||||
; filter_14
|
||||
mova m2, [%%p7]
|
||||
mova m3, [%%p6]
|
||||
mova m6, [%%p5]
|
||||
mova m7, [%%p4]
|
||||
PRELOAD 8, %%p3, P3
|
||||
PRELOAD 9, %%p2, P2
|
||||
%endif
|
||||
PRELOAD 10, %%q0, Q0
|
||||
PRELOAD 11, %%q1, Q1
|
||||
%if %2 == 16
|
||||
psllw m4, m2, 3
|
||||
paddw m5, m3, m3
|
||||
paddw m4, m6
|
||||
paddw m5, m7
|
||||
paddw m4, reg_P3
|
||||
paddw m5, reg_P2
|
||||
paddw m4, m1
|
||||
paddw m5, m0
|
||||
paddw m4, reg_Q0 ; q0+p1+p3+p5+p7*8
|
||||
psubw m5, m2 ; p0+p2+p4+p6*2-p7
|
||||
paddw m4, [pw_8]
|
||||
paddw m5, m4 ; q0+p0+p1+p2+p3+p4+p5+p6*2+p7*7+8
|
||||
|
||||
; below, we use r0-5 for storing pre-filter pixels for subsequent subtraction
|
||||
; at the end of the filter
|
||||
|
||||
mova [rsp+0*mmsize], m3
|
||||
FILTER_STEP m4, m5, F16M, 4, %%p6, m3, m2, m6, reg_Q1
|
||||
%endif
|
||||
mova m3, [%%q2]
|
||||
%if %2 == 16
|
||||
mova [rsp+1*mmsize], m6
|
||||
FILTER_STEP m4, m5, F16M, 4, %%p5, m6, m2, m7, m3
|
||||
%endif
|
||||
mova m6, [%%q3]
|
||||
%if %2 == 16
|
||||
mova [rsp+2*mmsize], m7
|
||||
FILTER_STEP m4, m5, F16M, 4, %%p4, m7, m2, reg_P3, m6
|
||||
mova m7, [%%q4]
|
||||
%if ARCH_X86_64
|
||||
mova [rsp+3*mmsize], reg_P3
|
||||
%else
|
||||
mova m4, reg_P3
|
||||
mova [rsp+3*mmsize], m4
|
||||
%endif
|
||||
FILTER_STEP m4, m5, F16M, 4, %%p3, reg_P3, m2, reg_P2, m7
|
||||
PRELOAD 8, %%q5, Q5
|
||||
%if ARCH_X86_64
|
||||
mova [rsp+4*mmsize], reg_P2
|
||||
%else
|
||||
mova m4, reg_P2
|
||||
mova [rsp+4*mmsize], m4
|
||||
%endif
|
||||
FILTER_STEP m4, m5, F16M, 4, %%p2, reg_P2, m2, m1, reg_Q5
|
||||
PRELOAD 9, %%q6, Q6
|
||||
mova [rsp+5*mmsize], m1
|
||||
FILTER_STEP m4, m5, F16M, 4, %%p1, m1, m2, m0, reg_Q6
|
||||
mova m1, [%%q7]
|
||||
FILTER_STEP m4, m5, F16M, 4, %%p0, m0, m2, reg_Q0, m1, 1
|
||||
FILTER_STEP m4, m5, F16M, 4, %%q0, reg_Q0, [rsp+0*mmsize], reg_Q1, m1, ARCH_X86_64
|
||||
FILTER_STEP m4, m5, F16M, 4, %%q1, reg_Q1, [rsp+1*mmsize], m3, m1, ARCH_X86_64
|
||||
FILTER_STEP m4, m5, F16M, 4, %%q2, m3, [rsp+2*mmsize], m6, m1, 1
|
||||
FILTER_STEP m4, m5, F16M, 4, %%q3, m6, [rsp+3*mmsize], m7, m1
|
||||
FILTER_STEP m4, m5, F16M, 4, %%q4, m7, [rsp+4*mmsize], reg_Q5, m1
|
||||
FILTER_STEP m4, m5, F16M, 4, %%q5, reg_Q5, [rsp+5*mmsize], reg_Q6, m1
|
||||
FILTER_STEP m4, m5, F16M, 4, %%q6, reg_Q6
|
||||
|
||||
mova m7, [%%p1]
|
||||
%else
|
||||
SWAP 1, 7
|
||||
%endif
|
||||
|
||||
mova m2, [%%p3]
|
||||
mova m1, [%%p2]
|
||||
|
||||
; reg_Q0-1 (m10-m11)
|
||||
; m0=p0
|
||||
; m1=p2
|
||||
; m2=p3
|
||||
; m3=q2
|
||||
; m4-5=free
|
||||
; m6=q3
|
||||
; m7=p1
|
||||
; m8-9 unused
|
||||
|
||||
; filter_6
|
||||
psllw m4, m2, 2
|
||||
paddw m5, m1, m1
|
||||
paddw m4, m7
|
||||
psubw m5, m2
|
||||
paddw m4, m0
|
||||
paddw m5, reg_Q0
|
||||
paddw m4, [pw_4]
|
||||
paddw m5, m4
|
||||
|
||||
%if ARCH_X86_64
|
||||
mova m8, m1
|
||||
mova m9, m7
|
||||
%else
|
||||
mova [rsp+0*mmsize], m1
|
||||
mova [rsp+1*mmsize], m7
|
||||
%endif
|
||||
%ifidn %1, v
|
||||
FILTER_STEP m4, m5, F8M, 3, %%p2, m1, m2, m7, reg_Q1
|
||||
%else
|
||||
FILTER_STEP m4, m5, F8M, 3, %%p2, m1, m2, m7, reg_Q1, 1
|
||||
%endif
|
||||
FILTER_STEP m4, m5, F8M, 3, %%p1, m7, m2, m0, m3, 1
|
||||
FILTER_STEP m4, m5, F8M, 3, %%p0, m0, m2, reg_Q0, m6, 1
|
||||
%if ARCH_X86_64
|
||||
FILTER_STEP m4, m5, F8M, 3, %%q0, reg_Q0, m8, reg_Q1, m6, ARCH_X86_64
|
||||
FILTER_STEP m4, m5, F8M, 3, %%q1, reg_Q1, m9, m3, m6, ARCH_X86_64
|
||||
%else
|
||||
FILTER_STEP m4, m5, F8M, 3, %%q0, reg_Q0, [rsp+0*mmsize], reg_Q1, m6, ARCH_X86_64
|
||||
FILTER_STEP m4, m5, F8M, 3, %%q1, reg_Q1, [rsp+1*mmsize], m3, m6, ARCH_X86_64
|
||||
%endif
|
||||
FILTER_STEP m4, m5, F8M, 3, %%q2, m3
|
||||
|
||||
UNSCRATCH 2, 10, %%q0
|
||||
UNSCRATCH 6, 11, %%q1
|
||||
%else
|
||||
SWAP 1, 7
|
||||
mova m2, [%%q0]
|
||||
mova m6, [%%q1]
|
||||
%endif
|
||||
UNSCRATCH 3, 13, rsp+(%%off+4)*mmsize, HEV
|
||||
|
||||
; m0=p0
|
||||
; m1=p2
|
||||
; m2=q0
|
||||
; m3=hev_mask
|
||||
; m4-5=free
|
||||
; m6=q1
|
||||
; m7=p1
|
||||
|
||||
; filter_4
|
||||
psubw m4, m7, m6 ; p1-q1
|
||||
psubw m5, m2, m0 ; q0-p0
|
||||
pand m4, m3
|
||||
pminsw m4, [pw_ %+ %%maxsgn]
|
||||
pmaxsw m4, [pw_ %+ %%minsgn] ; clip_intp2(p1-q1, 9) -> f
|
||||
paddw m4, m5
|
||||
paddw m5, m5
|
||||
paddw m4, m5 ; 3*(q0-p0)+f
|
||||
pminsw m4, [pw_ %+ %%maxsgn]
|
||||
pmaxsw m4, [pw_ %+ %%minsgn] ; clip_intp2(3*(q0-p0)+f, 9) -> f
|
||||
pand m4, reg_F4M
|
||||
paddw m5, m4, [pw_4]
|
||||
paddw m4, [pw_3]
|
||||
pminsw m5, [pw_ %+ %%maxsgn]
|
||||
pminsw m4, [pw_ %+ %%maxsgn]
|
||||
psraw m5, 3 ; min_intp2(f+4, 9)>>3 -> f1
|
||||
psraw m4, 3 ; min_intp2(f+3, 9)>>3 -> f2
|
||||
psubw m2, m5 ; q0-f1
|
||||
paddw m0, m4 ; p0+f2
|
||||
pandn m3, m5 ; f1 & !hev (for p1/q1 adj)
|
||||
pxor m4, m4
|
||||
mova m5, [pw_ %+ %%maxusgn]
|
||||
pmaxsw m2, m4
|
||||
pmaxsw m0, m4
|
||||
pminsw m2, m5
|
||||
pminsw m0, m5
|
||||
%if cpuflag(ssse3)
|
||||
pmulhrsw m3, [pw_16384] ; (f1+1)>>1
|
||||
%else
|
||||
paddw m3, [pw_1]
|
||||
psraw m3, 1
|
||||
%endif
|
||||
paddw m7, m3 ; p1+f
|
||||
psubw m6, m3 ; q1-f
|
||||
pmaxsw m7, m4
|
||||
pmaxsw m6, m4
|
||||
pminsw m7, m5
|
||||
pminsw m6, m5
|
||||
|
||||
; store
|
||||
%ifidn %1, v
|
||||
mova [%%p1], m7
|
||||
mova [%%p0], m0
|
||||
mova [%%q0], m2
|
||||
mova [%%q1], m6
|
||||
%else ; %1 == h
|
||||
%if %2 == 4
|
||||
TRANSPOSE4x4W 7, 0, 2, 6, 1
|
||||
movh [dst0q+strideq*0-4], m7
|
||||
movhps [dst0q+strideq*1-4], m7
|
||||
movh [dst0q+strideq*2-4], m0
|
||||
movhps [dst0q+stride3q -4], m0
|
||||
movh [dst4q+strideq*0-4], m2
|
||||
movhps [dst4q+strideq*1-4], m2
|
||||
movh [dst4q+strideq*2-4], m6
|
||||
movhps [dst4q+stride3q -4], m6
|
||||
%elif %2 == 8
|
||||
mova m3, [%%p3]
|
||||
mova m4, [%%q2]
|
||||
mova m5, [%%q3]
|
||||
|
||||
%if ARCH_X86_64
|
||||
TRANSPOSE8x8W 3, 1, 7, 0, 2, 6, 4, 5, 8
|
||||
%else
|
||||
TRANSPOSE8x8W 3, 1, 7, 0, 2, 6, 4, 5, [%%q2], [%%q0], 1
|
||||
mova m2, [%%q0]
|
||||
%endif
|
||||
|
||||
movu [dst0q+strideq*0-8], m3
|
||||
movu [dst0q+strideq*1-8], m1
|
||||
movu [dst0q+strideq*2-8], m7
|
||||
movu [dst0q+stride3q -8], m0
|
||||
movu [dst4q+strideq*0-8], m2
|
||||
movu [dst4q+strideq*1-8], m6
|
||||
movu [dst4q+strideq*2-8], m4
|
||||
movu [dst4q+stride3q -8], m5
|
||||
%else ; %2 == 16
|
||||
SCRATCH 2, 8, %%q0
|
||||
SCRATCH 6, 9, %%q1
|
||||
mova m2, [%%p7]
|
||||
mova m3, [%%p6]
|
||||
mova m4, [%%p5]
|
||||
mova m5, [%%p4]
|
||||
mova m6, [%%p3]
|
||||
|
||||
%if ARCH_X86_64
|
||||
TRANSPOSE8x8W 2, 3, 4, 5, 6, 1, 7, 0, 10
|
||||
%else
|
||||
mova [%%p1], m7
|
||||
TRANSPOSE8x8W 2, 3, 4, 5, 6, 1, 7, 0, [%%p1], [dst4q+strideq*0-16], 1
|
||||
%endif
|
||||
|
||||
mova [dst0q+strideq*0-16], m2
|
||||
mova [dst0q+strideq*1-16], m3
|
||||
mova [dst0q+strideq*2-16], m4
|
||||
mova [dst0q+stride3q -16], m5
|
||||
%if ARCH_X86_64
|
||||
mova [dst4q+strideq*0-16], m6
|
||||
%endif
|
||||
mova [dst4q+strideq*1-16], m1
|
||||
mova [dst4q+strideq*2-16], m7
|
||||
mova [dst4q+stride3q -16], m0
|
||||
|
||||
UNSCRATCH 2, 8, %%q0
|
||||
UNSCRATCH 6, 9, %%q1
|
||||
mova m0, [%%q2]
|
||||
mova m1, [%%q3]
|
||||
mova m3, [%%q4]
|
||||
mova m4, [%%q5]
|
||||
%if ARCH_X86_64
|
||||
mova m5, [%%q6]
|
||||
%endif
|
||||
mova m7, [%%q7]
|
||||
|
||||
%if ARCH_X86_64
|
||||
TRANSPOSE8x8W 2, 6, 0, 1, 3, 4, 5, 7, 8
|
||||
%else
|
||||
TRANSPOSE8x8W 2, 6, 0, 1, 3, 4, 5, 7, [%%q6], [dst4q+strideq*0], 1
|
||||
%endif
|
||||
|
||||
mova [dst0q+strideq*0], m2
|
||||
mova [dst0q+strideq*1], m6
|
||||
mova [dst0q+strideq*2], m0
|
||||
mova [dst0q+stride3q ], m1
|
||||
%if ARCH_X86_64
|
||||
mova [dst4q+strideq*0], m3
|
||||
%endif
|
||||
mova [dst4q+strideq*1], m4
|
||||
mova [dst4q+strideq*2], m5
|
||||
mova [dst4q+stride3q ], m7
|
||||
%endif ; %2
|
||||
%endif ; %1
|
||||
RET
|
||||
%endmacro
|
||||
|
||||
%macro LOOP_FILTER_CPUSETS 3
|
||||
INIT_XMM sse2
|
||||
LOOP_FILTER %1, %2, %3
|
||||
INIT_XMM ssse3
|
||||
LOOP_FILTER %1, %2, %3
|
||||
INIT_XMM avx
|
||||
LOOP_FILTER %1, %2, %3
|
||||
%endmacro
|
||||
|
||||
%macro LOOP_FILTER_WDSETS 2
|
||||
LOOP_FILTER_CPUSETS %1, 4, %2
|
||||
LOOP_FILTER_CPUSETS %1, 8, %2
|
||||
LOOP_FILTER_CPUSETS %1, 16, %2
|
||||
%endmacro
|
||||
|
||||
LOOP_FILTER_WDSETS h, 10
|
||||
LOOP_FILTER_WDSETS v, 10
|
||||
LOOP_FILTER_WDSETS h, 12
|
||||
LOOP_FILTER_WDSETS v, 12
|
||||
676
media/ffvpx/libavcodec/x86/vp9mc.asm
Normal file
676
media/ffvpx/libavcodec/x86/vp9mc.asm
Normal file
|
|
@ -0,0 +1,676 @@
|
|||
;******************************************************************************
|
||||
;* VP9 MC SIMD optimizations
|
||||
;*
|
||||
;* Copyright (c) 2013 Ronald S. Bultje <rsbultje gmail com>
|
||||
;*
|
||||
;* This file is part of FFmpeg.
|
||||
;*
|
||||
;* FFmpeg is free software; you can redistribute it and/or
|
||||
;* modify it under the terms of the GNU Lesser General Public
|
||||
;* License as published by the Free Software Foundation; either
|
||||
;* version 2.1 of the License, or (at your option) any later version.
|
||||
;*
|
||||
;* FFmpeg is distributed in the hope that it will be useful,
|
||||
;* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
;* Lesser General Public License for more details.
|
||||
;*
|
||||
;* You should have received a copy of the GNU Lesser General Public
|
||||
;* License along with FFmpeg; if not, write to the Free Software
|
||||
;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
;******************************************************************************
|
||||
|
||||
%include "libavutil/x86/x86util.asm"
|
||||
|
||||
SECTION_RODATA 32
|
||||
|
||||
cextern pw_256
|
||||
cextern pw_64
|
||||
|
||||
%macro F8_SSSE3_TAPS 8
|
||||
times 16 db %1, %2
|
||||
times 16 db %3, %4
|
||||
times 16 db %5, %6
|
||||
times 16 db %7, %8
|
||||
%endmacro
|
||||
|
||||
%macro F8_SSE2_TAPS 8
|
||||
times 8 dw %1
|
||||
times 8 dw %2
|
||||
times 8 dw %3
|
||||
times 8 dw %4
|
||||
times 8 dw %5
|
||||
times 8 dw %6
|
||||
times 8 dw %7
|
||||
times 8 dw %8
|
||||
%endmacro
|
||||
|
||||
%macro F8_16BPP_TAPS 8
|
||||
times 8 dw %1, %2
|
||||
times 8 dw %3, %4
|
||||
times 8 dw %5, %6
|
||||
times 8 dw %7, %8
|
||||
%endmacro
|
||||
|
||||
%macro FILTER 1
|
||||
const filters_%1 ; smooth
|
||||
F8_TAPS -3, -1, 32, 64, 38, 1, -3, 0
|
||||
F8_TAPS -2, -2, 29, 63, 41, 2, -3, 0
|
||||
F8_TAPS -2, -2, 26, 63, 43, 4, -4, 0
|
||||
F8_TAPS -2, -3, 24, 62, 46, 5, -4, 0
|
||||
F8_TAPS -2, -3, 21, 60, 49, 7, -4, 0
|
||||
F8_TAPS -1, -4, 18, 59, 51, 9, -4, 0
|
||||
F8_TAPS -1, -4, 16, 57, 53, 12, -4, -1
|
||||
F8_TAPS -1, -4, 14, 55, 55, 14, -4, -1
|
||||
F8_TAPS -1, -4, 12, 53, 57, 16, -4, -1
|
||||
F8_TAPS 0, -4, 9, 51, 59, 18, -4, -1
|
||||
F8_TAPS 0, -4, 7, 49, 60, 21, -3, -2
|
||||
F8_TAPS 0, -4, 5, 46, 62, 24, -3, -2
|
||||
F8_TAPS 0, -4, 4, 43, 63, 26, -2, -2
|
||||
F8_TAPS 0, -3, 2, 41, 63, 29, -2, -2
|
||||
F8_TAPS 0, -3, 1, 38, 64, 32, -1, -3
|
||||
; regular
|
||||
F8_TAPS 0, 1, -5, 126, 8, -3, 1, 0
|
||||
F8_TAPS -1, 3, -10, 122, 18, -6, 2, 0
|
||||
F8_TAPS -1, 4, -13, 118, 27, -9, 3, -1
|
||||
F8_TAPS -1, 4, -16, 112, 37, -11, 4, -1
|
||||
F8_TAPS -1, 5, -18, 105, 48, -14, 4, -1
|
||||
F8_TAPS -1, 5, -19, 97, 58, -16, 5, -1
|
||||
F8_TAPS -1, 6, -19, 88, 68, -18, 5, -1
|
||||
F8_TAPS -1, 6, -19, 78, 78, -19, 6, -1
|
||||
F8_TAPS -1, 5, -18, 68, 88, -19, 6, -1
|
||||
F8_TAPS -1, 5, -16, 58, 97, -19, 5, -1
|
||||
F8_TAPS -1, 4, -14, 48, 105, -18, 5, -1
|
||||
F8_TAPS -1, 4, -11, 37, 112, -16, 4, -1
|
||||
F8_TAPS -1, 3, -9, 27, 118, -13, 4, -1
|
||||
F8_TAPS 0, 2, -6, 18, 122, -10, 3, -1
|
||||
F8_TAPS 0, 1, -3, 8, 126, -5, 1, 0
|
||||
; sharp
|
||||
F8_TAPS -1, 3, -7, 127, 8, -3, 1, 0
|
||||
F8_TAPS -2, 5, -13, 125, 17, -6, 3, -1
|
||||
F8_TAPS -3, 7, -17, 121, 27, -10, 5, -2
|
||||
F8_TAPS -4, 9, -20, 115, 37, -13, 6, -2
|
||||
F8_TAPS -4, 10, -23, 108, 48, -16, 8, -3
|
||||
F8_TAPS -4, 10, -24, 100, 59, -19, 9, -3
|
||||
F8_TAPS -4, 11, -24, 90, 70, -21, 10, -4
|
||||
F8_TAPS -4, 11, -23, 80, 80, -23, 11, -4
|
||||
F8_TAPS -4, 10, -21, 70, 90, -24, 11, -4
|
||||
F8_TAPS -3, 9, -19, 59, 100, -24, 10, -4
|
||||
F8_TAPS -3, 8, -16, 48, 108, -23, 10, -4
|
||||
F8_TAPS -2, 6, -13, 37, 115, -20, 9, -4
|
||||
F8_TAPS -2, 5, -10, 27, 121, -17, 7, -3
|
||||
F8_TAPS -1, 3, -6, 17, 125, -13, 5, -2
|
||||
F8_TAPS 0, 1, -3, 8, 127, -7, 3, -1
|
||||
%endmacro
|
||||
|
||||
%define F8_TAPS F8_SSSE3_TAPS
|
||||
; int8_t ff_filters_ssse3[3][15][4][32]
|
||||
FILTER ssse3
|
||||
%define F8_TAPS F8_SSE2_TAPS
|
||||
; int16_t ff_filters_sse2[3][15][8][8]
|
||||
FILTER sse2
|
||||
%define F8_TAPS F8_16BPP_TAPS
|
||||
; int16_t ff_filters_16bpp[3][15][4][16]
|
||||
FILTER 16bpp
|
||||
|
||||
SECTION .text
|
||||
|
||||
%macro filter_sse2_h_fn 1
|
||||
%assign %%px mmsize/2
|
||||
cglobal vp9_%1_8tap_1d_h_ %+ %%px %+ _8, 6, 6, 15, dst, dstride, src, sstride, h, filtery
|
||||
pxor m5, m5
|
||||
mova m6, [pw_64]
|
||||
mova m7, [filteryq+ 0]
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
mova m8, [filteryq+ 16]
|
||||
mova m9, [filteryq+ 32]
|
||||
mova m10, [filteryq+ 48]
|
||||
mova m11, [filteryq+ 64]
|
||||
mova m12, [filteryq+ 80]
|
||||
mova m13, [filteryq+ 96]
|
||||
mova m14, [filteryq+112]
|
||||
%endif
|
||||
.loop:
|
||||
movh m0, [srcq-3]
|
||||
movh m1, [srcq-2]
|
||||
movh m2, [srcq-1]
|
||||
movh m3, [srcq+0]
|
||||
movh m4, [srcq+1]
|
||||
punpcklbw m0, m5
|
||||
punpcklbw m1, m5
|
||||
punpcklbw m2, m5
|
||||
punpcklbw m3, m5
|
||||
punpcklbw m4, m5
|
||||
pmullw m0, m7
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmullw m1, m8
|
||||
pmullw m2, m9
|
||||
pmullw m3, m10
|
||||
pmullw m4, m11
|
||||
%else
|
||||
pmullw m1, [filteryq+ 16]
|
||||
pmullw m2, [filteryq+ 32]
|
||||
pmullw m3, [filteryq+ 48]
|
||||
pmullw m4, [filteryq+ 64]
|
||||
%endif
|
||||
paddw m0, m1
|
||||
paddw m2, m3
|
||||
paddw m0, m4
|
||||
movh m1, [srcq+2]
|
||||
movh m3, [srcq+3]
|
||||
movh m4, [srcq+4]
|
||||
add srcq, sstrideq
|
||||
punpcklbw m1, m5
|
||||
punpcklbw m3, m5
|
||||
punpcklbw m4, m5
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmullw m1, m12
|
||||
pmullw m3, m13
|
||||
pmullw m4, m14
|
||||
%else
|
||||
pmullw m1, [filteryq+ 80]
|
||||
pmullw m3, [filteryq+ 96]
|
||||
pmullw m4, [filteryq+112]
|
||||
%endif
|
||||
paddw m0, m1
|
||||
paddw m3, m4
|
||||
paddw m0, m6
|
||||
paddw m2, m3
|
||||
paddsw m0, m2
|
||||
psraw m0, 7
|
||||
%ifidn %1, avg
|
||||
movh m1, [dstq]
|
||||
%endif
|
||||
packuswb m0, m0
|
||||
%ifidn %1, avg
|
||||
pavgb m0, m1
|
||||
%endif
|
||||
movh [dstq], m0
|
||||
add dstq, dstrideq
|
||||
dec hd
|
||||
jg .loop
|
||||
RET
|
||||
%endmacro
|
||||
|
||||
INIT_MMX mmxext
|
||||
filter_sse2_h_fn put
|
||||
filter_sse2_h_fn avg
|
||||
|
||||
INIT_XMM sse2
|
||||
filter_sse2_h_fn put
|
||||
filter_sse2_h_fn avg
|
||||
|
||||
%macro filter_h_fn 1
|
||||
%assign %%px mmsize/2
|
||||
cglobal vp9_%1_8tap_1d_h_ %+ %%px %+ _8, 6, 6, 11, dst, dstride, src, sstride, h, filtery
|
||||
mova m6, [pw_256]
|
||||
mova m7, [filteryq+ 0]
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
mova m8, [filteryq+32]
|
||||
mova m9, [filteryq+64]
|
||||
mova m10, [filteryq+96]
|
||||
%endif
|
||||
.loop:
|
||||
movh m0, [srcq-3]
|
||||
movh m1, [srcq-2]
|
||||
movh m2, [srcq-1]
|
||||
movh m3, [srcq+0]
|
||||
movh m4, [srcq+1]
|
||||
movh m5, [srcq+2]
|
||||
punpcklbw m0, m1
|
||||
punpcklbw m2, m3
|
||||
movh m1, [srcq+3]
|
||||
movh m3, [srcq+4]
|
||||
add srcq, sstrideq
|
||||
punpcklbw m4, m5
|
||||
punpcklbw m1, m3
|
||||
pmaddubsw m0, m7
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddubsw m2, m8
|
||||
pmaddubsw m4, m9
|
||||
pmaddubsw m1, m10
|
||||
%else
|
||||
pmaddubsw m2, [filteryq+32]
|
||||
pmaddubsw m4, [filteryq+64]
|
||||
pmaddubsw m1, [filteryq+96]
|
||||
%endif
|
||||
paddw m0, m4
|
||||
paddw m2, m1
|
||||
paddsw m0, m2
|
||||
pmulhrsw m0, m6
|
||||
%ifidn %1, avg
|
||||
movh m1, [dstq]
|
||||
%endif
|
||||
packuswb m0, m0
|
||||
%ifidn %1, avg
|
||||
pavgb m0, m1
|
||||
%endif
|
||||
movh [dstq], m0
|
||||
add dstq, dstrideq
|
||||
dec hd
|
||||
jg .loop
|
||||
RET
|
||||
%endmacro
|
||||
|
||||
INIT_MMX ssse3
|
||||
filter_h_fn put
|
||||
filter_h_fn avg
|
||||
|
||||
INIT_XMM ssse3
|
||||
filter_h_fn put
|
||||
filter_h_fn avg
|
||||
|
||||
%if ARCH_X86_64
|
||||
%macro filter_hx2_fn 1
|
||||
%assign %%px mmsize
|
||||
cglobal vp9_%1_8tap_1d_h_ %+ %%px %+ _8, 6, 6, 14, dst, dstride, src, sstride, h, filtery
|
||||
mova m13, [pw_256]
|
||||
mova m8, [filteryq+ 0]
|
||||
mova m9, [filteryq+32]
|
||||
mova m10, [filteryq+64]
|
||||
mova m11, [filteryq+96]
|
||||
.loop:
|
||||
movu m0, [srcq-3]
|
||||
movu m1, [srcq-2]
|
||||
movu m2, [srcq-1]
|
||||
movu m3, [srcq+0]
|
||||
movu m4, [srcq+1]
|
||||
movu m5, [srcq+2]
|
||||
movu m6, [srcq+3]
|
||||
movu m7, [srcq+4]
|
||||
add srcq, sstrideq
|
||||
SBUTTERFLY bw, 0, 1, 12
|
||||
SBUTTERFLY bw, 2, 3, 12
|
||||
SBUTTERFLY bw, 4, 5, 12
|
||||
SBUTTERFLY bw, 6, 7, 12
|
||||
pmaddubsw m0, m8
|
||||
pmaddubsw m1, m8
|
||||
pmaddubsw m2, m9
|
||||
pmaddubsw m3, m9
|
||||
pmaddubsw m4, m10
|
||||
pmaddubsw m5, m10
|
||||
pmaddubsw m6, m11
|
||||
pmaddubsw m7, m11
|
||||
paddw m0, m4
|
||||
paddw m1, m5
|
||||
paddw m2, m6
|
||||
paddw m3, m7
|
||||
paddsw m0, m2
|
||||
paddsw m1, m3
|
||||
pmulhrsw m0, m13
|
||||
pmulhrsw m1, m13
|
||||
packuswb m0, m1
|
||||
%ifidn %1, avg
|
||||
pavgb m0, [dstq]
|
||||
%endif
|
||||
mova [dstq], m0
|
||||
add dstq, dstrideq
|
||||
dec hd
|
||||
jg .loop
|
||||
RET
|
||||
%endmacro
|
||||
|
||||
INIT_XMM ssse3
|
||||
filter_hx2_fn put
|
||||
filter_hx2_fn avg
|
||||
|
||||
%if HAVE_AVX2_EXTERNAL
|
||||
INIT_YMM avx2
|
||||
filter_hx2_fn put
|
||||
filter_hx2_fn avg
|
||||
%endif
|
||||
|
||||
%endif ; ARCH_X86_64
|
||||
|
||||
%macro filter_sse2_v_fn 1
|
||||
%assign %%px mmsize/2
|
||||
%if ARCH_X86_64
|
||||
cglobal vp9_%1_8tap_1d_v_ %+ %%px %+ _8, 6, 8, 15, dst, dstride, src, sstride, h, filtery, src4, sstride3
|
||||
%else
|
||||
cglobal vp9_%1_8tap_1d_v_ %+ %%px %+ _8, 4, 7, 15, dst, dstride, src, sstride, filtery, src4, sstride3
|
||||
mov filteryq, r5mp
|
||||
%define hd r4mp
|
||||
%endif
|
||||
pxor m5, m5
|
||||
mova m6, [pw_64]
|
||||
lea sstride3q, [sstrideq*3]
|
||||
lea src4q, [srcq+sstrideq]
|
||||
sub srcq, sstride3q
|
||||
mova m7, [filteryq+ 0]
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
mova m8, [filteryq+ 16]
|
||||
mova m9, [filteryq+ 32]
|
||||
mova m10, [filteryq+ 48]
|
||||
mova m11, [filteryq+ 64]
|
||||
mova m12, [filteryq+ 80]
|
||||
mova m13, [filteryq+ 96]
|
||||
mova m14, [filteryq+112]
|
||||
%endif
|
||||
.loop:
|
||||
; FIXME maybe reuse loads from previous rows, or just
|
||||
; more generally unroll this to prevent multiple loads of
|
||||
; the same data?
|
||||
movh m0, [srcq]
|
||||
movh m1, [srcq+sstrideq]
|
||||
movh m2, [srcq+sstrideq*2]
|
||||
movh m3, [srcq+sstride3q]
|
||||
add srcq, sstrideq
|
||||
movh m4, [src4q]
|
||||
punpcklbw m0, m5
|
||||
punpcklbw m1, m5
|
||||
punpcklbw m2, m5
|
||||
punpcklbw m3, m5
|
||||
punpcklbw m4, m5
|
||||
pmullw m0, m7
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmullw m1, m8
|
||||
pmullw m2, m9
|
||||
pmullw m3, m10
|
||||
pmullw m4, m11
|
||||
%else
|
||||
pmullw m1, [filteryq+ 16]
|
||||
pmullw m2, [filteryq+ 32]
|
||||
pmullw m3, [filteryq+ 48]
|
||||
pmullw m4, [filteryq+ 64]
|
||||
%endif
|
||||
paddw m0, m1
|
||||
paddw m2, m3
|
||||
paddw m0, m4
|
||||
movh m1, [src4q+sstrideq]
|
||||
movh m3, [src4q+sstrideq*2]
|
||||
movh m4, [src4q+sstride3q]
|
||||
add src4q, sstrideq
|
||||
punpcklbw m1, m5
|
||||
punpcklbw m3, m5
|
||||
punpcklbw m4, m5
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmullw m1, m12
|
||||
pmullw m3, m13
|
||||
pmullw m4, m14
|
||||
%else
|
||||
pmullw m1, [filteryq+ 80]
|
||||
pmullw m3, [filteryq+ 96]
|
||||
pmullw m4, [filteryq+112]
|
||||
%endif
|
||||
paddw m0, m1
|
||||
paddw m3, m4
|
||||
paddw m0, m6
|
||||
paddw m2, m3
|
||||
paddsw m0, m2
|
||||
psraw m0, 7
|
||||
%ifidn %1, avg
|
||||
movh m1, [dstq]
|
||||
%endif
|
||||
packuswb m0, m0
|
||||
%ifidn %1, avg
|
||||
pavgb m0, m1
|
||||
%endif
|
||||
movh [dstq], m0
|
||||
add dstq, dstrideq
|
||||
dec hd
|
||||
jg .loop
|
||||
RET
|
||||
%endmacro
|
||||
|
||||
INIT_MMX mmxext
|
||||
filter_sse2_v_fn put
|
||||
filter_sse2_v_fn avg
|
||||
|
||||
INIT_XMM sse2
|
||||
filter_sse2_v_fn put
|
||||
filter_sse2_v_fn avg
|
||||
|
||||
%macro filter_v_fn 1
|
||||
%assign %%px mmsize/2
|
||||
%if ARCH_X86_64
|
||||
cglobal vp9_%1_8tap_1d_v_ %+ %%px %+ _8, 6, 8, 11, dst, dstride, src, sstride, h, filtery, src4, sstride3
|
||||
%else
|
||||
cglobal vp9_%1_8tap_1d_v_ %+ %%px %+ _8, 4, 7, 11, dst, dstride, src, sstride, filtery, src4, sstride3
|
||||
mov filteryq, r5mp
|
||||
%define hd r4mp
|
||||
%endif
|
||||
mova m6, [pw_256]
|
||||
lea sstride3q, [sstrideq*3]
|
||||
lea src4q, [srcq+sstrideq]
|
||||
sub srcq, sstride3q
|
||||
mova m7, [filteryq+ 0]
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
mova m8, [filteryq+32]
|
||||
mova m9, [filteryq+64]
|
||||
mova m10, [filteryq+96]
|
||||
%endif
|
||||
.loop:
|
||||
; FIXME maybe reuse loads from previous rows, or just
|
||||
; more generally unroll this to prevent multiple loads of
|
||||
; the same data?
|
||||
movh m0, [srcq]
|
||||
movh m1, [srcq+sstrideq]
|
||||
movh m2, [srcq+sstrideq*2]
|
||||
movh m3, [srcq+sstride3q]
|
||||
movh m4, [src4q]
|
||||
movh m5, [src4q+sstrideq]
|
||||
punpcklbw m0, m1
|
||||
punpcklbw m2, m3
|
||||
movh m1, [src4q+sstrideq*2]
|
||||
movh m3, [src4q+sstride3q]
|
||||
add srcq, sstrideq
|
||||
add src4q, sstrideq
|
||||
punpcklbw m4, m5
|
||||
punpcklbw m1, m3
|
||||
pmaddubsw m0, m7
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddubsw m2, m8
|
||||
pmaddubsw m4, m9
|
||||
pmaddubsw m1, m10
|
||||
%else
|
||||
pmaddubsw m2, [filteryq+32]
|
||||
pmaddubsw m4, [filteryq+64]
|
||||
pmaddubsw m1, [filteryq+96]
|
||||
%endif
|
||||
paddw m0, m4
|
||||
paddw m2, m1
|
||||
paddsw m0, m2
|
||||
pmulhrsw m0, m6
|
||||
%ifidn %1, avg
|
||||
movh m1, [dstq]
|
||||
%endif
|
||||
packuswb m0, m0
|
||||
%ifidn %1, avg
|
||||
pavgb m0, m1
|
||||
%endif
|
||||
movh [dstq], m0
|
||||
add dstq, dstrideq
|
||||
dec hd
|
||||
jg .loop
|
||||
RET
|
||||
%endmacro
|
||||
|
||||
INIT_MMX ssse3
|
||||
filter_v_fn put
|
||||
filter_v_fn avg
|
||||
|
||||
INIT_XMM ssse3
|
||||
filter_v_fn put
|
||||
filter_v_fn avg
|
||||
|
||||
%if ARCH_X86_64
|
||||
|
||||
%macro filter_vx2_fn 1
|
||||
%assign %%px mmsize
|
||||
cglobal vp9_%1_8tap_1d_v_ %+ %%px %+ _8, 6, 8, 14, dst, dstride, src, sstride, h, filtery, src4, sstride3
|
||||
mova m13, [pw_256]
|
||||
lea sstride3q, [sstrideq*3]
|
||||
lea src4q, [srcq+sstrideq]
|
||||
sub srcq, sstride3q
|
||||
mova m8, [filteryq+ 0]
|
||||
mova m9, [filteryq+32]
|
||||
mova m10, [filteryq+64]
|
||||
mova m11, [filteryq+96]
|
||||
.loop:
|
||||
; FIXME maybe reuse loads from previous rows, or just
|
||||
; more generally unroll this to prevent multiple loads of
|
||||
; the same data?
|
||||
movu m0, [srcq]
|
||||
movu m1, [srcq+sstrideq]
|
||||
movu m2, [srcq+sstrideq*2]
|
||||
movu m3, [srcq+sstride3q]
|
||||
movu m4, [src4q]
|
||||
movu m5, [src4q+sstrideq]
|
||||
movu m6, [src4q+sstrideq*2]
|
||||
movu m7, [src4q+sstride3q]
|
||||
add srcq, sstrideq
|
||||
add src4q, sstrideq
|
||||
SBUTTERFLY bw, 0, 1, 12
|
||||
SBUTTERFLY bw, 2, 3, 12
|
||||
SBUTTERFLY bw, 4, 5, 12
|
||||
SBUTTERFLY bw, 6, 7, 12
|
||||
pmaddubsw m0, m8
|
||||
pmaddubsw m1, m8
|
||||
pmaddubsw m2, m9
|
||||
pmaddubsw m3, m9
|
||||
pmaddubsw m4, m10
|
||||
pmaddubsw m5, m10
|
||||
pmaddubsw m6, m11
|
||||
pmaddubsw m7, m11
|
||||
paddw m0, m4
|
||||
paddw m1, m5
|
||||
paddw m2, m6
|
||||
paddw m3, m7
|
||||
paddsw m0, m2
|
||||
paddsw m1, m3
|
||||
pmulhrsw m0, m13
|
||||
pmulhrsw m1, m13
|
||||
packuswb m0, m1
|
||||
%ifidn %1, avg
|
||||
pavgb m0, [dstq]
|
||||
%endif
|
||||
mova [dstq], m0
|
||||
add dstq, dstrideq
|
||||
dec hd
|
||||
jg .loop
|
||||
RET
|
||||
%endmacro
|
||||
|
||||
INIT_XMM ssse3
|
||||
filter_vx2_fn put
|
||||
filter_vx2_fn avg
|
||||
|
||||
%if HAVE_AVX2_EXTERNAL
|
||||
INIT_YMM avx2
|
||||
filter_vx2_fn put
|
||||
filter_vx2_fn avg
|
||||
%endif
|
||||
|
||||
%endif ; ARCH_X86_64
|
||||
|
||||
%macro fpel_fn 6-8 0, 4
|
||||
%if %2 == 4
|
||||
%define %%srcfn movh
|
||||
%define %%dstfn movh
|
||||
%else
|
||||
%define %%srcfn movu
|
||||
%define %%dstfn mova
|
||||
%endif
|
||||
|
||||
%if %7 == 8
|
||||
%define %%pavg pavgb
|
||||
%define %%szsuf _8
|
||||
%elif %7 == 16
|
||||
%define %%pavg pavgw
|
||||
%define %%szsuf _16
|
||||
%else
|
||||
%define %%szsuf
|
||||
%endif
|
||||
|
||||
%if %2 <= mmsize
|
||||
cglobal vp9_%1%2 %+ %%szsuf, 5, 7, 4, dst, dstride, src, sstride, h, dstride3, sstride3
|
||||
lea sstride3q, [sstrideq*3]
|
||||
lea dstride3q, [dstrideq*3]
|
||||
%else
|
||||
cglobal vp9_%1%2 %+ %%szsuf, 5, 5, %8, dst, dstride, src, sstride, h
|
||||
%endif
|
||||
.loop:
|
||||
%%srcfn m0, [srcq]
|
||||
%%srcfn m1, [srcq+s%3]
|
||||
%%srcfn m2, [srcq+s%4]
|
||||
%%srcfn m3, [srcq+s%5]
|
||||
%if %2/mmsize == 8
|
||||
%%srcfn m4, [srcq+mmsize*4]
|
||||
%%srcfn m5, [srcq+mmsize*5]
|
||||
%%srcfn m6, [srcq+mmsize*6]
|
||||
%%srcfn m7, [srcq+mmsize*7]
|
||||
%endif
|
||||
lea srcq, [srcq+sstrideq*%6]
|
||||
%ifidn %1, avg
|
||||
%%pavg m0, [dstq]
|
||||
%%pavg m1, [dstq+d%3]
|
||||
%%pavg m2, [dstq+d%4]
|
||||
%%pavg m3, [dstq+d%5]
|
||||
%if %2/mmsize == 8
|
||||
%%pavg m4, [dstq+mmsize*4]
|
||||
%%pavg m5, [dstq+mmsize*5]
|
||||
%%pavg m6, [dstq+mmsize*6]
|
||||
%%pavg m7, [dstq+mmsize*7]
|
||||
%endif
|
||||
%endif
|
||||
%%dstfn [dstq], m0
|
||||
%%dstfn [dstq+d%3], m1
|
||||
%%dstfn [dstq+d%4], m2
|
||||
%%dstfn [dstq+d%5], m3
|
||||
%if %2/mmsize == 8
|
||||
%%dstfn [dstq+mmsize*4], m4
|
||||
%%dstfn [dstq+mmsize*5], m5
|
||||
%%dstfn [dstq+mmsize*6], m6
|
||||
%%dstfn [dstq+mmsize*7], m7
|
||||
%endif
|
||||
lea dstq, [dstq+dstrideq*%6]
|
||||
sub hd, %6
|
||||
jnz .loop
|
||||
RET
|
||||
%endmacro
|
||||
|
||||
%define d16 16
|
||||
%define s16 16
|
||||
%define d32 32
|
||||
%define s32 32
|
||||
INIT_MMX mmx
|
||||
fpel_fn put, 4, strideq, strideq*2, stride3q, 4
|
||||
fpel_fn put, 8, strideq, strideq*2, stride3q, 4
|
||||
INIT_MMX mmxext
|
||||
fpel_fn avg, 4, strideq, strideq*2, stride3q, 4, 8
|
||||
fpel_fn avg, 8, strideq, strideq*2, stride3q, 4, 8
|
||||
INIT_XMM sse
|
||||
fpel_fn put, 16, strideq, strideq*2, stride3q, 4
|
||||
fpel_fn put, 32, mmsize, strideq, strideq+mmsize, 2
|
||||
fpel_fn put, 64, mmsize, mmsize*2, mmsize*3, 1
|
||||
fpel_fn put, 128, mmsize, mmsize*2, mmsize*3, 1, 0, 8
|
||||
INIT_XMM sse2
|
||||
fpel_fn avg, 16, strideq, strideq*2, stride3q, 4, 8
|
||||
fpel_fn avg, 32, mmsize, strideq, strideq+mmsize, 2, 8
|
||||
fpel_fn avg, 64, mmsize, mmsize*2, mmsize*3, 1, 8
|
||||
INIT_YMM avx
|
||||
fpel_fn put, 32, strideq, strideq*2, stride3q, 4
|
||||
fpel_fn put, 64, mmsize, strideq, strideq+mmsize, 2
|
||||
fpel_fn put, 128, mmsize, mmsize*2, mmsize*3, 1
|
||||
%if HAVE_AVX2_EXTERNAL
|
||||
INIT_YMM avx2
|
||||
fpel_fn avg, 32, strideq, strideq*2, stride3q, 4, 8
|
||||
fpel_fn avg, 64, mmsize, strideq, strideq+mmsize, 2, 8
|
||||
%endif
|
||||
INIT_MMX mmxext
|
||||
fpel_fn avg, 8, strideq, strideq*2, stride3q, 4, 16
|
||||
INIT_XMM sse2
|
||||
fpel_fn avg, 16, strideq, strideq*2, stride3q, 4, 16
|
||||
fpel_fn avg, 32, mmsize, strideq, strideq+mmsize, 2, 16
|
||||
fpel_fn avg, 64, mmsize, mmsize*2, mmsize*3, 1, 16
|
||||
fpel_fn avg, 128, mmsize, mmsize*2, mmsize*3, 1, 16, 8
|
||||
%if HAVE_AVX2_EXTERNAL
|
||||
INIT_YMM avx2
|
||||
fpel_fn avg, 32, strideq, strideq*2, stride3q, 4, 16
|
||||
fpel_fn avg, 64, mmsize, strideq, strideq+mmsize, 2, 16
|
||||
fpel_fn avg, 128, mmsize, mmsize*2, mmsize*3, 1, 16
|
||||
%endif
|
||||
%undef s16
|
||||
%undef d16
|
||||
%undef s32
|
||||
%undef d32
|
||||
431
media/ffvpx/libavcodec/x86/vp9mc_16bpp.asm
Normal file
431
media/ffvpx/libavcodec/x86/vp9mc_16bpp.asm
Normal file
|
|
@ -0,0 +1,431 @@
|
|||
;******************************************************************************
|
||||
;* VP9 MC SIMD optimizations
|
||||
;*
|
||||
;* Copyright (c) 2015 Ronald S. Bultje <rsbultje gmail com>
|
||||
;*
|
||||
;* This file is part of FFmpeg.
|
||||
;*
|
||||
;* FFmpeg is free software; you can redistribute it and/or
|
||||
;* modify it under the terms of the GNU Lesser General Public
|
||||
;* License as published by the Free Software Foundation; either
|
||||
;* version 2.1 of the License, or (at your option) any later version.
|
||||
;*
|
||||
;* FFmpeg is distributed in the hope that it will be useful,
|
||||
;* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
;* Lesser General Public License for more details.
|
||||
;*
|
||||
;* You should have received a copy of the GNU Lesser General Public
|
||||
;* License along with FFmpeg; if not, write to the Free Software
|
||||
;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
;******************************************************************************
|
||||
|
||||
%include "libavutil/x86/x86util.asm"
|
||||
|
||||
SECTION_RODATA 32
|
||||
|
||||
pd_64: times 8 dd 64
|
||||
|
||||
cextern pw_1023
|
||||
cextern pw_4095
|
||||
|
||||
SECTION .text
|
||||
|
||||
%macro filter_h4_fn 1-2 12
|
||||
cglobal vp9_%1_8tap_1d_h_4_10, 6, 6, %2, dst, dstride, src, sstride, h, filtery
|
||||
mova m5, [pw_1023]
|
||||
.body:
|
||||
%if notcpuflag(sse4) && ARCH_X86_64
|
||||
pxor m11, m11
|
||||
%endif
|
||||
mova m6, [pd_64]
|
||||
mova m7, [filteryq+ 0]
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
mova m8, [filteryq+32]
|
||||
mova m9, [filteryq+64]
|
||||
mova m10, [filteryq+96]
|
||||
%endif
|
||||
.loop:
|
||||
movh m0, [srcq-6]
|
||||
movh m1, [srcq-4]
|
||||
movh m2, [srcq-2]
|
||||
movh m3, [srcq+0]
|
||||
movh m4, [srcq+2]
|
||||
punpcklwd m0, m1
|
||||
punpcklwd m2, m3
|
||||
pmaddwd m0, m7
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddwd m2, m8
|
||||
%else
|
||||
pmaddwd m2, [filteryq+32]
|
||||
%endif
|
||||
movu m1, [srcq+4]
|
||||
movu m3, [srcq+6]
|
||||
paddd m0, m2
|
||||
movu m2, [srcq+8]
|
||||
add srcq, sstrideq
|
||||
punpcklwd m4, m1
|
||||
punpcklwd m3, m2
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddwd m4, m9
|
||||
pmaddwd m3, m10
|
||||
%else
|
||||
pmaddwd m4, [filteryq+64]
|
||||
pmaddwd m3, [filteryq+96]
|
||||
%endif
|
||||
paddd m0, m4
|
||||
paddd m0, m3
|
||||
paddd m0, m6
|
||||
psrad m0, 7
|
||||
%if cpuflag(sse4)
|
||||
packusdw m0, m0
|
||||
%else
|
||||
packssdw m0, m0
|
||||
%endif
|
||||
%ifidn %1, avg
|
||||
movh m1, [dstq]
|
||||
%endif
|
||||
pminsw m0, m5
|
||||
%if notcpuflag(sse4)
|
||||
%if ARCH_X86_64
|
||||
pmaxsw m0, m11
|
||||
%else
|
||||
pxor m2, m2
|
||||
pmaxsw m0, m2
|
||||
%endif
|
||||
%endif
|
||||
%ifidn %1, avg
|
||||
pavgw m0, m1
|
||||
%endif
|
||||
movh [dstq], m0
|
||||
add dstq, dstrideq
|
||||
dec hd
|
||||
jg .loop
|
||||
RET
|
||||
|
||||
cglobal vp9_%1_8tap_1d_h_4_12, 6, 6, %2, dst, dstride, src, sstride, h, filtery
|
||||
mova m5, [pw_4095]
|
||||
jmp mangle(private_prefix %+ _ %+ vp9_%1_8tap_1d_h_4_10 %+ SUFFIX).body
|
||||
%endmacro
|
||||
|
||||
INIT_XMM sse2
|
||||
filter_h4_fn put
|
||||
filter_h4_fn avg
|
||||
|
||||
%macro filter_h_fn 1-2 12
|
||||
%assign %%px mmsize/2
|
||||
cglobal vp9_%1_8tap_1d_h_ %+ %%px %+ _10, 6, 6, %2, dst, dstride, src, sstride, h, filtery
|
||||
mova m5, [pw_1023]
|
||||
.body:
|
||||
%if notcpuflag(sse4) && ARCH_X86_64
|
||||
pxor m11, m11
|
||||
%endif
|
||||
mova m6, [pd_64]
|
||||
mova m7, [filteryq+ 0]
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
mova m8, [filteryq+32]
|
||||
mova m9, [filteryq+64]
|
||||
mova m10, [filteryq+96]
|
||||
%endif
|
||||
.loop:
|
||||
movu m0, [srcq-6]
|
||||
movu m1, [srcq-4]
|
||||
movu m2, [srcq-2]
|
||||
movu m3, [srcq+0]
|
||||
movu m4, [srcq+2]
|
||||
pmaddwd m0, m7
|
||||
pmaddwd m1, m7
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddwd m2, m8
|
||||
pmaddwd m3, m8
|
||||
pmaddwd m4, m9
|
||||
%else
|
||||
pmaddwd m2, [filteryq+32]
|
||||
pmaddwd m3, [filteryq+32]
|
||||
pmaddwd m4, [filteryq+64]
|
||||
%endif
|
||||
paddd m0, m2
|
||||
paddd m1, m3
|
||||
paddd m0, m4
|
||||
movu m2, [srcq+4]
|
||||
movu m3, [srcq+6]
|
||||
movu m4, [srcq+8]
|
||||
add srcq, sstrideq
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddwd m2, m9
|
||||
pmaddwd m3, m10
|
||||
pmaddwd m4, m10
|
||||
%else
|
||||
pmaddwd m2, [filteryq+64]
|
||||
pmaddwd m3, [filteryq+96]
|
||||
pmaddwd m4, [filteryq+96]
|
||||
%endif
|
||||
paddd m1, m2
|
||||
paddd m0, m3
|
||||
paddd m1, m4
|
||||
paddd m0, m6
|
||||
paddd m1, m6
|
||||
psrad m0, 7
|
||||
psrad m1, 7
|
||||
%if cpuflag(sse4)
|
||||
packusdw m0, m0
|
||||
packusdw m1, m1
|
||||
%else
|
||||
packssdw m0, m0
|
||||
packssdw m1, m1
|
||||
%endif
|
||||
punpcklwd m0, m1
|
||||
pminsw m0, m5
|
||||
%if notcpuflag(sse4)
|
||||
%if ARCH_X86_64
|
||||
pmaxsw m0, m11
|
||||
%else
|
||||
pxor m2, m2
|
||||
pmaxsw m0, m2
|
||||
%endif
|
||||
%endif
|
||||
%ifidn %1, avg
|
||||
pavgw m0, [dstq]
|
||||
%endif
|
||||
mova [dstq], m0
|
||||
add dstq, dstrideq
|
||||
dec hd
|
||||
jg .loop
|
||||
RET
|
||||
|
||||
cglobal vp9_%1_8tap_1d_h_ %+ %%px %+ _12, 6, 6, %2, dst, dstride, src, sstride, h, filtery
|
||||
mova m5, [pw_4095]
|
||||
jmp mangle(private_prefix %+ _ %+ vp9_%1_8tap_1d_h_ %+ %%px %+ _10 %+ SUFFIX).body
|
||||
%endmacro
|
||||
|
||||
INIT_XMM sse2
|
||||
filter_h_fn put
|
||||
filter_h_fn avg
|
||||
%if HAVE_AVX2_EXTERNAL
|
||||
INIT_YMM avx2
|
||||
filter_h_fn put
|
||||
filter_h_fn avg
|
||||
%endif
|
||||
|
||||
%macro filter_v4_fn 1-2 12
|
||||
%if ARCH_X86_64
|
||||
cglobal vp9_%1_8tap_1d_v_4_10, 6, 8, %2, dst, dstride, src, sstride, h, filtery, src4, sstride3
|
||||
%else
|
||||
cglobal vp9_%1_8tap_1d_v_4_10, 4, 7, %2, dst, dstride, src, sstride, filtery, src4, sstride3
|
||||
mov filteryq, r5mp
|
||||
%define hd r4mp
|
||||
%endif
|
||||
mova m5, [pw_1023]
|
||||
.body:
|
||||
%if notcpuflag(sse4) && ARCH_X86_64
|
||||
pxor m11, m11
|
||||
%endif
|
||||
mova m6, [pd_64]
|
||||
lea sstride3q, [sstrideq*3]
|
||||
lea src4q, [srcq+sstrideq]
|
||||
sub srcq, sstride3q
|
||||
mova m7, [filteryq+ 0]
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
mova m8, [filteryq+ 32]
|
||||
mova m9, [filteryq+ 64]
|
||||
mova m10, [filteryq+ 96]
|
||||
%endif
|
||||
.loop:
|
||||
; FIXME maybe reuse loads from previous rows, or just
|
||||
; more generally unroll this to prevent multiple loads of
|
||||
; the same data?
|
||||
movh m0, [srcq]
|
||||
movh m1, [srcq+sstrideq]
|
||||
movh m2, [srcq+sstrideq*2]
|
||||
movh m3, [srcq+sstride3q]
|
||||
add srcq, sstrideq
|
||||
movh m4, [src4q]
|
||||
punpcklwd m0, m1
|
||||
punpcklwd m2, m3
|
||||
pmaddwd m0, m7
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddwd m2, m8
|
||||
%else
|
||||
pmaddwd m2, [filteryq+ 32]
|
||||
%endif
|
||||
movh m1, [src4q+sstrideq]
|
||||
movh m3, [src4q+sstrideq*2]
|
||||
paddd m0, m2
|
||||
movh m2, [src4q+sstride3q]
|
||||
add src4q, sstrideq
|
||||
punpcklwd m4, m1
|
||||
punpcklwd m3, m2
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddwd m4, m9
|
||||
pmaddwd m3, m10
|
||||
%else
|
||||
pmaddwd m4, [filteryq+ 64]
|
||||
pmaddwd m3, [filteryq+ 96]
|
||||
%endif
|
||||
paddd m0, m4
|
||||
paddd m0, m3
|
||||
paddd m0, m6
|
||||
psrad m0, 7
|
||||
%if cpuflag(sse4)
|
||||
packusdw m0, m0
|
||||
%else
|
||||
packssdw m0, m0
|
||||
%endif
|
||||
%ifidn %1, avg
|
||||
movh m1, [dstq]
|
||||
%endif
|
||||
pminsw m0, m5
|
||||
%if notcpuflag(sse4)
|
||||
%if ARCH_X86_64
|
||||
pmaxsw m0, m11
|
||||
%else
|
||||
pxor m2, m2
|
||||
pmaxsw m0, m2
|
||||
%endif
|
||||
%endif
|
||||
%ifidn %1, avg
|
||||
pavgw m0, m1
|
||||
%endif
|
||||
movh [dstq], m0
|
||||
add dstq, dstrideq
|
||||
dec hd
|
||||
jg .loop
|
||||
RET
|
||||
|
||||
%if ARCH_X86_64
|
||||
cglobal vp9_%1_8tap_1d_v_4_12, 6, 8, %2, dst, dstride, src, sstride, h, filtery, src4, sstride3
|
||||
%else
|
||||
cglobal vp9_%1_8tap_1d_v_4_12, 4, 7, %2, dst, dstride, src, sstride, filtery, src4, sstride3
|
||||
mov filteryq, r5mp
|
||||
%endif
|
||||
mova m5, [pw_4095]
|
||||
jmp mangle(private_prefix %+ _ %+ vp9_%1_8tap_1d_v_4_10 %+ SUFFIX).body
|
||||
%endmacro
|
||||
|
||||
INIT_XMM sse2
|
||||
filter_v4_fn put
|
||||
filter_v4_fn avg
|
||||
|
||||
%macro filter_v_fn 1-2 13
|
||||
%assign %%px mmsize/2
|
||||
%if ARCH_X86_64
|
||||
cglobal vp9_%1_8tap_1d_v_ %+ %%px %+ _10, 6, 8, %2, dst, dstride, src, sstride, h, filtery, src4, sstride3
|
||||
%else
|
||||
cglobal vp9_%1_8tap_1d_v_ %+ %%px %+ _10, 4, 7, %2, dst, dstride, src, sstride, filtery, src4, sstride3
|
||||
mov filteryq, r5mp
|
||||
%define hd r4mp
|
||||
%endif
|
||||
mova m5, [pw_1023]
|
||||
.body:
|
||||
%if notcpuflag(sse4) && ARCH_X86_64
|
||||
pxor m12, m12
|
||||
%endif
|
||||
%if ARCH_X86_64
|
||||
mova m11, [pd_64]
|
||||
%endif
|
||||
lea sstride3q, [sstrideq*3]
|
||||
lea src4q, [srcq+sstrideq]
|
||||
sub srcq, sstride3q
|
||||
mova m7, [filteryq+ 0]
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
mova m8, [filteryq+ 32]
|
||||
mova m9, [filteryq+ 64]
|
||||
mova m10, [filteryq+ 96]
|
||||
%endif
|
||||
.loop:
|
||||
; FIXME maybe reuse loads from previous rows, or just
|
||||
; more generally unroll this to prevent multiple loads of
|
||||
; the same data?
|
||||
movu m0, [srcq]
|
||||
movu m1, [srcq+sstrideq]
|
||||
movu m2, [srcq+sstrideq*2]
|
||||
movu m3, [srcq+sstride3q]
|
||||
add srcq, sstrideq
|
||||
movu m4, [src4q]
|
||||
SBUTTERFLY wd, 0, 1, 6
|
||||
SBUTTERFLY wd, 2, 3, 6
|
||||
pmaddwd m0, m7
|
||||
pmaddwd m1, m7
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddwd m2, m8
|
||||
pmaddwd m3, m8
|
||||
%else
|
||||
pmaddwd m2, [filteryq+ 32]
|
||||
pmaddwd m3, [filteryq+ 32]
|
||||
%endif
|
||||
paddd m0, m2
|
||||
paddd m1, m3
|
||||
movu m2, [src4q+sstrideq]
|
||||
movu m3, [src4q+sstrideq*2]
|
||||
SBUTTERFLY wd, 4, 2, 6
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddwd m4, m9
|
||||
pmaddwd m2, m9
|
||||
%else
|
||||
pmaddwd m4, [filteryq+ 64]
|
||||
pmaddwd m2, [filteryq+ 64]
|
||||
%endif
|
||||
paddd m0, m4
|
||||
paddd m1, m2
|
||||
movu m4, [src4q+sstride3q]
|
||||
add src4q, sstrideq
|
||||
SBUTTERFLY wd, 3, 4, 6
|
||||
%if ARCH_X86_64 && mmsize > 8
|
||||
pmaddwd m3, m10
|
||||
pmaddwd m4, m10
|
||||
%else
|
||||
pmaddwd m3, [filteryq+ 96]
|
||||
pmaddwd m4, [filteryq+ 96]
|
||||
%endif
|
||||
paddd m0, m3
|
||||
paddd m1, m4
|
||||
%if ARCH_X86_64
|
||||
paddd m0, m11
|
||||
paddd m1, m11
|
||||
%else
|
||||
paddd m0, [pd_64]
|
||||
paddd m1, [pd_64]
|
||||
%endif
|
||||
psrad m0, 7
|
||||
psrad m1, 7
|
||||
%if cpuflag(sse4)
|
||||
packusdw m0, m1
|
||||
%else
|
||||
packssdw m0, m1
|
||||
%endif
|
||||
pminsw m0, m5
|
||||
%if notcpuflag(sse4)
|
||||
%if ARCH_X86_64
|
||||
pmaxsw m0, m12
|
||||
%else
|
||||
pxor m2, m2
|
||||
pmaxsw m0, m2
|
||||
%endif
|
||||
%endif
|
||||
%ifidn %1, avg
|
||||
pavgw m0, [dstq]
|
||||
%endif
|
||||
mova [dstq], m0
|
||||
add dstq, dstrideq
|
||||
dec hd
|
||||
jg .loop
|
||||
RET
|
||||
|
||||
%if ARCH_X86_64
|
||||
cglobal vp9_%1_8tap_1d_v_ %+ %%px %+ _12, 6, 8, %2, dst, dstride, src, sstride, h, filtery, src4, sstride3
|
||||
%else
|
||||
cglobal vp9_%1_8tap_1d_v_ %+ %%px %+ _12, 4, 7, %2, dst, dstride, src, sstride, filtery, src4, sstride3
|
||||
mov filteryq, r5mp
|
||||
%endif
|
||||
mova m5, [pw_4095]
|
||||
jmp mangle(private_prefix %+ _ %+ vp9_%1_8tap_1d_v_ %+ %%px %+ _10 %+ SUFFIX).body
|
||||
%endmacro
|
||||
|
||||
INIT_XMM sse2
|
||||
filter_v_fn put
|
||||
filter_v_fn avg
|
||||
%if HAVE_AVX2_EXTERNAL
|
||||
INIT_YMM avx2
|
||||
filter_v_fn put
|
||||
filter_v_fn avg
|
||||
%endif
|
||||
Loading…
Add table
Add a link
Reference in a new issue