Issue #1937 - Part 2: Update libaom source.

This commit is contained in:
Job Bautista 2022-06-25 18:15:40 +08:00 committed by roytam1
commit ecb7ec7377
948 changed files with 244949 additions and 88095 deletions

View file

@ -7,6 +7,8 @@ Andrey Norkin <anorkin@netflix.com>
Angie Chiang <angiebird@google.com>
Arild Fuldseth <arilfuld@cisco.com> <arild.fuldseth@gmail.com>
Arild Fuldseth <arilfuld@cisco.com> <arilfuld@cisco.com>
Aasaipriya Chandran <aasaipriya.c@ittiam.com>
Aasaipriya Chandran <aasaipriya.c@ittiam.com> Aasaipriya C <100778@ittiam.com>
Bohan Li <bohanli@google.com>
Changjun Yang <changjun.yang@intel.com>
Chi Yo Tsai <chiyotsai@google.com>
@ -56,14 +58,15 @@ Paul Wilkins <paulwilkins@google.com>
Peng Bin <binpengsmail@gmail.com>
Peng Bin <binpengsmail@gmail.com> <pengbin@kingsoft.com>
Peter de Rivaz <peter.derivaz@gmail.com> <peter.derivaz@argondesign.com>
Rachel Barker <rachelbarker@google.com> David Barker <david.barker@argondesign.com>
Ralph Giles <giles@xiph.org> <giles@entropywave.com>
Ralph Giles <giles@xiph.org> <giles@mozilla.com>
Remya Prakasan <remya.prakasan@ittiam.com>
Roger Zhou <youzhou@microsoft.com>
Ronald S. Bultje <rsbultje@gmail.com> <rbultje@google.com>
Ryan Lei <ryan.z.lei@intel.com>
Ryan Lei <ryan.z.lei@intel.com> <ryan.lei@intel.com>
Ryan Lei <ryan.z.lei@intel.com> <zlei3@ZLEI3-DESK.amr.corp.intel.com>
Ryan Lei <ryanlei@fb.com> <ryan.z.lei@intel.com>
Ryan Lei <ryanlei@fb.com> <ryan.lei@intel.com>
Ryan Lei <ryanlei@fb.com> <zlei3@ZLEI3-DESK.amr.corp.intel.com>
Sachin Kumar Garg <sachin.kumargarg@ittiam.com>
Sai Deng <sdeng@google.com>
Sami Pietilä <samipietila@google.com>
@ -82,6 +85,7 @@ Venkat Sanampudi <sanampudi.venkatarao@ittiam.com>
Wei-Ting Lin <weitinglin@google.com>
Wei-Ting Lin <weitinglin@google.com> <weitingco@gmail.com>
Wenyao Liu <wenyao.liu@cidana.com>
Will Bresnahan <bill.wresnahan@gmail.com>
Yaowu Xu <yaowu@google.com> <adam@xuyaowu.com>
Yaowu Xu <yaowu@google.com> <yaowu@xuyaowu.com>
Yaowu Xu <yaowu@google.com> <yaowu@yaowu-macbookpro.roam.corp.google.com>

View file

@ -3,7 +3,7 @@
Aamir Anis <aanis@google.com>
Aaron Watry <awatry@gmail.com>
Aasaipriya <aasaipriya.c@ittiam.com>
Aasaipriya Chandran <aasaipriya.c@ittiam.com>
Abo Talib Mahfoodh <ab.mahfoodh@gmail.com>
Adrian Grange <agrange@google.com>
Ahmad Sharif <asharif@google.com>
@ -12,6 +12,7 @@ Alexander Bokov <alexanderbokov@google.com>
Alexander Voronov <avoronov@graphics.cs.msu.ru>
Aex Converse <aconverse@google.com>
Alexis Ballier <aballier@gentoo.org>
Alex Peterson <petersonab@google.com>
Alok Ahuja <waveletcoeff@gmail.com>
Alpha Lam <hclam@google.com>
A.Mahfoodh <ab.mahfoodh@gmail.com>
@ -22,9 +23,11 @@ Andrew Russell <anrussell@google.com>
Andrey Norkin <anorkin@netflix.com>
Angie Chiang <angiebird@google.com>
Aniket Dhok <aniket.dhok@ittiam.com>
Aniket Wanare <Aniket.wanare@ittiam.com>
Ankur Saxena <ankurs@nvidia.com>
Arild Fuldseth <arilfuld@cisco.com>
Aron Rosenberg <arosenberg@logitech.com>
Arun Singh Negi <arun.negi@ittiam.com>
Attila Nagy <attilanagy@google.com>
Bohan Li <bohanli@google.com>
Brennan Shacklett <bshacklett@mozilla.com>
@ -34,9 +37,11 @@ Changjun Yang <changjun.yang@intel.com>
Charles 'Buck' Krasic <ckrasic@google.com>
Cheng Chen <chengchen@google.com>
Cherma Rajan A <cherma.rajan@ittiam.com>
Chethan Kumar R E <chethan.kumar@ittiam.com>
Chi Yo Tsai <chiyotsai@google.com>
Chm <chm@rock-chips.com>
Christian Duvivier <cduvivier@google.com>
Christopher Degawa <christopher.degawa@intel.com>
Cyril Concolato <cconcolato@netflix.com>
Dake He <dkhe@google.com>
Damon Shen <yjshen@google.com>
@ -45,13 +50,11 @@ Daniele Castagna <dcastagna@chromium.org>
Daniel Kang <ddkang@google.com>
Daniel Max Valenzuela <daniel.vt@samsung.com>
Danil Chapovalov <danilchap@google.com>
David Barker <david.barker@argondesign.com>
David Major <dmajor@mozilla.com>
David Michael Barr <b@rr-dav.id.au>
David Turner <david.turner@argondesign.com>
Deb Mukherjee <debargha@google.com>
Deepa K G <deepa.kg@ittiam.com>
Deng <zhipin.deng@intel.com>
Di Chen <chendixi@google.com>
Dim Temp <dimtemp0@gmail.com>
Dmitry Kovalev <dkovalev@google.com>
@ -90,7 +93,7 @@ Hui Su <huisu@google.com>
Ilie Halip <ilie.halip@gmail.com>
Ilya Brailovskiy <brailovs@lab126.com>
Imdad Sardharwalla <imdad.sardharwalla@argondesign.com>
iole moccagatta <iole.moccagatta@gmail.com>
Iole Moccagatta <iole.moccagatta@gmail.com>
Ivan Krasin <krasin@chromium.org>
Ivan Maltz <ivanmaltz@google.com>
Jacek Caban <cjacek@gmail.com>
@ -103,7 +106,8 @@ Jan Gerber <j@mailb.org>
Jan Kratochvil <jan.kratochvil@redhat.com>
Janne Salonen <jsalonen@google.com>
Jayasanker J <jayasanker.j@ittiam.com>
Jean-Marc Valin <jmvalin@mozilla.com>
Jayashri Murugan <jayashri.murugan@ittiam.com>
Jean-Marc Valin <jmvalin@jmvalin.ca>
Jean-Yves Avenard <jyavenard@mozilla.com>
Jeff Faust <jfaust@google.com>
Jeff Muizelaar <jmuizelaar@mozilla.com>
@ -122,33 +126,43 @@ John Stark <jhnstrk@gmail.com>
Jonathan Matthews <jonathan.matthews@argondesign.com>
Joshua Bleecher Snyder <josh@treelinelabs.com>
Joshua Litt <joshualitt@google.com>
Josh Verdejo <joverdejo@google.com>
Julia Robson <juliamrobson@gmail.com>
Justin Clift <justin@salasaga.org>
Justin Lebar <justin.lebar@gmail.com>
Katsuhisa Yuasa <berupon@gmail.com>
Kavi Ramamurthy <kavii@google.com>
KO Myung-Hun <komh@chollian.net>
Krishna Malladi <kmalladi@google.com>
Kyle Siefring <kylesiefring@gmail.com>
Larisa Markeeva <lmarkeeva@google.com>
Lauren Partin <lpartin@google.com>
Lawrence Velázquez <larryv@macports.org>
leolzhao <leolzhao@tencent.com>
Lester Lu <kslu@google.com>
liang zhao <leolzhao@tencent.com>
Linfeng Zhang <linfengz@google.com>
Link.Meng <monthev@gmail.com>
Logan Goldberg <logangw@google.com>
Lokeshwar Reddy B <lokeshwar.reddy@ittiam.com>
Lou Quillio <louquillio@google.com>
Luca Barbato <lu_zero@gentoo.org>
Luc Trudeau <ltrudeau@mozilla.com>
Luca Versari <veluca@google.com>
Luc Trudeau <luc@trud.ca>
Madhu Peringassery Krishnan <mpkrishnan@tencent.com>
Makoto Kato <makoto.kt@gmail.com>
Mans Rullgard <mans@mansr.com>
Marco Paniconi <marpan@google.com>
Mark Mentovai <mark@chromium.org>
Mark Wachsler <wachsler@google.com>
Martin Ettl <ettl.martin78@googlemail.com>
Martin Storsjo <martin@martin.st>
Maryla <maryla@google.com>
Matthew Heaney <matthewjheaney@chromium.org>
Matthieu Vaudano <matthieu.vaudano@allegrodvt.com>
Mattias Hansson <mattias.hansson@arm.com>
Maxym Dmytrychenko <maxim.d33@gmail.com>
Michael Bebenita <mbebenita@mozilla.com>
Michael Bebenita <mbebenita@gmail.com>
Michael Horowitz <mhoro@webrtc.org>
Michael Kohler <michaelkohler@live.com>
Michelle Findlay-Olynyk <mfo@google.com>
@ -160,8 +174,10 @@ Mingliang Chen <mlchen@google.com>
Mirko Bonadei <mbonadei@google.com>
Monty Montgomery <cmontgomery@mozilla.com>
Morton Jonuschat <yabawock@gmail.com>
Mudassir Galagnath <mudassir.galaganath@ittiam.com>
Mufaddal Chakera <mufaddal.chakera@ittiam.com>
Nathan E. Egge <negge@mozilla.com>
Neeraj Gadgil <neeraj.gadgil@ittiam.com>
Neil Birkbeck <birkbeck@google.com>
Nico Weber <thakis@chromium.org>
Nithya V S <nithya.vs@ittiam.com>
@ -178,8 +194,11 @@ Peng Bin <binpengsmail@gmail.com>
Pengchong Jin <pengchong@google.com>
Peter Boström <pbos@google.com>
Peter de Rivaz <peter.derivaz@gmail.com>
Peter Kasting <pkasting@chromium.org>
Philip Jägenstedt <philipj@opera.com>
Priit Laes <plaes@plaes.org>
Qiu Jianlin <jianlin.qiu@intel.com>
Rachel Barker <rachelbarker@google.com>
Rafael Ávila de Espíndola <rafael.espindola@gmail.com>
Rafaël Carré <funman@videolan.org>
Ralph Giles <giles@xiph.org>
@ -189,17 +208,19 @@ Remya Prakasan <remya.prakasan@ittiam.com>
Remy Foray <remy.foray@allegrodvt.com>
Rob Bradford <rob@linux.intel.com>
Robert-André Mauchin <zebob.m@gmail.com>
RogerZhou <youzhou@microsoft.com>
Robert Chin <robertchin@google.com>
Roger Zhou <youzhou@microsoft.com>
Rohit Athavale <rathaval@xilinx.com>
Ronald S. Bultje <rsbultje@gmail.com>
Rostislav Pehlivanov <rpehlivanov@mozilla.com>
Ruiling Song <ruiling.song@intel.com>
Rui Ueyama <ruiu@google.com>
Rupert Swarbrick <rupert.swarbrick@argondesign.com>
Ryan Lei <ryan.lei@intel.com>
Ryan Lei <ryanlei@fb.com>
Ryan Overbeck <rover@google.com>
Sachin Kumar Garg <sachin.kumargarg@ittiam.com>
Sai Deng <sdeng@google.com>
Sami Boukortt <sboukortt@google.com>
Sami Pietilä <samipietila@google.com>
Sarah Parker <sarahparker@google.com>
Sasi Inguva <isasi@google.com>
@ -212,6 +233,7 @@ Sean Purser-Haskell <seanhaskell@google.com>
Sebastien Alaiwan <sebastien.alaiwan@allegrodvt.com>
Sergey Kolomenkin <kolomenkin@gmail.com>
Sergey Ulanov <sergeyu@chromium.org>
S Hamsalekha <hamsalekha.s@ittiam.com>
Shimon Doodkin <helpmepro1@gmail.com>
Shunyao Li <shunyaoli@google.com>
SmilingWolf <lupo996@gmail.com>
@ -220,11 +242,13 @@ Stanislav Vitvitskyy <vitvitskyy@google.com>
Stefan Holmer <holmer@google.com>
Steinar Midtskogen <stemidts@cisco.com>
Suman Sunkara <sunkaras@google.com>
susannad <susannad@google.com>
Taekhyun Kim <takim@nvidia.com>
Takanori MATSUURA <t.matsuu@gmail.com>
Tamar Levy <tamar.levy@intel.com>
Tao Bai <michaelbai@chromium.org>
Tarek AMARA <amatarek@justin.tv>
Tarundeep Singh <tarundeep.singh@ittiam.com>
Tero Rintaluoma <teror@google.com>
Thijs Vermeir <thijsvermeir@gmail.com>
Thomas Daede <tdaede@mozilla.com>
@ -241,14 +265,22 @@ Urvang Joshi <urvang@google.com>
Venkat Sanampudi <sanampudi.venkatarao@ittiam.com>
Victoria Zhislina <niva213@gmail.com>
Vignesh Venkatasubramanian <vigneshv@google.com>
Vikas Prasad <vikas.prasad@ittiam.com>
Vincent Rabaud <vrabaud@google.com>
Vishesh <vishesh.garg@ittiam.com>
Vishnu Teja Manyam <vishnu.teja@ittiam.com>
Vitalii Dziumenko <vdziumenko@luxoft.com>
Vitalii Dziumenko <vdziumenko@luxoft.corp-partner.google.com>
Wan-Teh Chang <wtc@google.com>
Wei-Ting Lin <weitinglin@google.com>
Wenyao Liu <wenyao.liu@cidana.com>
Will Bresnahan <bill.wresnahan@gmail.com>
Xiaoqing Zhu <xzhu@netflix.com>
Xing Jin <ddvfinite@gmail.com>
Xin Zhao <xinzzhao@tencent.com>
Yaowu Xu <yaowu.google.com>
Yannis Guyon <yguyon@google.com>
Yaowu Xu <yaowu@google.com>
Yeqing Wu <yeqing_wu@apple.com>
Yi Luo <luoyi@google.com>
Yongzhe Wang <yongzhe@google.com>
Yue Chen <yuec@google.com>
@ -256,5 +288,5 @@ Yunqing Wang <yunqingwang@google.com>
Yury Gitman <yuryg@google.com>
Yushin Cho <ycho@mozilla.com>
Zhijie Yang <zhijie.yang@broadcom.com>
zhipin deng <zhipin.deng@intel.com>
Zhipin Deng <zhipin.deng@intel.com>
Zoe Liu <zoeliu@gmail.com>

View file

@ -1,3 +1,411 @@
2022-06-17 v3.4.0
This release includes compression efficiency and perceptual quality
improvements, speedup and memory optimizations, and some new features.
There are no ABI or API breaking changes in this release.
- New Features
* New --dist-metric flag with "qm-psnr" value to use quantization
matrices in the distortion computation for RD search. The default
value is "psnr".
* New command line option "--auto-intra-tools-off=1" to make
all-intra encoding faster for high bit rate under
"--deltaq-mode=3" mode.
* New rate control library aom_av1_rc for real-time hardware
encoders. Supports CBR for both one spatial layer and SVC.
* New image format AOM_IMG_FMT_NV12 can be used as input to the
encoder. The presence of AOM_IMG_FMT_NV12 can be detected at
compile time by checking if the macro AOM_HAVE_IMG_FMT_NV12 is
defined.
* New codec controls for the encoder:
o AV1E_SET_AUTO_INTRA_TOOLS_OFF. Only in effect if
--deltaq-mode=3.
o AV1E_SET_RTC_EXTERNAL_RC
o AV1E_SET_FP_MT. Only supported if libaom is built with
-DCONFIG_FRAME_PARALLEL_ENCODE=1.
o AV1E_GET_TARGET_SEQ_LEVEL_IDX
* New key-value pairs for the key-value API:
o --auto-intra-tools-off=0 (default) or 1. Only in effect if
--deltaq-mode=3.
o --strict-level-conformance=0 (default) or 1
o --fp-mt=0 (default) or 1. Only supported if libaom is built
with -DCONFIG_FRAME_PARALLEL_ENCODE=1.
* New aomenc options (not supported by the key-value API):
o --nv12
- Compression Efficiency Improvements
* Correctly calculate SSE for high bitdepth in skip mode, 0.2% to
0.6% coding gain.
* RTC at speed 9/10: BD-rate gain of ~4/5%
* RTC screen content coding: many improvements for real-time screen
at speed 10 (quality, speedup, and rate control), up to high
resolutions (1080p).
* RTC-SVC: fixes to make intra-only frames work for spatial layers.
* RTC-SVC: quality improvements for temporal layers.
* AV1 RT: A new passive rate control strategy for screen content, an
average of 7.5% coding gain, with some clips of 20+%. The feature
is turned off by default due to higher bit rate variation.
- Perceptual Quality Improvements
* RTC: Visual quality improvements for high speeds (9/10)
* Improvements in coding quality for all intra mode
- Speedup and Memory Optimizations
* ~10% speedup in good quality mode encoding.
* ~7% heap memory reduction in good quality encoding mode for speed
5 and 6.
* Ongoing improvements to intra-frame encoding performance on Arm
* Faster encoding speed for "--deltaq-mode=3" mode.
* ~10% speedup for speed 5/6, ~15% speedup for speed 7/8, and
~10% speedup for speed 9/10 in real time encoding mode
* ~20% heap memory reduction in still-picture encoding mode for
360p-720p resolutions with multiple threads
* ~13% speedup for speed 6 and ~12% speedup for speed 9 in
still-picture encoding mode.
* Optimizations to improve multi-thread efficiency for still-picture
encoding mode.
- Bug Fixes
* b/204460717: README.md: replace master with main
* b/210677928: libaom disable_order is surprising for
max_reference_frames=3
* b/222461449: -DCONFIG_TUNE_BUTTERAUGLI=1 broken
* b/227207606: write_greyscale writes incorrect chroma in highbd
mode
* b/229955363: Integer-overflow in linsolve_wiener
* https://crbug.com/aomedia/2032
* https://crbug.com/aomedia/2397
* https://crbug.com/aomedia/2563
* https://crbug.com/aomedia/2815
* https://crbug.com/aomedia/3009
* https://crbug.com/aomedia/3018
* https://crbug.com/aomedia/3045
* https://crbug.com/aomedia/3101
* https://crbug.com/aomedia/3130
* https://crbug.com/aomedia/3173
* https://crbug.com/aomedia/3184
* https://crbug.com/aomedia/3187
* https://crbug.com/aomedia/3190
* https://crbug.com/aomedia/3195
* https://crbug.com/aomedia/3197
* https://crbug.com/aomedia/3201
* https://crbug.com/aomedia/3202
* https://crbug.com/aomedia/3204
* https://crbug.com/aomedia/3205
* https://crbug.com/aomedia/3207
* https://crbug.com/aomedia/3208
* https://crbug.com/aomedia/3209
* https://crbug.com/aomedia/3213
* https://crbug.com/aomedia/3214
* https://crbug.com/aomedia/3219
* https://crbug.com/aomedia/3222
* https://crbug.com/aomedia/3223
* https://crbug.com/aomedia/3225
* https://crbug.com/aomedia/3226
* https://crbug.com/aomedia/3228
* https://crbug.com/aomedia/3232
* https://crbug.com/aomedia/3236
* https://crbug.com/aomedia/3237
* https://crbug.com/aomedia/3238
* https://crbug.com/aomedia/3240
* https://crbug.com/aomedia/3243
* https://crbug.com/aomedia/3244
* https://crbug.com/aomedia/3246
* https://crbug.com/aomedia/3248
* https://crbug.com/aomedia/3250
* https://crbug.com/aomedia/3251
* https://crbug.com/aomedia/3252
* https://crbug.com/aomedia/3255
* https://crbug.com/aomedia/3257
* https://crbug.com/aomedia/3259
* https://crbug.com/aomedia/3260
* https://crbug.com/aomedia/3267
* https://crbug.com/aomedia/3268
* https://crbug.com/aomedia/3269
* https://crbug.com/aomedia/3276
* https://crbug.com/aomedia/3278
* https://crbug.com/chromium/1290068
* https://crbug.com/chromium/1303237
* https://crbug.com/chromium/1304990
* https://crbug.com/chromium/1321141
* https://crbug.com/chromium/1321388
* https://crbug.com/oss-fuzz/44846
* https://crbug.com/oss-fuzz/44856
* https://crbug.com/oss-fuzz/44862
* https://crbug.com/oss-fuzz/44904
* https://crbug.com/oss-fuzz/45056
2022-01-28 v3.3.0
This release includes compression efficiency and perceptual quality
improvements, speedup and memory optimizations, some new features, and
several bug fixes.
- New Features
* AV1 RT: Introducing CDEF search level 5
* Changed real time speed 4 to behave the same as real time speed 5
* Add --deltaq-strength
* rtc: Allow scene-change and overshoot detection for svc
* rtc: Intra-only frame for svc
* AV1 RT: Option 2 for codec control AV1E_SET_ENABLE_CDEF to disable
CDEF on non-ref frames
* New codec controls AV1E_SET_LOOPFILTER_CONTROL and
AOME_GET_LOOPFILTER_LEVEL
* Improvements to three pass encoding
- Compression Efficiency Improvements
* Overall compression gains: 0.6%
- Perceptual Quality Improvements
* Improves the perceptual quality of high QP encoding for delta-q mode 4
* Auto select noise synthesis level for all intra
- Speedup and Memory Optimizations
* Added many SSE2 optimizations.
* Good quality 2-pass encoder speedups:
o Speed 2: 9%
o Speed 3: 12.5%
o Speed 4: 8%
o Speed 5: 3%
o Speed 6: 4%
* Real time mode encoder speedups:
o Speed 5: 2.6% BDRate gain, 4% speedup
o Speed 6: 3.5% BDRate gain, 4% speedup
o Speed 9: 1% BDRate gain, 3% speedup
o Speed 10: 3% BDRate gain, neutral speedup
* All intra encoding speedups (AVIF):
o Single thread - speed 6: 8%
o Single thread - speed 9: 15%
o Multi thread(8) - speed 6: 14%
o Multi thread(8) - speed 9: 34%
- Bug Fixes
* Issue 3163: Segmentation fault when using --enable-keyframe-filtering=2
* Issue 2436: Integer overflow in av1_warp_affine_c()
* Issue 3226: armv7 build failure due to gcc-11
* Issue 3195: Bug report on libaom (AddressSanitizer: heap-buffer-overflow)
* Issue 3191: Bug report on libaom (AddressSanitizer: SEGV on unknown
address)
* Issue 3176: Some SSE2/SADx4AvgTest.* tests fail on Windows
* Issue 3175: Some SSE2/SADSkipTest.* tests fail on Windows
2021-10-13 v3.2.0
This release includes compression efficiency and perceptual quality
improvements, speedup and memory optimizations, as well as some new
features.
- New Features
* Introduced speeds 7, 8, and 9 for all intra mode.
* Introduced speed 10 for real time mode.
* Introduced an API that allows external partition decisions.
* SVC: added support for compound prediction.
* SVC: added support for fixed SVC modes.
- Compression Efficiency Improvements
* Intra-mode search improvement.
* Improved real time (RT) mode BDrate savings by ~5% (RT speed 5)
and ~12% (RT speed 6). The improvement was measured on the video
conference set.
* Improved real time mode for nonrd path (speed 7, 8, 9): BDrate
gains of ~3-5%.
* Rate control and RD adjustments based on ML research in VP9.
Gains of ~0.5-1.0% for HD.
- Perceptual Quality Improvements
* Added a new mode --deltaq-mode=3 to improve perceptual quality
based on a differential contrast model for still images.
* Added a new mode deltaq-mode=4 to improve perceptual quality
based on user rated cq_level data set for still images.
* Weighting of some intra mode and partition size choices to better
manage and retain texture.
- Speedup and Memory Optimizations
* Further improved 2-pass good quality encoder speed:
o Speed 2 speedup: 18%
o Speed 3 speedup: 22%
o Speed 4 speedup: 37%
o Speed 5 speedup: 30%
o Speed 6 speedup: 20%
* Optimized the real time encoder (measured on the video conference
set):
o RT speed 5 speedup: 110%
o RT speed 6 speedup: 77%
- Bug Fixes
* Issue 3069: Fix one-pass mode keyframe placement off-by-one error.
* Issue 3156: Fix a bug in av1_quantize_lp AVX2 optimization.
2021-09-29 v3.1.3
This release includes several bug fixes.
- Bug fixes:
The following four cmake changes should help the people building
libaom using MSVC.
1. exports: use CMAKE_SHARED_LIBRARY_PREFIX to determine lib name
https://aomedia-review.googlesource.com/c/aom/+/142342
2. aom_install: Install lib dlls to bindir
https://aomedia-review.googlesource.com/c/aom/+/146546
3. aom_install: use relpath for install
https://aomedia-review.googlesource.com/c/aom/+/146550
4. aom_install: don't exclude msvc from install
https://aomedia-review.googlesource.com/c/aom/+/146547
aom/aom_encoder.h: remove configure option reference
https://aomedia-review.googlesource.com/c/aom/+/146743
Issue 3113: Tests for detecting chroma subsampling in
av1_copy_and_extend_frame() do not work when y_width or y_height is
1
Issue 3115: image2yuvconfig() should calculate uv_crop_width and
uv_crop_height from y_crop_width and y_crop_height
Issue 3140: rc_overshoot_pct is documented as having a range of
0-1000, but is range checked against 0-100
Issue 3147: Build failure on Apple M1 arm64
2021-07-20 v3.1.2
This release includes several bug fixes.
- Bug fixes:
exports.cmake: use APPLE and WIN32 and use def for mingw-w64
https://aomedia-review.googlesource.com/c/aom/+/139882
Issue 2993: Incorrect spatial_id when decoding base layer of
multi-layer stream
Issue 3080: Chroma Resampling by Encoder on Y4M Inputs Files Tagged
as C420mpeg2
Issue 3081: Use of uninitialized value $version_extra in
concatenation (.) or string at aom/build/cmake/version.pl line 88.
2021-06-08 v3.1.1
This release includes several bug fixes.
- Bug fixes:
Issue 2965: Cherry-picked the following four commits for the
tune=butteraugli mode.
1. Add libjxl to pkg_config if enabled:
https://aomedia-review.googlesource.com/c/aom/+/136044
2. Declare set_mb_butteraugli_rdmult_scaling static:
https://aomedia-review.googlesource.com/c/aom/+/134506
3. Add color range detection in tune=butteraugli mode:
https://aomedia-review.googlesource.com/c/aom/+/135521
4. Enable tune=butteraugli in all-intra mode:
https://aomedia-review.googlesource.com/c/aom/+/136082
Issue 3021: Fix vmaf model initialization error when not set to
tune=vmaf
Issue 3050: Compilation fails with -DCONFIG_TUNE_VMAF=1
Issue 3054: Consistent crash on near-static screen content, keyframe
related
2021-05-03 v3.1.0
This release adds an "all intra" mode to the encoder, which significantly
speeds up the encoding of AVIF still images at speed 6.
- Upgrading:
All intra mode for encoding AVIF still images and AV1 all intra videos:
AOM_USAGE_ALL_INTRA (2) can be passed as the 'usage' argument to
aom_codec_enc_config_default().
New encoder control IDs added:
- AV1E_SET_ENABLE_DIAGONAL_INTRA: Enable diagonal (D45 to D203) intra
prediction modes (0: false, 1: true (default)). Also available as
"enable-diagonal-intra" for the aom_codec_set_option() function.
New aom_tune_metric enum value: AOM_TUNE_BUTTERAUGLI. The new aomenc option
--tune=butteraugli was added to optimize the encoders perceptual quality by
optimizing the Butteraugli metric. Install libjxl (JPEG XL) and then pass
-DCONFIG_TUNE_BUTTERAUGLI=1 to the cmake command to enable it.
Addition of support for libvmaf 2.x.
- Enhancements:
Heap memory consumption for encoding AVIF still images is significantly
reduced.
- Bug fixes:
Issue 2601: third_party/libaom fails licensecheck
Issue 2950: Conditional expression for rc->this_key_frame_forced is always
true in find_next_key_frame()
Issue 2988: "make install" installs the aom.h header twice
Issue 2992: Incorrectly printing the temporal_id twice in dump_obu tool
Issue 2998:
Issue 2999:
Issue 3000:
2021-02-24 v3.0.0
This release includes compression efficiency improvement, speed improvement
for realtime mode, as well as some new APIs.
- Upgrading:
Support for PSNR calculation based on stream bit-depth.
New encoder control IDs added:
- AV1E_SET_ENABLE_RECT_TX
- AV1E_SET_VBR_CORPUS_COMPLEXITY_LAP
- AV1E_GET_BASELINE_GF_INTERVAL
- AV1E_SET_ENABLE_DNL_DENOISING
New decoder control IDs added:
- AOMD_GET_FWD_KF_PRESENT
- AOMD_GET_FRAME_FLAGS
- AOMD_GET_ALTREF_PRESENT
- AOMD_GET_TILE_INFO
- AOMD_GET_SCREEN_CONTENT_TOOLS_INFO
- AOMD_GET_STILL_PICTURE
- AOMD_GET_SB_SIZE
- AOMD_GET_SHOW_EXISTING_FRAME_FLAG
- AOMD_GET_S_FRAME_INFO
New aom_tune_content enum value: AOM_CONTENT_FILM
New aom_tune_metric enum value: AOM_TUNE_VMAF_NEG_MAX_GAIN
Coefficient and mode update can be turned off via
AV1E_SET_{COEFF/MODE}_COST_UPD_FREQ.
New key & value API added, available with aom_codec_set_option() function.
Scaling API expanded to include 1/4, 3/4 and 1/8.
- Enhancements:
Better multithreading performance with realtime mode.
New speed 9 setting for faster realtime encoding.
Smaller binary size with low bitdepth and realtime only build.
Temporal denoiser and its optimizations on x86 and Neon.
Optimizations for scaling.
Faster encoding with speed settings 2 to 6 for good encoding mode.
Improved documentation throughout the library, with function level
documentation, tree view and support for the dot tool.
- Bug fixes:
Aside from those mentioned in v2.0.1 and v2.0.2, this release includes the
following bug fixes:
Issue 2940: Segfault when encoding with --use-16bit-internal and --limit > 1
Issue 2941: Decoder mismatch with --rt --bit-depth=10 and --cpu-used=8
Issue 2895: mingw-w64 i686 gcc fails to build
Issue 2874: Separate ssse3 functions from sse2 file.
2021-02-09 v2.0.2
This release includes several bug fixes.

View file

@ -8,9 +8,30 @@
# License 1.0 was not distributed with this source code in the PATENTS file, you
# can obtain it at www.aomedia.org/license/patent.
#
cmake_minimum_required(VERSION 3.5)
if(CONFIG_TFLITE)
cmake_minimum_required(VERSION 3.11)
else()
cmake_minimum_required(VERSION 3.7)
endif()
set(AOM_ROOT "${CMAKE_CURRENT_SOURCE_DIR}")
set(AOM_CONFIG_DIR "${CMAKE_CURRENT_BINARY_DIR}")
if("${AOM_ROOT}" STREQUAL "${AOM_CONFIG_DIR}")
message(
FATAL_ERROR "Building from within the aom source tree is not supported.\n"
"Hint: Run these commands\n"
"$ rm -rf CMakeCache.txt CMakeFiles\n"
"$ mkdir -p ../aom_build\n" "$ cd ../aom_build\n"
"And re-run CMake from the aom_build directory.")
endif()
project(AOM C CXX)
# GENERATED source property global visibility.
if(POLICY CMP0118)
cmake_policy(SET CMP0118 NEW)
endif()
if(NOT EMSCRIPTEN)
if(NOT CMAKE_BUILD_TYPE AND NOT CMAKE_CONFIGURATION_TYPES)
set(CMAKE_BUILD_TYPE
@ -20,24 +41,32 @@ if(NOT EMSCRIPTEN)
endif()
endif()
set(AOM_ROOT "${CMAKE_CURRENT_SOURCE_DIR}")
set(AOM_CONFIG_DIR "${CMAKE_CURRENT_BINARY_DIR}")
if("${AOM_ROOT}" STREQUAL "${AOM_CONFIG_DIR}")
message(
FATAL_ERROR "Building from within the aom source tree is not supported.\n"
"Hint: Run these commands\n"
"$ rm -rf CMakeCache.txt CMakeFiles\n"
"$ mkdir -p ../aom_build\n" "$ cd ../aom_build\n"
"And re-run CMake from the aom_build directory.")
endif()
# Updating version info.
# Library version info. Update LT_CURRENT, LT_REVISION and LT_AGE when making a
# public release by following the guidelines in the libtool document:
# https://www.gnu.org/software/libtool/manual/libtool.html#Updating-version-info
set(SO_VERSION 2)
set(SO_FILE_VERSION 2.0.2)
#
# c=<current>, r=<revision>, a=<age>
#
# libtool generates a .so file as .so.[c-a].a.r, while -version-info c:r:a is
# passed to libtool.
#
# We set SO_FILE_VERSION = [c-a].a.r
set(LT_CURRENT 7)
set(LT_REVISION 0)
set(LT_AGE 4)
math(EXPR SO_VERSION "${LT_CURRENT} - ${LT_AGE}")
set(SO_FILE_VERSION "${SO_VERSION}.${LT_AGE}.${LT_REVISION}")
unset(LT_CURRENT)
unset(LT_REVISION)
unset(LT_AGE)
# Enable generators like Xcode and Visual Studio to place projects in folders.
set_property(GLOBAL PROPERTY USE_FOLDERS TRUE)
include("${AOM_ROOT}/build/cmake/aom_configure.cmake")
if(CONFIG_THREE_PASS)
include("${AOM_ROOT}/common/ivf_dec.cmake")
endif()
include("${AOM_ROOT}/aom_dsp/aom_dsp.cmake")
include("${AOM_ROOT}/aom_mem/aom_mem.cmake")
include("${AOM_ROOT}/aom_ports/aom_ports.cmake")
@ -88,6 +117,7 @@ list(APPEND AOM_LIBYUV_SOURCES
"${AOM_ROOT}/third_party/libyuv/include/libyuv/row.h"
"${AOM_ROOT}/third_party/libyuv/include/libyuv/scale.h"
"${AOM_ROOT}/third_party/libyuv/include/libyuv/scale_row.h"
"${AOM_ROOT}/third_party/libyuv/source/convert_argb.cc"
"${AOM_ROOT}/third_party/libyuv/source/cpu_id.cc"
"${AOM_ROOT}/third_party/libyuv/source/planar_functions.cc"
"${AOM_ROOT}/third_party/libyuv/source/row_any.cc"
@ -104,7 +134,8 @@ list(APPEND AOM_LIBYUV_SOURCES
"${AOM_ROOT}/third_party/libyuv/source/scale_mips.cc"
"${AOM_ROOT}/third_party/libyuv/source/scale_neon.cc"
"${AOM_ROOT}/third_party/libyuv/source/scale_neon64.cc"
"${AOM_ROOT}/third_party/libyuv/source/scale_win.cc")
"${AOM_ROOT}/third_party/libyuv/source/scale_win.cc"
"${AOM_ROOT}/third_party/libyuv/source/scale_uv.cc")
list(APPEND AOM_SOURCES
"${AOM_CONFIG_DIR}/config/aom_config.c"
@ -113,6 +144,7 @@ list(APPEND AOM_SOURCES
"${AOM_ROOT}/aom/aom_codec.h"
"${AOM_ROOT}/aom/aom_decoder.h"
"${AOM_ROOT}/aom/aom_encoder.h"
"${AOM_ROOT}/aom/aom_external_partition.h"
"${AOM_ROOT}/aom/aom_frame_buffer.h"
"${AOM_ROOT}/aom/aom_image.h"
"${AOM_ROOT}/aom/aom_integer.h"
@ -127,6 +159,10 @@ list(APPEND AOM_SOURCES
"${AOM_ROOT}/aom/src/aom_integer.c")
list(APPEND AOM_COMMON_APP_UTIL_SOURCES
"${AOM_ROOT}/av1/arg_defs.c"
"${AOM_ROOT}/av1/arg_defs.h"
"${AOM_ROOT}/common/args_helper.c"
"${AOM_ROOT}/common/args_helper.h"
"${AOM_ROOT}/common/args.c"
"${AOM_ROOT}/common/args.h"
"${AOM_ROOT}/common/av1_config.c"
@ -139,10 +175,11 @@ list(APPEND AOM_COMMON_APP_UTIL_SOURCES
"${AOM_ROOT}/common/rawenc.c"
"${AOM_ROOT}/common/rawenc.h"
"${AOM_ROOT}/common/y4menc.c"
"${AOM_ROOT}/common/y4menc.h")
"${AOM_ROOT}/common/y4menc.h"
"${AOM_ROOT}/common/ivfdec.c"
"${AOM_ROOT}/common/ivfdec.h")
list(APPEND AOM_DECODER_APP_UTIL_SOURCES "${AOM_ROOT}/common/ivfdec.c"
"${AOM_ROOT}/common/ivfdec.h" "${AOM_ROOT}/common/obudec.c"
list(APPEND AOM_DECODER_APP_UTIL_SOURCES "${AOM_ROOT}/common/obudec.c"
"${AOM_ROOT}/common/obudec.h" "${AOM_ROOT}/common/video_reader.c"
"${AOM_ROOT}/common/video_reader.h")
@ -173,6 +210,10 @@ list(APPEND AOM_WEBM_ENCODER_SOURCES "${AOM_ROOT}/common/webmenc.cc"
include_directories(${AOM_ROOT} ${AOM_CONFIG_DIR} ${AOM_ROOT}/apps
${AOM_ROOT}/common ${AOM_ROOT}/examples ${AOM_ROOT}/stats)
if(CONFIG_RUNTIME_CPU_DETECT AND ANDROID_NDK)
include_directories(${ANDROID_NDK}/sources/android/cpufeatures)
endif()
# Targets
add_library(aom_version ${AOM_VERSION_SOURCES})
add_dummy_source_file_to_target(aom_version c)
@ -263,13 +304,48 @@ if(NOT MSVC AND NOT APPLE)
endif()
endif()
if(CONFIG_AV1_RC_RTC AND CONFIG_AV1_ENCODER AND NOT BUILD_SHARED_LIBS)
list(APPEND AOM_AV1_RC_SOURCES "${AOM_ROOT}/av1/ratectrl_rtc.h"
"${AOM_ROOT}/av1/ratectrl_rtc.cc")
add_library(aom_av1_rc ${AOM_AV1_RC_SOURCES})
target_link_libraries(aom_av1_rc ${AOM_LIB_LINK_TYPE} aom)
if(NOT MSVC AND NOT APPLE)
target_link_libraries(aom_av1_rc ${AOM_LIB_LINK_TYPE} m)
endif()
endif()
if(CONFIG_AV1_ENCODER AND NOT CONFIG_REALTIME_ONLY AND NOT BUILD_SHARED_LIBS)
list(APPEND AOM_AV1_RC_QMODE_SOURCES
"${AOM_ROOT}/av1/ratectrl_qmode_interface.h"
"${AOM_ROOT}/av1/ratectrl_qmode_interface.cc"
"${AOM_ROOT}/av1/reference_manager.h"
"${AOM_ROOT}/av1/reference_manager.cc"
"${AOM_ROOT}/av1/ratectrl_qmode.h"
"${AOM_ROOT}/av1/ratectrl_qmode.cc")
add_library(av1_rc_qmode ${AOM_AV1_RC_QMODE_SOURCES})
target_link_libraries(av1_rc_qmode ${AOM_LIB_LINK_TYPE} aom)
if(NOT MSVC AND NOT APPLE)
target_link_libraries(av1_rc_qmode ${AOM_LIB_LINK_TYPE} m)
endif()
set_target_properties(av1_rc_qmode PROPERTIES LINKER_LANGUAGE CXX)
endif()
# List of object and static library targets.
set(AOM_LIB_TARGETS ${AOM_LIB_TARGETS} aom_rtcd aom_mem aom_scale aom)
if(CONFIG_AV1_RC_RTC AND CONFIG_AV1_ENCODER AND NOT BUILD_SHARED_LIBS)
set(AOM_LIB_TARGETS ${AOM_LIB_TARGETS} aom_av1_rc)
endif()
if(CONFIG_AV1_ENCODER AND NOT CONFIG_REALTIME_ONLY AND NOT BUILD_SHARED_LIBS)
set(AOM_LIB_TARGETS ${AOM_LIB_TARGETS} av1_rc_qmode)
endif()
if(BUILD_SHARED_LIBS)
set(AOM_LIB_TARGETS ${AOM_LIB_TARGETS} aom_static)
endif()
# Setup dependencies.
if(CONFIG_THREE_PASS)
setup_ivf_dec_targets()
endif()
setup_aom_dsp_targets()
setup_aom_mem_targets()
setup_aom_ports_targets()
@ -297,19 +373,23 @@ file(WRITE "${AOM_GEN_SRC_DIR}/usage_exit.cc"
#
if(ENABLE_EXAMPLES OR ENABLE_TESTS OR ENABLE_TOOLS)
add_library(aom_common_app_util OBJECT ${AOM_COMMON_APP_UTIL_SOURCES})
set_property(TARGET ${example} PROPERTY FOLDER examples)
if(CONFIG_AV1_DECODER)
add_library(aom_decoder_app_util OBJECT ${AOM_DECODER_APP_UTIL_SOURCES})
set_property(TARGET ${example} PROPERTY FOLDER examples)
# obudec depends on internal headers that require *rtcd.h
add_dependencies(aom_decoder_app_util aom_rtcd)
endif()
if(CONFIG_AV1_ENCODER)
add_library(aom_encoder_app_util OBJECT ${AOM_ENCODER_APP_UTIL_SOURCES})
set_property(TARGET ${example} PROPERTY FOLDER examples)
endif()
endif()
if((CONFIG_AV1_DECODER OR CONFIG_AV1_ENCODER) AND ENABLE_EXAMPLES)
add_executable(resize_util "${AOM_ROOT}/examples/resize_util.c"
$<TARGET_OBJECTS:aom_common_app_util>)
set_property(TARGET ${example} PROPERTY FOLDER examples)
list(APPEND AOM_APP_TARGETS resize_util)
endif()
@ -376,6 +456,14 @@ if(CONFIG_AV1_DECODER AND ENABLE_EXAMPLES)
list(APPEND AOM_APP_TARGETS ${AOM_DECODER_EXAMPLE_TARGETS})
endif()
if(CONFIG_LIBYUV OR CONFIG_TUNE_BUTTERAUGLI)
add_library(yuv OBJECT ${AOM_LIBYUV_SOURCES})
if(NOT MSVC)
target_compile_options(yuv PRIVATE -Wno-unused-parameter)
endif()
include_directories("${AOM_ROOT}/third_party/libyuv/include")
endif()
if(CONFIG_AV1_ENCODER)
if(ENABLE_EXAMPLES)
add_executable(aomenc "${AOM_ROOT}/apps/aomenc.c"
@ -397,6 +485,10 @@ if(CONFIG_AV1_ENCODER)
add_executable(noise_model "${AOM_ROOT}/examples/noise_model.c"
$<TARGET_OBJECTS:aom_common_app_util>
$<TARGET_OBJECTS:aom_encoder_app_util>)
add_executable(photon_noise_table
"${AOM_ROOT}/examples/photon_noise_table.c"
$<TARGET_OBJECTS:aom_common_app_util>
$<TARGET_OBJECTS:aom_encoder_app_util>)
add_executable(scalable_encoder "${AOM_ROOT}/examples/scalable_encoder.c"
$<TARGET_OBJECTS:aom_common_app_util>
$<TARGET_OBJECTS:aom_encoder_app_util>)
@ -407,8 +499,8 @@ if(CONFIG_AV1_ENCODER)
# Maintain a list of encoder example targets.
list(APPEND AOM_ENCODER_EXAMPLE_TARGETS aomenc lossless_encoder noise_model
set_maps simple_encoder scalable_encoder twopass_encoder
svc_encoder_rtc)
photon_noise_table set_maps simple_encoder scalable_encoder
twopass_encoder svc_encoder_rtc)
endif()
if(ENABLE_TOOLS)
@ -432,17 +524,131 @@ if(CONFIG_AV1_ENCODER)
list(APPEND AOM_APP_TARGETS ${AOM_ENCODER_EXAMPLE_TARGETS}
${AOM_ENCODER_TOOL_TARGETS})
if(CONFIG_TUNE_VMAF)
find_library(VMAF libvmaf.a vmaf)
if(NOT VMAF)
message(FATAL_ERROR "VMAF library not found.")
if(CONFIG_TUNE_BUTTERAUGLI)
find_package(PkgConfig)
# Use find_library() with STATIC_LINK_JXL for static build since
# pkg_check_modules() with LIBJXL_STATIC is not working.
if(STATIC_LINK_JXL OR NOT PKG_CONFIG_FOUND)
find_library(LIBJXL_LIBRARIES libjxl.a)
find_library(LIBHWY_LIBRARIES libhwy.a)
find_library(LIBSKCMS_LIBRARIES libskcms.a)
find_library(LIBBROTLICOMMON_LIBRARIES libbrotlicommon-static.a)
find_library(LIBBROTLIENC_LIBRARIES libbrotlienc-static.a)
find_library(LIBBROTLIDEC_LIBRARIES libbrotlidec-static.a)
find_path(LIBJXL_INCLUDE_DIRS butteraugli.h PATH_SUFFIXES jxl)
if(LIBJXL_LIBRARIES
AND LIBHWY_LIBRARIES
AND LIBSKCMS_LIBRARIES
AND LIBBROTLICOMMON_LIBRARIES
AND LIBBROTLIENC_LIBRARIES
AND LIBBROTLIDEC_LIBRARIES
AND LIBJXL_INCLUDE_DIRS)
message(STATUS "Found JXL library: ${LIBJXL_LIBRARIES} "
"${LIBHWY_LIBRARIES} ${LIBSKCMS_LIBRARIES} "
"${LIBBROTLICOMMON_LIBRARIES} ${LIBBROTLIENC_LIBRARIES}"
"${LIBBROTLIDEC_LIBRARIES}")
message(STATUS "Found JXL include: ${LIBJXL_INCLUDE_DIRS}")
else()
message(FATAL_ERROR "JXL library not found.")
endif()
target_link_libraries(aom
PRIVATE ${LIBJXL_LIBRARIES} ${LIBHWY_LIBRARIES}
${LIBSKCMS_LIBRARIES}
${LIBBROTLIENC_LIBRARIES}
${LIBBROTLIDEC_LIBRARIES}
${LIBBROTLICOMMON_LIBRARIES})
target_include_directories(aom_dsp_encoder PRIVATE ${LIBJXL_INCLUDE_DIRS})
else()
pkg_check_modules(LIBJXL REQUIRED libjxl)
target_link_libraries(aom PRIVATE ${LIBJXL_LDFLAGS} ${LIBJXL_LIBRARIES})
target_include_directories(aom_dsp_encoder PRIVATE ${LIBJXL_INCLUDE_DIRS})
if(LIBJXL_CFLAGS)
append_compiler_flag("${LIBJXL_CFLAGS}")
endif()
pkg_check_modules(LIBHWY REQUIRED libhwy)
target_link_libraries(aom PRIVATE ${LIBHWY_LDFLAGS} ${LIBHWY_LIBRARIES})
target_include_directories(aom_dsp_encoder
PRIVATE ${LIBLIBHWY_INCLUDE_DIRS})
if(LIBHWY_CFLAGS)
append_compiler_flag("${LIBHWY_CFLAGS}")
endif()
endif()
set_target_properties(aom PROPERTIES LINKER_LANGUAGE CXX)
if(BUILD_SHARED_LIBS)
set_target_properties(aom_static PROPERTIES LINKER_LANGUAGE CXX)
endif()
list(APPEND AOM_LIB_TARGETS yuv)
target_sources(aom PRIVATE $<TARGET_OBJECTS:yuv>)
if(BUILD_SHARED_LIBS)
target_sources(aom_static PRIVATE $<TARGET_OBJECTS:yuv>)
endif()
endif()
if(CONFIG_TFLITE)
include(FetchContent)
set(TFLITE_TAG "v2.6.1")
message(STATUS "Fetching TFLite ${TFLITE_TAG}...")
# static linking makes life with TFLite much easier
set(TFLITE_C_BUILD_SHARED_LIBS OFF)
# We don't care about comparing against these delegates (yet), and disabling
# it reduces compile time meaningfully
set(TFLITE_ENABLE_RUY OFF)
set(TFLITE_ENABLE_XNNPACK OFF)
fetchcontent_declare(tflite
GIT_REPOSITORY https://github.com/tensorflow/tensorflow
GIT_TAG ${TFLITE_TAG}
GIT_SHALLOW TRUE)
fetchcontent_getproperties(tflite)
if(NOT tflite_POPULATED)
fetchcontent_populate(tflite)
# Some of the subprojects (e.g. Eigen) are very noisy and emit status
# messages all the time. Temporary ignore status messages while adding
# this to silence it. Ugly but effective.
set(OLD_CMAKE_MESSAGE_LOG_LEVEL ${CMAKE_MESSAGE_LOG_LEVEL})
set(CMAKE_MESSAGE_LOG_LEVEL WARNING)
add_subdirectory(${tflite_SOURCE_DIR}/tensorflow/lite/c
${tflite_BINARY_DIR})
set(CMAKE_MESSAGE_LOG_LEVEL ${OLD_CMAKE_MESSAGE_LOG_LEVEL})
endif()
# Disable some noisy warnings in tflite
target_compile_options(tensorflow-lite PRIVATE -w)
# tensorflowlite_c is implicitly declared by this FetchContent
include_directories(${tflite_SOURCE_DIR})
target_link_libraries(aom PRIVATE tensorflow-lite)
endif()
if(CONFIG_TUNE_VMAF)
find_package(PkgConfig)
if(PKG_CONFIG_FOUND)
pkg_check_modules(VMAF REQUIRED libvmaf)
if(BUILD_SHARED_LIBS)
target_link_libraries(aom PRIVATE ${VMAF_LDFLAGS} ${VMAF_LIBRARIES})
else()
target_link_libraries(aom
PRIVATE ${VMAF_LDFLAGS} ${VMAF_LIBRARIES} -static)
endif()
target_include_directories(aom PRIVATE ${VMAF_INCLUDE_DIRS})
target_include_directories(aom_dsp_encoder PRIVATE ${VMAF_INCLUDE_DIRS})
if(VMAF_CFLAGS)
append_compiler_flag("${VMAF_CFLAGS}")
endif()
else()
message(FATAL_ERROR "CONFIG_TUNE_VMAF error: pkg-config not found.")
endif()
message("-- Found VMAF library: " ${VMAF})
set_target_properties(aom PROPERTIES LINKER_LANGUAGE CXX)
if(BUILD_SHARED_LIBS)
set_target_properties(aom_static PROPERTIES LINKER_LANGUAGE CXX)
endif()
target_link_libraries(aom PRIVATE ${VMAF})
endif()
endif()
@ -524,12 +730,6 @@ endforeach()
if(ENABLE_EXAMPLES OR ENABLE_TESTS OR ENABLE_TOOLS)
if(CONFIG_LIBYUV)
add_library(yuv OBJECT ${AOM_LIBYUV_SOURCES})
if(NOT MSVC)
target_compile_options(yuv PRIVATE -Wno-unused-parameter)
endif()
include_directories("${AOM_ROOT}/third_party/libyuv/include")
# Add to existing targets.
foreach(aom_app ${AOM_APP_TARGETS})
target_sources(${aom_app} PRIVATE $<TARGET_OBJECTS:yuv>)
@ -622,6 +822,17 @@ if(ENABLE_EXAMPLES AND "${CMAKE_GENERATOR}" MATCHES "Makefiles$")
endif()
if(BUILD_SHARED_LIBS)
if(NOT WIN32 AND NOT APPLE)
# The -z defs linker option reports unresolved symbol references from object
# files when building a shared library.
if("${CMAKE_VERSION}" VERSION_LESS "3.13")
# target_link_options() is not available before CMake 3.13.
target_link_libraries(aom PRIVATE -Wl,-z,defs)
else()
target_link_options(aom PRIVATE LINKER:-z,defs)
endif()
endif()
include("${AOM_ROOT}/build/cmake/exports.cmake")
setup_exports_target()
endif()
@ -630,13 +841,44 @@ endif()
set_user_flags()
# Aomedia documentation rule.
set(DOXYGEN_VERSION_VALUE 0)
if(ENABLE_DOCS)
include(FindDoxygen)
if(DOXYGEN_FOUND)
# Check if Doxygen version is >= minimum required version(i.e. 1.8.10).
set(MINIMUM_DOXYGEN_VERSION 1008010)
if(DOXYGEN_VERSION)
# Strip SHA1 from version string if present.
string(REGEX
REPLACE "^([0-9]+\\.[0-9]+\\.[0-9]+).*" "\\1" DOXYGEN_VERSION
${DOXYGEN_VERSION})
# Replace dots with semicolons to create a list.
string(REGEX REPLACE "\\." ";" DOXYGEN_VERSION_LIST ${DOXYGEN_VERSION})
# Parse version components from the list.
list(GET DOXYGEN_VERSION_LIST 0 DOXYGEN_MAJOR)
list(GET DOXYGEN_VERSION_LIST 1 DOXYGEN_MINOR)
list(GET DOXYGEN_VERSION_LIST 2 DOXYGEN_PATCH)
endif()
# Construct a version value for comparison.
math(EXPR DOXYGEN_MAJOR "${DOXYGEN_MAJOR}*1000000")
math(EXPR DOXYGEN_MINOR "${DOXYGEN_MINOR}*1000")
math(EXPR DOXYGEN_VERSION_VALUE
"${DOXYGEN_MAJOR} + ${DOXYGEN_MINOR} + ${DOXYGEN_PATCH}")
if(${DOXYGEN_VERSION_VALUE} LESS ${MINIMUM_DOXYGEN_VERSION})
set(DOXYGEN_FOUND NO)
endif()
endif()
if(DOXYGEN_FOUND)
include("${AOM_ROOT}/docs.cmake")
setup_documentation_targets()
else()
message("--- Cannot find doxygen, ENABLE_DOCS turned off.")
message(
"--- Cannot find doxygen(version 1.8.10 or newer), ENABLE_DOCS turned off."
)
set(ENABLE_DOCS OFF)
endif()
endif()
@ -652,12 +894,14 @@ endif()
if(ENABLE_EXAMPLES)
foreach(example ${AOM_EXAMPLE_TARGETS})
list(APPEND AOM_DIST_EXAMPLES $<TARGET_FILE:${example}>)
set_property(TARGET ${example} PROPERTY FOLDER examples)
endforeach()
endif()
if(ENABLE_TOOLS)
foreach(tool ${AOM_TOOL_TARGETS})
list(APPEND AOM_DIST_TOOLS $<TARGET_FILE:${tool}>)
set_property(TARGET ${tool} PROPERTY FOLDER tools)
endforeach()
endif()
@ -694,6 +938,10 @@ foreach(var ${all_cmake_vars})
endif()
endforeach()
if(NOT CONFIG_AV1_DECODER)
list(FILTER aom_source_vars EXCLUDE REGEX "_DECODER_")
endif()
# Libaom_srcs.txt generation.
set(libaom_srcs_txt_file "${AOM_CONFIG_DIR}/libaom_srcs.txt")
file(WRITE "${libaom_srcs_txt_file}" "# This file is generated. DO NOT EDIT.\n")
@ -703,6 +951,9 @@ foreach(aom_source_var ${aom_source_vars})
foreach(file ${${aom_source_var}})
if(NOT "${file}" MATCHES "${AOM_CONFIG_DIR}")
string(REPLACE "${AOM_ROOT}/" "" file "${file}")
if(NOT CONFIG_AV1_DECODER AND "${file}" MATCHES "aom_decoder")
continue()
endif()
file(APPEND "${libaom_srcs_txt_file}" "${file}\n")
endif()
endforeach()
@ -733,6 +984,9 @@ foreach(aom_source_var ${aom_source_vars})
if(NOT "${file}" MATCHES "${AOM_CONFIG_DIR}")
string(REPLACE "${AOM_ROOT}" "//third_party/libaom/source/libaom" file
"${file}")
if(NOT CONFIG_AV1_DECODER AND "${file}" MATCHES "aom_decoder")
continue()
endif()
file(APPEND "${libaom_srcs_gni_file}" " \"${file}\",\n")
endif()
endforeach()

View file

@ -1,3 +1,5 @@
README.md {#LREADME}
=========
# AV1 Codec Library
## Contents
@ -40,23 +42,24 @@
5. [Support](#support)
6. [Bug reports](#bug-reports)
## Building the library and applications
## Building the library and applications {#building-the-library-and-applications}
### Prerequisites
### Prerequisites {#prerequisites}
1. [CMake](https://cmake.org) version 3.5 or higher.
1. [CMake](https://cmake.org). See CMakeLists.txt for the minimum version
required.
2. [Git](https://git-scm.com/).
3. [Perl](https://www.perl.org/).
4. For x86 targets, [yasm](http://yasm.tortall.net/), which is preferred, or a
recent version of [nasm](http://www.nasm.us/). If you download yasm with
the intention to work with Visual Studio, please download win32.exe or
win64.exe and rename it into yasm.exe. DO NOT download or use vsyasm.exe.
5. Building the documentation requires [doxygen](http://doxygen.org).
6. Building the unit tests requires [Python](https://www.python.org/).
7. Emscripten builds require the portable
5. Building the documentation requires
[doxygen version 1.8.10 or newer](http://doxygen.org).
6. Emscripten builds require the portable
[EMSDK](https://kripken.github.io/emscripten-site/index.html).
### Get the code
### Get the code {#get-the-code}
The AV1 library source code is stored in the Alliance for Open Media Git
repository:
@ -67,7 +70,7 @@ repository:
$ cd aom
~~~
### Basic build
### Basic build {#basic-build}
CMake replaces the configure step typical of many projects. Running CMake will
produce configuration and build files for the currently selected CMake
@ -85,7 +88,7 @@ successfully. The compiler chosen varies by host platform, but a general rule
applies: On systems where cc and c++ are present in $PATH at the time CMake is
run the generated build will use cc and c++ by default.
### Configuration options
### Configuration options {#configuration-options}
The AV1 codec library has a great many configuration options. These come in two
varieties:
@ -106,7 +109,7 @@ configuration options can be found at the top of the CMakeLists.txt file found
in the root of the AV1 repository, and AV1 codec configuration options can
currently be found in the file `build/cmake/aom_config_defaults.cmake`.
### Dylib builds
### Dylib builds {#dylib-builds}
A dylib (shared object) build of the AV1 codec library can be enabled via the
CMake built in variable `BUILD_SHARED_LIBS`:
@ -118,7 +121,7 @@ CMake built in variable `BUILD_SHARED_LIBS`:
This is currently only supported on non-Windows targets.
### Debugging
### Debugging {#debugging}
Depending on the generator used there are multiple ways of going about
debugging AV1 components. For single configuration generators like the Unix
@ -147,7 +150,7 @@ generic at generation time:
$ cmake path/to/aom -DAOM_TARGET_CPU=generic
~~~
### Cross compiling
### Cross compiling {#cross-compiling}
For the purposes of building the AV1 codec and applications and relative to the
scope of this guide, all builds for architectures differing from the native host
@ -197,7 +200,7 @@ In addition to the above it's important to note that the toolchain files
suffixed with gcc behave differently than the others. These toolchain files
attempt to obey the $CROSS environment variable.
### Sanitizers
### Sanitizers {#sanitizers}
Sanitizer integration is built-in to the CMake build system. To enable a
sanitizer, add `-DSANITIZE=<type>` to the CMake command line. For example, to
@ -211,7 +214,7 @@ enable address sanitizer:
Sanitizers available vary by platform, target, and compiler. Consult your
compiler documentation to determine which, if any, are available.
### Microsoft Visual Studio builds
### Microsoft Visual Studio builds {#microsoft-visual-studio-builds}
Building the AV1 codec library in Microsoft Visual Studio is supported. Visual
Studio 2017 (15.0) or later is required. The following example demonstrates
@ -241,7 +244,7 @@ generating projects and a solution for the Microsoft IDE:
NOTE: The build system targets Windows 7 or later by compiling files with
`-D_WIN32_WINNT=0x0601`.
### Xcode builds
### Xcode builds {#xcode-builds}
Building the AV1 codec library in Xcode is supported. The following example
demonstrates generating an Xcode project:
@ -250,7 +253,7 @@ demonstrates generating an Xcode project:
$ cmake path/to/aom -G Xcode
~~~
### Emscripten builds
### Emscripten builds {#emscripten-builds}
Building the AV1 codec library with Emscripten is supported. Typically this is
used to hook into the AOMAnalyzer GUI application. These instructions focus on
@ -261,7 +264,7 @@ It is assumed here that you have already downloaded and installed the EMSDK,
installed and activated at least one toolchain, and setup your environment
appropriately using the emsdk\_env script.
1. Download [AOMAnalyzer](https://people.xiph.org/~mbebenita/analyzer/).
1. Build [AOM Analyzer](https://github.com/xiph/aomanalyzer).
2. Configure the build:
@ -293,7 +296,7 @@ appropriately using the emsdk\_env script.
$ path/to/AOMAnalyzer path/to/examples/inspect.js path/to/av1/input/file
~~~
### Extra build flags
### Extra build flags {#extra-build-flags}
Three variables allow for passing of additional flags to the build system.
@ -312,10 +315,10 @@ These flags can be used, for example, to enable asserts in a release build:
-DAOM_EXTRA_CXX_FLAGS=-UNDEBUG
~~~
### Build with VMAF support
### Build with VMAF support {#build-with-vmaf}
After installing
[libvmaf.a](https://github.com/Netflix/vmaf/blob/master/resource/doc/libvmaf.md),
[libvmaf.a](https://github.com/Netflix/vmaf/tree/master/libvmaf),
you can use it with the encoder:
~~~
@ -323,22 +326,22 @@ you can use it with the encoder:
~~~
Please note that the default VMAF model
("/usr/local/share/model/vmaf_v0.6.1.pkl")
("/usr/local/share/model/vmaf_v0.6.1.json")
will be used unless you set the following flag when running the encoder:
~~~
# --vmaf-model-path=path/to/model
~~~
## Testing the AV1 codec
## Testing the AV1 codec {#testing-the-av1-codec}
### Testing basics
### Testing basics {#testing-basics}
There are several methods of testing the AV1 codec. All of these methods require
the presence of the AV1 source code and a working build of the AV1 library and
applications.
#### 1. Unit tests:
#### 1. Unit tests: {#1_unit-tests}
The unit tests can be run at build time:
@ -352,7 +355,7 @@ The unit tests can be run at build time:
$ make runtests
~~~
#### 2. Example tests:
#### 2. Example tests: {#2_example-tests}
The example tests require a bash shell and can be run in the following manner:
@ -367,7 +370,7 @@ The example tests require a bash shell and can be run in the following manner:
$ path/to/aom/test/examples.sh --bin-path examples
~~~
#### 3. Encoder tests:
#### 3. Encoder tests: {#3_encoder-tests}
When making a change to the encoder run encoder tests to confirm that your
change has a positive or negligible impact on encode quality. When running these
@ -418,7 +421,7 @@ report that can be viewed in a web browser:
You can view the report by opening mytweak.html in a web browser.
### IDE hosted tests
### IDE hosted tests {#ide-hosted-tests}
By default the generated projects files created by CMake will not include the
runtests and testdata rules when generating for IDEs like Microsoft Visual
@ -434,11 +437,13 @@ options in MSVS and Xcode. To enable the test rules in IDEs the
$ cmake path/to/aom -DENABLE_IDE_TEST_HOSTING=1 -G Xcode
~~~
### Downloading the test data
### Downloading the test data {#downloading-the-test-data}
The fastest and easiest way to obtain the test data is to use CMake to generate
a build using the Unix Makefiles generator, and then to build only the testdata
rule:
rule. By default the test files will be downloaded to the current directory. The
`LIBAOM_TEST_DATA_PATH` environment variable can be used to set a
custom one.
~~~
$ cmake path/to/aom -G "Unix Makefiles"
@ -448,7 +453,7 @@ rule:
The above make command will only download and verify the test data.
### Adding a new test data file
### Adding a new test data file {#adding-a-new-test-data-file}
First, add the new test data file to the `aom-test-data` bucket of the
`aomedia-testing` project on Google Cloud Platform. You may need to ask someone
@ -470,19 +475,19 @@ the SHA1 checksum of the new test data file to `test/test-data.sha1`. (The SHA1
checksum of a file can be calculated by running the `sha1sum` command on the
file.)
### Additional test data
### Additional test data {#additional-test-data}
The test data mentioned above is strictly intended for unit testing.
Additional input data for testing the encoder can be obtained from:
https://media.xiph.org/video/derf/
### Sharded testing
### Sharded testing {#sharded-testing}
The AV1 codec library unit tests are built upon gtest which supports sharding of
test jobs. Sharded test runs can be achieved in a couple of ways.
#### 1. Running test\_libaom directly:
#### 1. Running test\_libaom directly: {#1_running-test_libaom-directly}
~~~
# Set the environment variable GTEST_TOTAL_SHARDS to control the number of
@ -496,7 +501,7 @@ test jobs. Sharded test runs can be achieved in a couple of ways.
To create a test shard for each CPU core available on the current system set
`GTEST_TOTAL_SHARDS` to the number of CPU cores on your system minus one.
#### 2. Running the tests via the CMake build:
#### 2. Running the tests via the CMake build: {#2_running-the-tests-via-the-cmake-build}
~~~
# For IDE based builds, ENABLE_IDE_TEST_HOSTING must be enabled. See
@ -515,14 +520,14 @@ CMake. A system with 24 cores can run 24 test shards using a value of 24 with
the `-j` parameter. When CMake is unable to detect the number of cores 10 shards
is the default maximum value.
## Coding style
## Coding style {#coding-style}
We are using the Google C Coding Style defined by the
[Google C++ Style Guide](https://google.github.io/styleguide/cppguide.html).
The coding style used by this project is enforced with clang-format using the
configuration contained in the
[.clang-format](https://chromium.googlesource.com/webm/aom/+/master/.clang-format)
[.clang-format](https://chromium.googlesource.com/webm/aom/+/main/.clang-format)
file in the root of the repository.
You can download clang-format using your system's package manager, or directly
@ -556,27 +561,27 @@ Some Git installations have clang-format integration. Here are some examples:
$ git clang-format -f -p
~~~
## Submitting patches
## Submitting patches {#submitting-patches}
We manage the submission of patches using the
[Gerrit](https://www.gerritcodereview.com/) code review tool. This tool
implements a workflow on top of the Git version control system to ensure that
all changes get peer reviewed and tested prior to their distribution.
### Login cookie
### Login cookie {#login-cookie}
Browse to [AOMedia Git index](https://aomedia.googlesource.com/) and login with
your account (Gmail credentials, for example). Next, follow the
`Generate Password` Password link at the top of the page. Youll be given
instructions for creating a cookie to use with our Git repos.
### Contributor agreement
### Contributor agreement {#contributor-agreement}
You will be required to execute a
[contributor agreement](http://aomedia.org/license) to ensure that the AOMedia
Project has the right to distribute your changes.
### Testing your code
### Testing your code {#testing-your-code}
The testing basics are covered in the [testing section](#testing-the-av1-codec)
above.
@ -584,7 +589,7 @@ above.
In addition to the local tests, many more (e.g. asan, tsan, valgrind) will run
through Jenkins instances upon upload to gerrit.
### Commit message hook
### Commit message hook {#commit-message-hook}
Gerrit requires that each submission include a unique Change-Id. You can assign
one manually using git commit --amend, but its easier to automate it with the
@ -604,15 +609,15 @@ See the Gerrit
[documentation](https://gerrit-review.googlesource.com/Documentation/user-changeid.html)
for more information.
### Upload your change
### Upload your change {#upload-your-change}
The command line to upload your patch looks like this:
~~~
$ git push https://aomedia-review.googlesource.com/aom HEAD:refs/for/master
$ git push https://aomedia-review.googlesource.com/aom HEAD:refs/for/main
~~~
### Incorporating reviewer comments
### Incorporating reviewer comments {#incorporating-reviewer-comments}
If you previously uploaded a change to Gerrit and the Approver has asked for
changes, follow these steps:
@ -631,7 +636,7 @@ In general, you should not rebase your changes when doing updates in response to
review. Doing so can make it harder to follow the evolution of your change in
the diff view.
### Submitting your change
### Submitting your change {#submitting-your-change}
Once your change has been Approved and Verified, you can “submit” it through the
Gerrit UI. This will usually automatically rebase your change onto the branch
@ -648,18 +653,18 @@ must rebase your changes manually:
If there are any conflicts, resolve them as you normally would with Git. When
youre done, reupload your change.
### Viewing the status of uploaded changes
### Viewing the status of uploaded changes {#viewing-the-status-of-uploaded-changes}
To check the status of a change that you uploaded, open
[Gerrit](https://aomedia-review.googlesource.com/), sign in, and click My >
Changes.
## Support
## Support {#support}
This library is an open source project supported by its community. Please
please email aomediacodec@jointdevelopment.kavi.com for help.
## Bug reports
## Bug reports {#bug-reports}
Bug reports can be filed in the Alliance for Open Media
[issue tracker](https://bugs.chromium.org/p/aomedia/issues/list).

View file

@ -41,27 +41,45 @@ extern "C" {
/*!\brief Control functions
*
* The set of macros define the control functions of AOM interface
* The range for common control IDs is 230-255(max).
*/
enum aom_com_control_id {
/* TODO(https://crbug.com/aomedia/2671): The encoder overlaps the range of
* these values for its control ids, see the NOTEs in aom/aomcx.h. These
* should be migrated to something like the AOM_DECODER_CTRL_ID_START range
* next time we're ready to break the ABI.
/*!\brief Codec control function to get a pointer to a reference frame
*
* av1_ref_frame_t* parameter
*/
AV1_GET_REFERENCE = 128, /**< get a pointer to a reference frame,
av1_ref_frame_t* parameter */
AV1_SET_REFERENCE = 129, /**< write a frame into a reference buffer,
av1_ref_frame_t* parameter */
AV1_COPY_REFERENCE = 130, /**< get a copy of reference frame from the decoderm
av1_ref_frame_t* parameter */
AOM_COMMON_CTRL_ID_MAX,
AV1_GET_REFERENCE = 230,
AV1_GET_NEW_FRAME_IMAGE =
192, /**< get a pointer to the new frame, aom_image_t* parameter */
AV1_COPY_NEW_FRAME_IMAGE = 193, /**< copy the new frame to an external buffer,
aom_image_t* parameter */
/*!\brief Codec control function to write a frame into a reference buffer
*
* av1_ref_frame_t* parameter
*/
AV1_SET_REFERENCE = 231,
/*!\brief Codec control function to get a copy of reference frame from the
* decoder
*
* av1_ref_frame_t* parameter
*/
AV1_COPY_REFERENCE = 232,
/*!\brief Codec control function to get a pointer to the new frame
*
* aom_image_t* parameter
*/
AV1_GET_NEW_FRAME_IMAGE = 233,
/*!\brief Codec control function to copy the new frame to an external buffer
*
* aom_image_t* parameter
*/
AV1_COPY_NEW_FRAME_IMAGE = 234,
/*!\brief Start point of control IDs for aom_dec_control_id.
* Any new common control IDs should be added above.
*/
AOM_DECODER_CTRL_ID_START = 256
// No common control IDs should be added after AOM_DECODER_CTRL_ID_START.
};
/*!\brief AV1 specific reference frame data struct

View file

@ -9,6 +9,57 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
///////////////////////////////////////////////////////////////////////////////
// Internal implementation details
///////////////////////////////////////////////////////////////////////////////
//
// There are two levels of interfaces used to access the AOM codec: the
// the aom_codec_iface and the aom_codec_ctx.
//
// 1. aom_codec_iface_t
// (Related files: aom/aom_codec.h, aom/src/aom_codec.c,
// aom/internal/aom_codec_internal.h, av1/av1_cx_iface.c,
// av1/av1_dx_iface.c)
//
// Used to initialize the codec context, which contains the configuration for
// for modifying the encoder/decoder during run-time. See the other
// documentation in this header file for more details. For the most part,
// users will call helper functions, such as aom_codec_iface_name,
// aom_codec_get_caps, etc., to interact with it.
//
// The main purpose of the aom_codec_iface_t is to provide a way to generate
// a default codec config, find out what capabilities the implementation has,
// and create an aom_codec_ctx_t (which is actually used to interact with the
// codec).
//
// Note that the implementations for the AV1 algorithm are located in
// av1/av1_cx_iface.c and av1/av1_dx_iface.c
//
//
// 2. aom_codec_ctx_t
// (Related files: aom/aom_codec.h, av1/av1_cx_iface.c, av1/av1_dx_iface.c,
// aom/aomcx.h, aom/aomdx.h, aom/src/aom_encoder.c, aom/src/aom_decoder.c)
//
// The actual interface between user code and the codec. It stores the name
// of the codec, a pointer back to the aom_codec_iface_t that initialized it,
// initialization flags, a config for either encoder or the decoder, and a
// pointer to internal data.
//
// The codec is configured / queried through calls to aom_codec_control,
// which takes a control ID (listed in aomcx.h and aomdx.h) and a parameter.
// In the case of "getter" control IDs, the parameter is modified to have
// the requested value; in the case of "setter" control IDs, the codec's
// configuration is changed based on the parameter. Note that a aom_codec_err_t
// is returned, which indicates if the operation was successful or not.
//
// Note that for the encoder, the aom_codec_alg_priv_t points to the
// the aom_codec_alg_priv structure in av1/av1_cx_iface.c, and for the decoder,
// the struct in av1/av1_dx_iface.c. Variables such as AV1_COMP cpi are stored
// here and also used in the core algorithm.
//
// At the end, aom_codec_destroy should be called for each initialized
// aom_codec_ctx_t.
/*!\defgroup codec Common Algorithm Interface
* This abstraction allows applications to easily support multiple video
* formats with minimal code duplication. This section describes the interface
@ -23,13 +74,16 @@
* video codec algorithm.
*
* An application instantiates a specific codec instance by using
* aom_codec_init() and a pointer to the algorithm's interface structure:
* aom_codec_dec_init() or aom_codec_enc_init() and a pointer to the
* algorithm's interface structure:
* <pre>
* my_app.c:
* extern aom_codec_iface_t my_codec;
* {
* aom_codec_ctx_t algo;
* res = aom_codec_init(&algo, &my_codec);
* int threads = 4;
* aom_codec_dec_cfg_t cfg = { threads, 0, 0, 1 };
* res = aom_codec_dec_init(&algo, &my_codec, &cfg, 0);
* }
* </pre>
*
@ -95,7 +149,7 @@ extern "C" {
* types, removing or reassigning enums, adding/removing/rearranging
* fields to structures
*/
#define AOM_CODEC_ABI_VERSION (5 + AOM_IMAGE_ABI_VERSION) /**<\hideinitializer*/
#define AOM_CODEC_ABI_VERSION (7 + AOM_IMAGE_ABI_VERSION) /**<\hideinitializer*/
/*!\brief Algorithm return codes */
typedef enum {
@ -185,13 +239,17 @@ typedef int64_t aom_codec_pts_t;
* Contains function pointers and other data private to the codec
* implementation. This structure is opaque to the application. Common
* functions used with this structure:
* - aom_codec_iface_name: get the name of the codec
* - aom_codec_get_caps: returns the capabilities of the codec (see
* aom_encoder.h for more details)
* - aom_codec_enc_config_default: generate the default config to use
* when initializing the encoder
* - aom_codec_iface_name(aom_codec_iface_t *iface): get the
* name of the codec
* - aom_codec_get_caps(aom_codec_iface_t *iface): returns
* the capabilities of the codec
* - aom_codec_enc_config_default: generate the default config for
* initializing the encoder (see documention in aom_encoder.h)
* - aom_codec_dec_init, aom_codec_enc_init: initialize the codec context
* structure (see documentation on aom_codec_ctx for more information).
* structure (see documentation on aom_codec_ctx).
*
* To get access to the AV1 encoder and decoder, use aom_codec_av1_cx() and
* aom_codec_av1_dx().
*/
typedef const struct aom_codec_iface aom_codec_iface_t;
@ -202,6 +260,27 @@ typedef const struct aom_codec_iface aom_codec_iface_t;
*/
typedef struct aom_codec_priv aom_codec_priv_t;
/*!\brief Compressed Frame Flags
*
* This type represents a bitfield containing information about a compressed
* frame that may be useful to an application. The most significant 16 bits
* can be used by an algorithm to provide additional detail, for example to
* support frame types that are codec specific (MPEG-1 D-frames for example)
*/
typedef uint32_t aom_codec_frame_flags_t;
#define AOM_FRAME_IS_KEY 0x1 /**< frame is the start of a GOP */
/*!\brief frame can be dropped without affecting the stream (no future frame
* depends on this one) */
#define AOM_FRAME_IS_DROPPABLE 0x2
/*!\brief this is an INTRA_ONLY frame */
#define AOM_FRAME_IS_INTRAONLY 0x10
/*!\brief this is an S-frame */
#define AOM_FRAME_IS_SWITCH 0x20
/*!\brief this is an error-resilient frame */
#define AOM_FRAME_IS_ERROR_RESILIENT 0x40
/*!\brief this is a key-frame dependent recovery-point frame */
#define AOM_FRAME_IS_DELAYED_RANDOM_ACCESS_POINT 0x80
/*!\brief Iterator
*
* Opaque storage used for iterating over lists.
@ -266,31 +345,27 @@ typedef enum aom_superblock_size {
/*!\brief Return the version information (as an integer)
*
* Returns a packed encoding of the library version number. This will only
* include
* the major.minor.patch component of the version number. Note that this encoded
* value should be accessed through the macros provided, as the encoding may
* change
* in the future.
* include the major.minor.patch component of the version number. Note that this
* encoded value should be accessed through the macros provided, as the encoding
* may change in the future.
*
*/
int aom_codec_version(void);
/*!\brief Return the version major number */
/*!\brief Return the major version number */
#define aom_codec_version_major() ((aom_codec_version() >> 16) & 0xff)
/*!\brief Return the version minor number */
/*!\brief Return the minor version number */
#define aom_codec_version_minor() ((aom_codec_version() >> 8) & 0xff)
/*!\brief Return the version patch number */
/*!\brief Return the patch version number */
#define aom_codec_version_patch() ((aom_codec_version() >> 0) & 0xff)
/*!\brief Return the version information (as a string)
*
* Returns a printable string containing the full library version number. This
* may
* contain additional text following the three digit version number, as to
* indicate
* release candidates, prerelease versions, etc.
* may contain additional text following the three digit version number, as to
* indicate release candidates, prerelease versions, etc.
*
*/
const char *aom_codec_version_str(void);
@ -298,8 +373,7 @@ const char *aom_codec_version_str(void);
/*!\brief Return the version information (as a string)
*
* Returns a printable "extra string". This is the component of the string
* returned
* by aom_codec_version_str() following the three digit version number.
* returned by aom_codec_version_str() following the three digit version number.
*
*/
const char *aom_codec_version_extra_str(void);
@ -405,17 +479,38 @@ aom_codec_caps_t aom_codec_get_caps(aom_codec_iface_t *iface);
* ctx->err will be set to the same value as the return value.
*
* \param[in] ctx Pointer to this instance's context
* \param[in] ctrl_id Algorithm specific control identifier
* \param[in] ctrl_id Algorithm specific control identifier.
* Must be nonzero.
*
* \retval #AOM_CODEC_OK
* The control request was processed.
* \retval #AOM_CODEC_ERROR
* The control request was not processed.
* \retval #AOM_CODEC_INVALID_PARAM
* The data was not valid.
* The control ID was zero, or the data was not valid.
*/
aom_codec_err_t aom_codec_control(aom_codec_ctx_t *ctx, int ctrl_id, ...);
/*!\brief Key & Value API
*
* aom_codec_set_option() takes a context, a key (option name) and a value. If
* the context is non-null and an error occurs, ctx->err will be set to the same
* value as the return value.
*
* \param[in] ctx Pointer to this instance's context
* \param[in] name The name of the option (key)
* \param[in] value The value of the option
*
* \retval #AOM_CODEC_OK
* The value of the option was set.
* \retval #AOM_CODEC_INVALID_PARAM
* The data was not valid.
* \retval #AOM_CODEC_ERROR
* The option was not successfully set.
*/
aom_codec_err_t aom_codec_set_option(aom_codec_ctx_t *ctx, const char *name,
const char *value);
/*!\brief aom_codec_control wrapper macro (adds type-checking, less flexible)
*
* This macro allows for type safe conversions across the variadic parameter

View file

@ -31,17 +31,28 @@ extern "C" {
#endif
#include "aom/aom_codec.h"
#include "aom/aom_external_partition.h"
/*!\brief Current ABI version number
*
* \hideinitializer
* \internal
* If this file is altered in any way that changes the ABI, this value
* must be bumped. Examples include, but are not limited to, changing
* types, removing or reassigning enums, adding/removing/rearranging
* fields to structures
*
* Note: In the definition of AOM_ENCODER_ABI_VERSION, 3 is the value of
* AOM_EXT_PART_ABI_VERSION in libaom v3.2.0. The old value of
* AOM_EXT_PART_ABI_VERSION is used so as to not break the ABI version check in
* aom_codec_enc_init_ver() when an application compiled against libaom v3.2.0
* passes the old value of AOM_ENCODER_ABI_VERSION to aom_codec_enc_init_ver().
* The external partition API is still experimental. When it is declared stable,
* we will replace 3 with AOM_EXT_PART_ABI_VERSION in the definition of
* AOM_ENCODER_ABI_VERSION.
*/
#define AOM_ENCODER_ABI_VERSION \
(8 + AOM_CODEC_ABI_VERSION) /**<\hideinitializer*/
(10 + AOM_CODEC_ABI_VERSION + /*AOM_EXT_PART_ABI_VERSION=*/3)
/*! \brief Encoder capabilities bitfield
*
@ -78,27 +89,6 @@ typedef struct aom_fixed_buf {
size_t sz; /**< Length of the buffer, in chars */
} aom_fixed_buf_t; /**< alias for struct aom_fixed_buf */
/*!\brief Compressed Frame Flags
*
* This type represents a bitfield containing information about a compressed
* frame that may be useful to an application. The most significant 16 bits
* can be used by an algorithm to provide additional detail, for example to
* support frame types that are codec specific (MPEG-1 D-frames for example)
*/
typedef uint32_t aom_codec_frame_flags_t;
#define AOM_FRAME_IS_KEY 0x1 /**< frame is the start of a GOP */
/*!\brief frame can be dropped without affecting the stream (no future frame
* depends on this one) */
#define AOM_FRAME_IS_DROPPABLE 0x2
/*!\brief this is an INTRA_ONLY frame */
#define AOM_FRAME_IS_INTRAONLY 0x10
/*!\brief this is an S-frame */
#define AOM_FRAME_IS_SWITCH 0x20
/*!\brief this is an error-resilient frame */
#define AOM_FRAME_IS_ERROR_RESILIENT 0x40
/*!\brief this is a key-frame dependent recovery-point frame */
#define AOM_FRAME_IS_DELAYED_RANDOM_ACCESS_POINT 0x80
/*!\brief Error Resilient flags
*
* These flags define which error resilient features to enable in the
@ -152,17 +142,19 @@ typedef struct aom_codec_cx_pkt {
unsigned int samples[4]; /**< Number of samples, total/y/u/v */
uint64_t sse[4]; /**< sum squared error, total/y/u/v */
double psnr[4]; /**< PSNR, total/y/u/v */
} psnr; /**< data for PSNR packet */
aom_fixed_buf_t raw; /**< data for arbitrary packets */
/* This packet size is fixed to allow codecs to extend this
* interface without having to manage storage for raw packets,
* i.e., if it's smaller than 128 bytes, you can store in the
* packet list directly.
*/
char pad[128 - sizeof(enum aom_codec_cx_pkt_kind)]; /**< fixed sz */
} data; /**< packet data */
} aom_codec_cx_pkt_t; /**< alias for struct aom_codec_cx_pkt */
/*!\brief Number of samples, total/y/u/v when
* input bit-depth < stream bit-depth.*/
unsigned int samples_hbd[4];
/*!\brief sum squared error, total/y/u/v when
* input bit-depth < stream bit-depth.*/
uint64_t sse_hbd[4];
/*!\brief PSNR, total/y/u/v when
* input bit-depth < stream bit-depth.*/
double psnr_hbd[4];
} psnr; /**< data for PSNR packet */
aom_fixed_buf_t raw; /**< data for arbitrary packets */
} data; /**< packet data */
} aom_codec_cx_pkt_t; /**< alias for struct aom_codec_cx_pkt */
/*!\brief Rational Number
*
@ -173,11 +165,19 @@ typedef struct aom_rational {
int den; /**< fraction denominator */
} aom_rational_t; /**< alias for struct aom_rational */
/*!\brief Multi-pass Encoding Pass */
/*!\brief Multi-pass Encoding Pass
*
* AOM_RC_LAST_PASS is kept for backward compatibility.
* If passes is not given and pass==2, the codec will assume passes=2.
* For new code, it is recommended to use AOM_RC_SECOND_PASS and set
* the "passes" member to 2 via the key & val API for two-pass encoding.
*/
enum aom_enc_pass {
AOM_RC_ONE_PASS, /**< Single pass mode */
AOM_RC_FIRST_PASS, /**< First pass of multi-pass mode */
AOM_RC_LAST_PASS /**< Final pass of multi-pass mode */
AOM_RC_ONE_PASS = 0, /**< Single pass mode */
AOM_RC_FIRST_PASS = 1, /**< First pass of multi-pass mode */
AOM_RC_SECOND_PASS = 2, /**< Second pass of multi-pass mode */
AOM_RC_THIRD_PASS = 3, /**< Third pass of multi-pass mode */
AOM_RC_LAST_PASS = 2, /**< Final pass of two-pass mode */
};
/*!\brief Rate control mode */
@ -202,6 +202,22 @@ enum aom_kf_mode {
AOM_KF_DISABLED = 0 /**< Encoder does not place keyframes. */
};
/*!\brief Frame super-resolution mode. */
typedef enum {
/**< Frame super-resolution is disabled for all frames. */
AOM_SUPERRES_NONE,
/**< All frames are coded at the specified scale and super-resolved. */
AOM_SUPERRES_FIXED,
/**< All frames are coded at a random scale and super-resolved. */
AOM_SUPERRES_RANDOM,
/**< Super-resolution scale for each frame is determined based on the q index
of that frame. */
AOM_SUPERRES_QTHRESH,
/**< Full-resolution or super-resolution and the scale (in case of
super-resolution) are automatically selected for each frame. */
AOM_SUPERRES_AUTO,
} aom_superres_mode;
/*!\brief Encoder Config Options
*
* This type allows to enumerate and control flags defined for encoder control
@ -358,7 +374,8 @@ typedef struct cfg_options {
* /algo/_eflag_*. The lower order 16 bits are reserved for common use.
*/
typedef long aom_enc_frame_flags_t;
#define AOM_EFLAG_FORCE_KF (1 << 0) /**< Force this frame to be a keyframe */
/*!\brief Force this frame to be a keyframe */
#define AOM_EFLAG_FORCE_KF (1 << 0)
/*!\brief Encoder configuration structure
*
@ -546,10 +563,8 @@ typedef struct aom_codec_enc_cfg {
* Similar to spatial resampling, frame super-resolution integrates
* upscaling after the encode/decode process. Taking control of upscaling and
* using restoration filters should allow it to outperform normal resizing.
*
* Valid values are 0 to 4 as defined in enum SUPERRES_MODE.
*/
unsigned int rc_superres_mode;
aom_superres_mode rc_superres_mode;
/*!\brief Frame super-resolution denominator.
*
@ -559,7 +574,7 @@ typedef struct aom_codec_enc_cfg {
*
* Valid denominators are 8 to 16.
*
* Used only by SUPERRES_FIXED.
* Used only by AOM_SUPERRES_FIXED.
*/
unsigned int rc_superres_denominator;
@ -578,7 +593,7 @@ typedef struct aom_codec_enc_cfg {
* The q level threshold after which superres is used.
* Valid values are 1 to 63.
*
* Used only by SUPERRES_QTHRESH
* Used only by AOM_SUPERRES_QTHRESH
*/
unsigned int rc_superres_qthresh;
@ -587,7 +602,7 @@ typedef struct aom_codec_enc_cfg {
* The q level threshold after which superres is used for key frames.
* Valid values are 1 to 63.
*
* Used only by SUPERRES_QTHRESH
* Used only by AOM_SUPERRES_QTHRESH
*/
unsigned int rc_superres_kf_qthresh;
@ -617,7 +632,7 @@ typedef struct aom_codec_enc_cfg {
/*!\brief Target data rate
*
* Target bandwidth to use for this stream, in kilobits per second.
* Target bitrate to use for this stream, in kilobits per second.
*/
unsigned int rc_target_bitrate;
@ -651,25 +666,19 @@ typedef struct aom_codec_enc_cfg {
/*!\brief Rate control adaptation undershoot control
*
* This value, expressed as a percentage of the target bitrate,
* controls the maximum allowed adaptation speed of the codec.
* This factor controls the maximum amount of bits that can
* be subtracted from the target bitrate in order to compensate
* for prior overshoot.
* This value, controls the tolerance of the VBR algorithm to undershoot
* and is used as a trigger threshold for more aggressive adaptation of Q.
*
* Valid values in the range 0-1000.
* Valid values in the range 0-100.
*/
unsigned int rc_undershoot_pct;
/*!\brief Rate control adaptation overshoot control
*
* This value, expressed as a percentage of the target bitrate,
* controls the maximum allowed adaptation speed of the codec.
* This factor controls the maximum amount of bits that can
* be added to the target bitrate in order to compensate for
* prior undershoot.
* This value, controls the tolerance of the VBR algorithm to overshoot
* and is used as a trigger threshold for more aggressive adaptation of Q.
*
* Valid values in the range 0-1000.
* Valid values in the range 0-100.
*/
unsigned int rc_overshoot_pct;
@ -879,27 +888,11 @@ typedef struct aom_codec_enc_cfg {
*/
unsigned int use_fixed_qp_offsets;
/*!\brief Number of fixed QP offsets
*
* This defines the number of elements in the fixed_qp_offsets array.
*/
#define FIXED_QP_OFFSET_COUNT 5
/*!\brief Array of fixed QP offsets
/*!\brief Deprecated and ignored. DO NOT USE.
*
* This array specifies fixed QP offsets (range: 0 to 63) for frames at
* different levels of the pyramid. It is a comma-separated list of 5 values:
* - QP offset for keyframe
* - QP offset for ALTREF frame
* - QP offset for 1st level internal ARF
* - QP offset for 2nd level internal ARF
* - QP offset for 3rd level internal ARF
* Notes:
* - QP offset for leaf level frames is not explicitly specified. These frames
* use the worst quality allowed (--cq-level).
* - This option is only relevant for --end-usage=q.
* TODO(aomedia:3269): Remove fixed_qp_offsets in libaom v4.0.0.
*/
int fixed_qp_offsets[FIXED_QP_OFFSET_COUNT];
int fixed_qp_offsets[5];
/*!\brief Options defined per config file
*
@ -914,7 +907,7 @@ typedef struct aom_codec_enc_cfg {
* function directly, to ensure that the ABI version number parameter
* is properly initialized.
*
* If the library was configured with --disable-multithread, this call
* If the library was configured with -DCONFIG_MULTITHREAD=0, this call
* is not thread safe and should be guarded with a lock if being used
* in a multithreaded context.
*
@ -952,8 +945,8 @@ aom_codec_err_t aom_codec_enc_init_ver(aom_codec_ctx_t *ctx,
* \param[in] iface Pointer to the algorithm interface to use.
* \param[out] cfg Configuration buffer to populate.
* \param[in] usage Algorithm specific usage value. For AV1, must be
* set to AOM_USAGE_GOOD_QUALITY (0) or
* AOM_USAGE_REALTIME (1).
* set to AOM_USAGE_GOOD_QUALITY (0),
* AOM_USAGE_REALTIME (1), or AOM_USAGE_ALL_INTRA (2).
*
* \retval #AOM_CODEC_OK
* The configuration was populated.
@ -1012,6 +1005,8 @@ aom_fixed_buf_t *aom_codec_get_global_headers(aom_codec_ctx_t *ctx);
#define AOM_USAGE_GOOD_QUALITY (0)
/*!\brief usage parameter analogous to AV1 REALTIME mode. */
#define AOM_USAGE_REALTIME (1)
/*!\brief usage parameter analogous to AV1 all intra mode. */
#define AOM_USAGE_ALL_INTRA (2)
/*!\brief Encode a frame
*
@ -1019,15 +1014,20 @@ aom_fixed_buf_t *aom_codec_get_global_headers(aom_codec_ctx_t *ctx);
* time stamp (PTS) \ref MUST be strictly increasing.
*
* When the last frame has been passed to the encoder, this function should
* continue to be called, with the img parameter set to NULL. This will
* signal the end-of-stream condition to the encoder and allow it to encode
* any held buffers. Encoding is complete when aom_codec_encode() is called
* and aom_codec_get_cx_data() returns no data.
* continue to be called in a loop, with the img parameter set to NULL. This
* will signal the end-of-stream condition to the encoder and allow it to
* encode any held buffers. Encoding is complete when aom_codec_encode() is
* called with img set to NULL and aom_codec_get_cx_data() returns no data.
*
* \param[in] ctx Pointer to this instance's context
* \param[in] img Image data to encode, NULL to flush.
* \param[in] pts Presentation time stamp, in timebase units.
* \param[in] duration Duration to show frame, in timebase units.
* Encoding sample values outside the range
* [0..(1<<img->bit_depth)-1] is undefined behavior.
* \param[in] pts Presentation time stamp, in timebase units. If img
* is NULL, pts is ignored.
* \param[in] duration Duration to show frame, in timebase units. If img
* is not NULL, duration must be nonzero. If img is
* NULL, duration is ignored.
* \param[in] flags Flags to use for encoding this frame.
*
* \retval #AOM_CODEC_OK

View file

@ -0,0 +1,452 @@
/*
* Copyright (c) 2021, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AOM_AOM_AOM_EXTERNAL_PARTITION_H_
#define AOM_AOM_AOM_EXTERNAL_PARTITION_H_
/*!\defgroup aom_encoder AOMedia AOM/AV1 Encoder
* \ingroup aom
*
* @{
*/
#include <stdint.h>
/*!\file
* \brief Provides function pointer definitions for the external partition.
*
* \note The external partition API should be considered experimental. Until the
* external partition API is declared stable, breaking changes may be made to
* this API in a future libaom release.
*/
/*!\brief Current ABI version number
*
* \internal
* If this file is altered in any way that changes the ABI, this value
* must be bumped. Examples include, but are not limited to, changing
* types, removing or reassigning enums, adding/removing/rearranging
* fields to structures.
*/
#define AOM_EXT_PART_ABI_VERSION 8
#ifdef __cplusplus
extern "C" {
#endif
/*!\brief Abstract external partition model handler
*/
typedef void *aom_ext_part_model_t;
/*!\brief Number of features to determine whether to skip partition none and
* do partition split directly. The same as "FEATURE_SIZE_SMS_SPLIT".
*/
#define AOM_EXT_PART_SIZE_DIRECT_SPLIT 17
/*!\brief Number of features to use simple motion search to prune out
* rectangular partition in some direction. The same as
* "FEATURE_SIZE_SMS_PRUNE_PART".
*/
#define AOM_EXT_PART_SIZE_PRUNE_PART 25
/*!\brief Number of features to prune split and rectangular partition
* after PARTITION_NONE.
*/
#define AOM_EXT_PART_SIZE_PRUNE_NONE 4
/*!\brief Number of features to terminates partition after partition none using
* simple_motion_search features and the rate, distortion, and rdcost of
* PARTITION_NONE. The same as "FEATURE_SIZE_SMS_TERM_NONE".
*/
#define AOM_EXT_PART_SIZE_TERM_NONE 28
/*!\brief Number of features to terminates partition after partition split.
*/
#define AOM_EXT_PART_SIZE_TERM_SPLIT 31
/*!\brief Number of features to prune rectangular partition using stats
* collected after partition split.
*/
#define AOM_EXT_PART_SIZE_PRUNE_RECT 9
/*!\brief Number of features to prune AB partition using stats
* collected after rectangular partition..
*/
#define AOM_EXT_PART_SIZE_PRUNE_AB 10
/*!\brief Number of features to prune 4-way partition using stats
* collected after AB partition.
*/
#define AOM_EXT_PART_SIZE_PRUNE_4_WAY 18
/*!\brief Decision mode of the external partition model.
* AOM_EXT_PART_WHOLE_TREE: the external partition model should provide the
* whole partition tree for the superblock.
*
* AOM_EXT_PART_RECURSIVE: the external partition model provides the partition
* decision of the current block only. The decision process starts from
* the superblock size, down to the smallest block size (4x4) recursively.
*/
typedef enum aom_ext_part_decision_mode {
AOM_EXT_PART_WHOLE_TREE = 0,
AOM_EXT_PART_RECURSIVE = 1,
} aom_ext_part_decision_mode_t;
/*!\brief Config information sent to the external partition model.
*
* For example, the maximum superblock size determined by the sequence header.
*/
typedef struct aom_ext_part_config {
int superblock_size; ///< super block size (either 64x64 or 128x128)
} aom_ext_part_config_t;
/*!\brief Features pass to the external model to make partition decisions.
* Specifically, features collected before NONE partition.
* Features "f" are used to determine:
* partition_none_allowed, partition_horz_allowed, partition_vert_allowed,
* do_rectangular_split, do_square_split
* Features "f_part2" are used to determine:
* prune_horz, prune_vert.
*/
typedef struct aom_partition_features_before_none {
/*! features to determine whether skip partition none and do split directly */
float f[AOM_EXT_PART_SIZE_DIRECT_SPLIT];
/*! features to determine whether to prune rectangular partition */
float f_part2[AOM_EXT_PART_SIZE_PRUNE_PART];
} aom_partition_features_before_none_t;
/*!\brief Features pass to the external model to make partition decisions.
* Specifically, features collected after NONE partition.
*/
typedef struct aom_partition_features_none {
/*! features to prune split and rectangular partition */
float f[AOM_EXT_PART_SIZE_PRUNE_NONE];
/*! features to determine termination of partition */
float f_terminate[AOM_EXT_PART_SIZE_TERM_NONE];
} aom_partition_features_none_t;
/*!\brief Features pass to the external model to make partition decisions.
* Specifically, features collected after SPLIT partition.
*/
typedef struct aom_partition_features_split {
/*! features to determine termination of partition */
float f_terminate[AOM_EXT_PART_SIZE_TERM_SPLIT];
/*! features to determine pruning rect partition */
float f_prune_rect[AOM_EXT_PART_SIZE_PRUNE_RECT];
} aom_partition_features_split_t;
/*!\brief Features pass to the external model to make partition decisions.
* Specifically, features collected after RECTANGULAR partition.
*/
typedef struct aom_partition_features_rect {
/*! features to determine pruning AB partition */
float f[AOM_EXT_PART_SIZE_PRUNE_AB];
} aom_partition_features_rect_t;
/*!\brief Features pass to the external model to make partition decisions.
* Specifically, features collected after AB partition: HORZ_A, HORZ_B, VERT_A,
* VERT_B.
*/
typedef struct aom_partition_features_ab {
/*! features to determine pruning 4-way partition */
float f[AOM_EXT_PART_SIZE_PRUNE_4_WAY];
} aom_partition_features_ab_t;
/*!\brief Feature id to tell the external model the current stage in partition
* pruning and what features to use to make decisions accordingly.
*/
typedef enum {
AOM_EXT_PART_FEATURE_BEFORE_NONE,
AOM_EXT_PART_FEATURE_BEFORE_NONE_PART2,
AOM_EXT_PART_FEATURE_AFTER_NONE,
AOM_EXT_PART_FEATURE_AFTER_NONE_PART2,
AOM_EXT_PART_FEATURE_AFTER_SPLIT,
AOM_EXT_PART_FEATURE_AFTER_SPLIT_PART2,
AOM_EXT_PART_FEATURE_AFTER_RECT,
AOM_EXT_PART_FEATURE_AFTER_AB
} AOM_EXT_PART_FEATURE_ID;
/*!\brief Features collected from the tpl process.
*
* The tpl process collects information that help measure the inter-frame
* dependency.
* The tpl process is computed in the unit of tpl_bsize_1d (16x16).
* Therefore, the max number of units inside a superblock is
* 128x128 / (16x16) = 64. Change it if the tpl process changes.
*/
typedef struct aom_sb_tpl_features {
int available; ///< If tpl stats are available
int tpl_unit_length; ///< The block length of tpl process
int num_units; ///< The number of units inside the current superblock
int64_t intra_cost[64]; ///< The intra cost of each unit
int64_t inter_cost[64]; ///< The inter cost of each unit
int64_t mc_dep_cost[64]; ///< The motion compensated dependency cost
} aom_sb_tpl_features_t;
/*!\brief Features collected from the simple motion process.
*
* The simple motion process collects information by applying motion compensated
* prediction on each block.
* The block size is 16x16, which could be changed. If it is changed, update
* comments and the array size here.
*/
typedef struct aom_sb_simple_motion_features {
int unit_length; ///< The block length of the simple motion process
int num_units; ///< The number of units inside the current superblock
int block_sse[64]; ///< Sum of squared error of each unit
int block_var[64]; ///< Variance of each unit
} aom_sb_simple_motion_features_t;
/*!\brief Features of each super block.
*
* Features collected for each super block before partition search.
*/
typedef struct aom_sb_features {
/*! Features from motion search */
aom_sb_simple_motion_features_t motion_features;
/*! Features from tpl process */
aom_sb_tpl_features_t tpl_features;
} aom_sb_features_t;
/*!\brief Features pass to the external model to make partition decisions.
*
* The encoder sends these features to the external model through
* "func()" defined in .....
*
* NOTE: new member variables may be added to this structure in the future.
* Once new features are finalized, bump the major version of libaom.
*/
typedef struct aom_partition_features {
// Features for the current supervised multi-stage ML model.
/*! Feature ID to indicate active features */
AOM_EXT_PART_FEATURE_ID id;
/*! Features collected before NONE partition */
aom_partition_features_before_none_t before_part_none;
/*! Features collected after NONE partition */
aom_partition_features_none_t after_part_none;
/*! Features collected after SPLIT partition */
aom_partition_features_split_t after_part_split;
/*! Features collected after RECTANGULAR partition */
aom_partition_features_rect_t after_part_rect;
/*! Features collected after AB partition */
aom_partition_features_ab_t after_part_ab;
// Features for a new ML model.
aom_sb_features_t sb_features; ///< Features collected for the super block
int mi_row; ///< Mi_row position of the block
int mi_col; ///< Mi_col position of the block
int frame_width; ///< Frame width
int frame_height; ///< Frame height
int block_size; ///< As "BLOCK_SIZE" in av1/common/enums.h
/*!
* Valid partition types. A bitmask is used. "1" represents the
* corresponding type is vaild. The bitmask follows the enum order for
* PARTITION_TYPE in "enums.h" to represent one partition type at a bit.
* For example, 0x01 stands for only PARTITION_NONE is valid,
* 0x09 (00...001001) stands for PARTITION_NONE and PARTITION_SPLIT are valid.
*/
int valid_partition_types;
int update_type; ///< Frame update type, defined in ratectrl.h
int qindex; ///< Quantization index, range: [0, 255]
int rdmult; ///< Rate-distortion multiplier
int pyramid_level; ///< The level of this frame in the hierarchical structure
int has_above_block; ///< Has above neighbor block
int above_block_width; ///< Width of the above block, -1 if not exist
int above_block_height; ///< Height of the above block, -1 if not exist
int has_left_block; ///< Has left neighbor block
int left_block_width; ///< Width of the left block, -1 if not exist
int left_block_height; ///< Height of the left block, -1 if not exist
/*!
* The following parameters are collected from applying simple motion search.
* Sum of squared error (SSE) and variance of motion compensated residual
* are good indicators of block partitioning.
* If a block is a square, we also apply motion search for its 4 sub blocks.
* If not a square, their values are -1.
* If a block is able to split horizontally, we apply motion search and get
* stats for horizontal blocks. If not, their values are -1.
* If a block is able to split vertically, we apply motion search and get
* stats for vertical blocks. If not, their values are -1.
*/
unsigned int block_sse; ///< SSE of motion compensated residual
unsigned int block_var; ///< Variance of motion compensated residual
unsigned int sub_block_sse[4]; ///< SSE of sub blocks.
unsigned int sub_block_var[4]; ///< Variance of sub blocks.
unsigned int horz_block_sse[2]; ///< SSE of horz sub blocks
unsigned int horz_block_var[2]; ///< Variance of horz sub blocks
unsigned int vert_block_sse[2]; ///< SSE of vert sub blocks
unsigned int vert_block_var[2]; ///< Variance of vert sub blocks
/*!
* The following parameters are calculated from tpl model.
* If tpl model is not available, their values are -1.
*/
int64_t tpl_intra_cost; ///< Intra cost, ref to "TplDepStats" in tpl_model.h
int64_t tpl_inter_cost; ///< Inter cost in tpl model
int64_t tpl_mc_dep_cost; ///< Motion compensated dependency cost in tpl model
} aom_partition_features_t;
/*!\brief Partition decisions received from the external model.
*
* The encoder receives partition decisions and encodes the superblock
* with the given partition type.
* The encoder receives it from "func()" define in ....
*
* NOTE: new member variables may be added to this structure in the future.
* Once new features are finalized, bump the major version of libaom.
*/
typedef struct aom_partition_decision {
// Decisions for directly set partition types
int is_final_decision; ///< The flag whether it's the final decision
int num_nodes; ///< The number of leaf nodes
int partition_decision[2048]; ///< Partition decisions
int current_decision; ///< Partition decision for the current block
// Decisions for partition type pruning
int terminate_partition_search; ///< Terminate further partition search
int partition_none_allowed; ///< Allow partition none type
int partition_rect_allowed[2]; ///< Allow rectangular partitions
int do_rectangular_split; ///< Try rectangular split partition
int do_square_split; ///< Try square split partition
int prune_rect_part[2]; ///< Prune rectangular partition
int horza_partition_allowed; ///< Allow HORZ_A partitioin
int horzb_partition_allowed; ///< Allow HORZ_B partitioin
int verta_partition_allowed; ///< Allow VERT_A partitioin
int vertb_partition_allowed; ///< Allow VERT_B partitioin
int partition_horz4_allowed; ///< Allow HORZ4 partition
int partition_vert4_allowed; ///< Allow VERT4 partition
} aom_partition_decision_t;
/*!\brief Encoding stats for the given partition decision.
*
* The encoding stats collected by encoding the superblock with the
* given partition types.
* The encoder sends the stats to the external model for training
* or inference though "func()" defined in ....
*/
typedef struct aom_partition_stats {
int rate; ///< Rate cost of the block
int64_t dist; ///< Distortion of the block
int64_t rdcost; ///< Rate-distortion cost of the block
} aom_partition_stats_t;
/*!\brief Enum for return status.
*/
typedef enum aom_ext_part_status {
AOM_EXT_PART_OK = 0, ///< Status of success
AOM_EXT_PART_ERROR = 1, ///< Status of failure
AOM_EXT_PART_TEST = 2, ///< Status used for tests
} aom_ext_part_status_t;
/*!\brief Callback of creating an external partition model.
*
* The callback is invoked by the encoder to create an external partition
* model.
*
* \param[in] priv Callback's private data
* \param[in] part_config Config information pointer for model creation
* \param[out] ext_part_model Pointer to the model
*/
typedef aom_ext_part_status_t (*aom_ext_part_create_model_fn_t)(
void *priv, const aom_ext_part_config_t *part_config,
aom_ext_part_model_t *ext_part_model);
/*!\brief Callback of sending features to the external partition model.
*
* The callback is invoked by the encoder to send features to the external
* partition model.
*
* \param[in] ext_part_model The external model
* \param[in] part_features Pointer to the features
*/
typedef aom_ext_part_status_t (*aom_ext_part_send_features_fn_t)(
aom_ext_part_model_t ext_part_model,
const aom_partition_features_t *part_features);
/*!\brief Callback of receiving partition decisions from the external
* partition model.
*
* The callback is invoked by the encoder to receive partition decisions from
* the external partition model.
*
* \param[in] ext_part_model The external model
* \param[in] ext_part_decision Pointer to the partition decisions
*/
typedef aom_ext_part_status_t (*aom_ext_part_get_decision_fn_t)(
aom_ext_part_model_t ext_part_model,
aom_partition_decision_t *ext_part_decision);
/*!\brief Callback of sending stats to the external partition model.
*
* The callback is invoked by the encoder to send encoding stats to
* the external partition model.
*
* \param[in] ext_part_model The external model
* \param[in] ext_part_stats Pointer to the encoding stats
*/
typedef aom_ext_part_status_t (*aom_ext_part_send_partition_stats_fn_t)(
aom_ext_part_model_t ext_part_model,
const aom_partition_stats_t *ext_part_stats);
/*!\brief Callback of deleting the external partition model.
*
* The callback is invoked by the encoder to delete the external partition
* model.
*
* \param[in] ext_part_model The external model
*/
typedef aom_ext_part_status_t (*aom_ext_part_delete_model_fn_t)(
aom_ext_part_model_t ext_part_model);
/*!\brief Callback function set for external partition model.
*
* Uses can enable external partition model by registering a set of
* callback functions with the flag: AV1E_SET_EXTERNAL_PARTITION_MODEL
*/
typedef struct aom_ext_part_funcs {
/*!
* Create an external partition model.
*/
aom_ext_part_create_model_fn_t create_model;
/*!
* Send features to the external partition model to make partition decisions.
*/
aom_ext_part_send_features_fn_t send_features;
/*!
* Get partition decisions from the external partition model.
*/
aom_ext_part_get_decision_fn_t get_partition_decision;
/*!
* Send stats of the current partition to the external model.
*/
aom_ext_part_send_partition_stats_fn_t send_partition_stats;
/*!
* Delete the external partition model.
*/
aom_ext_part_delete_model_fn_t delete_model;
/*!
* The decision mode of the model.
*/
aom_ext_part_decision_mode_t decision_mode;
/*!
* Private data for the external partition model.
*/
void *priv;
} aom_ext_part_funcs_t;
/*!@} - end defgroup aom_encoder*/
#ifdef __cplusplus
} // extern "C"
#endif
#endif // AOM_AOM_AOM_EXTERNAL_PARTITION_H_

View file

@ -58,7 +58,7 @@ typedef struct aom_codec_frame_buffer {
* must return 0. Any failure the callback must return a value less than 0.
*
* \param[in] priv Callback's private data
* \param[in] new_size Size in bytes needed by the buffer
* \param[in] min_size Size in bytes needed by the buffer
* \param[in,out] fb Pointer to aom_codec_frame_buffer_t
*/
typedef int (*aom_get_frame_buffer_cb_fn_t)(void *priv, size_t min_size,

View file

@ -48,6 +48,11 @@ typedef enum aom_img_fmt {
AOM_IMG_FMT_AOMI420 = AOM_IMG_FMT_PLANAR | 4,
AOM_IMG_FMT_I422 = AOM_IMG_FMT_PLANAR | 5,
AOM_IMG_FMT_I444 = AOM_IMG_FMT_PLANAR | 6,
/*!\brief Allows detection of the presence of AOM_IMG_FMT_NV12 at compile time.
*/
#define AOM_HAVE_IMG_FMT_NV12 1
AOM_IMG_FMT_NV12 =
AOM_IMG_FMT_PLANAR | 7, /**< 4:2:0 with U and V interleaved */
AOM_IMG_FMT_I42016 = AOM_IMG_FMT_I420 | AOM_IMG_FMT_HIGHBITDEPTH,
AOM_IMG_FMT_YV1216 = AOM_IMG_FMT_YV12 | AOM_IMG_FMT_HIGHBITDEPTH,
AOM_IMG_FMT_I42216 = AOM_IMG_FMT_I422 | AOM_IMG_FMT_HIGHBITDEPTH,
@ -124,8 +129,12 @@ typedef enum aom_matrix_coefficients {
/*!\brief List of supported color range */
typedef enum aom_color_range {
AOM_CR_STUDIO_RANGE = 0, /**< Y [16..235], UV [16..240] */
AOM_CR_FULL_RANGE = 1 /**< YUV/RGB [0..255] */
AOM_CR_STUDIO_RANGE = 0, /**<- Y [16..235], UV [16..240] (bit depth 8) */
/**<- Y [64..940], UV [64..960] (bit depth 10) */
/**<- Y [256..3760], UV [256..3840] (bit depth 12) */
AOM_CR_FULL_RANGE = 1 /**<- YUV/RGB [0..255] (bit depth 8) */
/**<- YUV/RGB [0..1023] (bit depth 10) */
/**<- YUV/RGB [0..4095] (bit depth 12) */
} aom_color_range_t; /**< alias for enum aom_color_range */
/*!\brief List of chroma sample positions */
@ -195,10 +204,12 @@ typedef struct aom_image {
unsigned int y_chroma_shift; /**< subsampling order, Y */
/* Image data pointers. */
#define AOM_PLANE_PACKED 0 /**< To be used for all packed formats */
#define AOM_PLANE_Y 0 /**< Y (Luminance) plane */
#define AOM_PLANE_U 1 /**< U (Chroma) plane */
#define AOM_PLANE_V 2 /**< V (Chroma) plane */
#define AOM_PLANE_PACKED 0 /**< To be used for all packed formats */
#define AOM_PLANE_Y 0 /**< Y (Luminance) plane */
#define AOM_PLANE_U 1 /**< U (Chroma) plane */
#define AOM_PLANE_V 2 /**< V (Chroma) plane */
/* planes[AOM_PLANE_V] = NULL and stride[AOM_PLANE_V] = 0 when fmt ==
* AOM_IMG_FMT_NV12 */
unsigned char *planes[3]; /**< pointer to the top left pixel for each plane */
int stride[3]; /**< stride between rows for each plane */
size_t sz; /**< data size */
@ -300,7 +311,8 @@ aom_image_t *aom_img_alloc_with_border(aom_image_t *img, aom_img_fmt_t fmt,
/*!\brief Set the rectangle identifying the displayed portion of the image
*
* Updates the displayed rectangle (aka viewport) on the image surface to
* match the specified coordinates and size.
* match the specified coordinates and size. Specifically, sets img->d_w,
* img->d_h, and elements of the img->planes[] array.
*
* \param[in] img Image descriptor
* \param[in] x leftmost column
@ -309,7 +321,7 @@ aom_image_t *aom_img_alloc_with_border(aom_image_t *img, aom_img_fmt_t fmt,
* \param[in] h height
* \param[in] border A border that is padded on four sides of the image.
*
* \return 0 if the requested rectangle is valid, nonzero otherwise.
* \return 0 if the requested rectangle is valid, nonzero (-1) otherwise.
*/
int aom_img_set_rect(aom_image_t *img, unsigned int x, unsigned int y,
unsigned int w, unsigned int h, unsigned int border);
@ -360,6 +372,9 @@ int aom_img_plane_height(const aom_image_t *img, int plane);
* \param[in] data Metadata contents
* \param[in] sz Metadata contents size
* \param[in] insert_flag Metadata insert flag
*
* \return Returns 0 on success. If img or data is NULL, sz is 0, or memory
* allocation fails, it returns -1.
*/
int aom_img_add_metadata(aom_image_t *img, uint32_t type, const uint8_t *data,
size_t sz, aom_metadata_insert_flags_t insert_flag);
@ -410,6 +425,9 @@ void aom_img_remove_metadata(aom_image_t *img);
* \param[in] data Metadata data pointer
* \param[in] sz Metadata size
* \param[in] insert_flag Metadata insert flag
*
* \return Returns the newly allocated aom_metadata struct. If data is NULL,
* sz is 0, or memory allocation fails, it returns NULL.
*/
aom_metadata_t *aom_img_metadata_alloc(uint32_t type, const uint8_t *data,
size_t sz,

View file

@ -22,22 +22,7 @@
#define AOM_INLINE inline
#endif
#if defined(AOM_EMULATE_INTTYPES)
typedef signed char int8_t;
typedef signed short int16_t;
typedef signed int int32_t;
typedef unsigned char uint8_t;
typedef unsigned short uint16_t;
typedef unsigned int uint32_t;
#ifndef _UINTPTR_T_DEFINED
typedef size_t uintptr_t;
#endif
#else
/* Most platforms have the C99 standard integer types. */
/* Assume platforms have the C99 standard integer types. */
#if defined(__cplusplus)
#if !defined(__STDC_FORMAT_MACROS)
@ -49,27 +34,7 @@ typedef size_t uintptr_t;
#endif // __cplusplus
#include <stdint.h>
#endif
/* VS2010 defines stdint.h, but not inttypes.h */
#if defined(_MSC_VER) && _MSC_VER < 1800
#define PRId64 "I64d"
#else
#include <inttypes.h>
#endif
#if !defined(INT8_MAX)
#define INT8_MAX 127
#endif
#if !defined(INT32_MAX)
#define INT32_MAX 2147483647
#endif
#if !defined(INT32_MIN)
#define INT32_MIN (-2147483647 - 1)
#endif
#if defined(__cplusplus)
extern "C" {

View file

@ -18,10 +18,23 @@
*/
#include "aom/aom.h"
#include "aom/aom_encoder.h"
#include "aom/aom_external_partition.h"
/*!\file
* \brief Provides definitions for using AOM or AV1 encoder algorithm within the
* aom Codec Interface.
*
* Several interfaces are excluded with CONFIG_REALTIME_ONLY build:
* Global motion
* Warped motion
* OBMC
* TPL model
* Loop restoration
*
* The following features are also disabled with CONFIG_REALTIME_ONLY:
* CNN
* 4X rectangular blocks
* 4X rectangular transform in intra prediction
*/
#ifdef __cplusplus
@ -31,11 +44,19 @@ extern "C" {
/*!\name Algorithm interface for AV1
*
* This interface provides the capability to encode raw AV1 streams.
* @{
*@{
*/
/*!\brief A single instance of the AV1 encoder.
*\deprecated This access mechanism is provided for backwards compatibility;
* prefer aom_codec_av1_cx().
*/
extern aom_codec_iface_t aom_codec_av1_cx_algo;
/*!\brief The interface to the AV1 encoder.
*/
extern aom_codec_iface_t *aom_codec_av1_cx(void);
/*!@} - end algorithm interface member group*/
/*!@} - end algorithm interface member group */
/*
* Algorithm Flags
@ -147,6 +168,7 @@ extern aom_codec_iface_t *aom_codec_av1_cx(void);
*
* This set of macros define the control functions available for AVx
* encoder interface.
* The range of encode control ID is 7-229(max).
*
* \sa #aom_codec_control(aom_codec_ctx_t *ctx, int ctrl_id, ...)
*/
@ -185,9 +207,14 @@ enum aome_enc_control_id {
* encoding process, values greater than 0 will increase encoder speed at
* the expense of quality.
*
* Valid range: 0..8. 0 runs the slowest, and 8 runs the fastest;
* Valid range: 0..10. 0 runs the slowest, and 10 runs the fastest;
* quality improves as speed decreases (since more compression
* possibilities are explored).
*
* NOTE: 10 is only allowed in AOM_USAGE_REALTIME. In AOM_USAGE_GOOD_QUALITY
* and AOM_USAGE_ALL_INTRA, 9 is the highest allowed value. However,
* AOM_USAGE_GOOD_QUALITY treats 7..9 the same as 6. Also, AOM_USAGE_REALTIME
* treats 0..4 the same as 5.
*/
AOME_SET_CPUUSED = 13,
@ -201,7 +228,14 @@ enum aome_enc_control_id {
/* NOTE: enum 15 unused */
/*!\brief Codec control function to set sharpness, unsigned int parameter.
/*!\brief Codec control function to set the sharpness parameter,
* unsigned int parameter.
*
* This parameter controls the level at which rate-distortion optimization of
* transform coefficients favours sharpness in the block.
*
* Valid range: 0..7. The default is 0. Values 1-7 will avoid eob and skip
* block optimization and will change rdmult in favour of block sharpness.
*/
AOME_SET_SHARPNESS = AOME_SET_ENABLEAUTOALTREF + 2, // 16
@ -241,6 +275,8 @@ enum aome_enc_control_id {
/*!\brief Codec control function to set visual tuning, aom_tune_metric (int)
* parameter
*
* The default is AOM_TUNE_PSNR.
*/
AOME_SET_TUNING = AOME_SET_ARNR_STRENGTH + 2, // 24
@ -365,6 +401,8 @@ enum aome_enc_control_id {
*
* - 0 = disable
* - 1 = enable (default)
*
* \note Excluded from CONFIG_REALTIME_ONLY build.
*/
AV1E_SET_ENABLE_TPL_MODEL = 35,
@ -372,7 +410,8 @@ enum aome_enc_control_id {
* unsigned int parameter
*
* - 0 = disable
* - 1 = enable (default)
* - 1 = enable without overlay (default)
* - 2 = enable with overlay
*/
AV1E_SET_ENABLE_KEYFRAME_FILTERING = 36,
@ -382,7 +421,7 @@ enum aome_enc_control_id {
* AV1 has a bitstream feature to reduce decoding dependency between frames
* by turning off backward update of probability context used in encoding
* and decoding. This allows staged parallel processing of more than one
* video frames in the decoder. This control function provides a mean to
* video frames in the decoder. This control function provides a means to
* turn this feature on or off for bitstreams produced by encoder.
*
* - 0 = disable (default)
@ -418,10 +457,12 @@ enum aome_enc_control_id {
* AV1 has a segment based feature that allows encoder to adaptively change
* quantization parameter for each segment within a frame to improve the
* subjective quality. This control makes encoder operate in one of the
* several AQ_modes supported.
* several AQ modes supported.
*
* - 0 = disable (default)
* - 1 = enable
* - 1 = variance
* - 2 = complexity
* - 3 = cyclic refresh
*/
AV1E_SET_AQ_MODE = 40,
@ -429,7 +470,7 @@ enum aome_enc_control_id {
* int parameter
*
* One AV1 encoder speed feature is to enable quality boost by lowering
* frame level Q periodically. This control function provides a mean to
* frame level Q periodically. This control function provides a means to
* turn on/off this feature.
*
* - 0 = disable (default)
@ -450,6 +491,7 @@ enum aome_enc_control_id {
*
* - AOM_CONTENT_DEFAULT = Regular video content (default)
* - AOM_CONTENT_SCREEN = Screen capture content
* - AOM_CONTENT_FILM = Film content
*/
AV1E_SET_TUNE_CONTENT = 43,
@ -570,18 +612,18 @@ enum aome_enc_control_id {
AV1E_SET_RENDER_SIZE = 53,
/*!\brief Control to set target sequence level index for a certain operating
* point(OP), int parameter
* Possible values are in the form of "ABxy"(pad leading zeros if less than
* 4 digits).
* point (OP), int parameter
* Possible values are in the form of "ABxy".
* - AB: OP index.
* - xy: Target level index for the OP. Can be values 0~23(corresponding to
* level 2.0 ~ 7.3) or 24(keep level stats only for level monitoring) or
* 31(maximum level parameter, no level-based constraints).
* - xy: Target level index for the OP. Can be values 0~23 (corresponding to
* level 2.0 ~ 7.3, note levels 2.2, 2.3, 3.2, 3.3, 4.2, 4.3, 7.0, 7.1, 7.2
* & 7.3 are undefined) or 24 (keep level stats only for level monitoring)
* or 31 (maximum level parameter, no level-based constraints).
*
* E.g.:
* - "0" means target level index 0 for the 0th OP;
* - "111" means target level index 11 for the 1st OP;
* - "1021" means target level index 21 for the 10th OP.
* - "0" means target level index 0 (2.0) for the 0th OP;
* - "109" means target level index 9 (4.1) for the 1st OP;
* - "1019" means target level index 19 (6.3) for the 10th OP.
*
* If the target level is not specified for an OP, the maximum level parameter
* of 31 is used as default.
@ -617,7 +659,8 @@ enum aome_enc_control_id {
* in-loop filter aiming to remove coding artifacts
*
* - 0 = disable
* - 1 = enable (default)
* - 1 = enable for all frames (default)
* - 2 = disable for non-reference frames
*/
AV1E_SET_ENABLE_CDEF = 58,
@ -626,6 +669,8 @@ enum aome_enc_control_id {
*
* - 0 = disable
* - 1 = enable (default)
*
* \note Excluded from CONFIG_REALTIME_ONLY build.
*/
AV1E_SET_ENABLE_RESTORATION = 59,
@ -641,6 +686,8 @@ enum aome_enc_control_id {
*
* - 0 = disable
* - 1 = enable (default)
*
* \note Excluded from CONFIG_REALTIME_ONLY build.
*/
AV1E_SET_ENABLE_OBMC = 61,
@ -847,7 +894,17 @@ enum aome_enc_control_id {
*/
AV1E_SET_ENABLE_FLIP_IDTX = 81,
/* Note: enum value 82 unused */
/*!\brief Codec control function to turn on / off rectangular transforms, int
* parameter
*
* This will enable or disable usage of rectangular transforms. NOTE:
* Rectangular transforms only enabled when corresponding rectangular
* partitions are.
*
* - 0 = disable
* - 1 = enable (default)
*/
AV1E_SET_ENABLE_RECT_TX = 82,
/*!\brief Codec control function to turn on / off dist-wtd compound mode
* at sequence level, int parameter
@ -892,7 +949,7 @@ enum aome_enc_control_id {
AV1E_SET_ENABLE_DUAL_FILTER = 86,
/*!\brief Codec control function to turn on / off delta quantization in chroma
* planes usage for a sequence, int parameter
* planes for a sequence, int parameter
*
* - 0 = disable (default)
* - 1 = enable
@ -960,6 +1017,8 @@ enum aome_enc_control_id {
*
* - 0 = disable
* - 1 = enable (default)
*
* \note Excluded from CONFIG_REALTIME_ONLY build.
*/
AV1E_SET_ENABLE_GLOBAL_MOTION = 95,
@ -968,6 +1027,8 @@ enum aome_enc_control_id {
*
* - 0 = disable
* - 1 = enable (default)
*
* \note Excluded from CONFIG_REALTIME_ONLY build.
*/
AV1E_SET_ENABLE_WARPED_MOTION = 96,
@ -979,15 +1040,14 @@ enum aome_enc_control_id {
*
* - 0 = disable
* - 1 = enable (default)
*
* \note Excluded from CONFIG_REALTIME_ONLY build.
*/
AV1E_SET_ALLOW_WARPED_MOTION = 97,
/*!\brief Codec control function to turn on / off filter intra usage at
* sequence level, int parameter
*
* \attention If AV1E_SET_ENABLE_FILTER_INTRA is 0, then this flag is
* forced to 0.
*
* - 0 = disable
* - 1 = enable (default)
*/
@ -1025,8 +1085,6 @@ enum aome_enc_control_id {
/*!\brief Codec control function to turn on / off frame superresolution, int
* parameter
*
* \attention If AV1E_SET_ENABLE_SUPERRES is 0, then this flag is forced to 0.
*
* - 0 = disable
* - 1 = enable (default)
*/
@ -1061,7 +1119,9 @@ enum aome_enc_control_id {
*
* - 0 = deltaq signaling off
* - 1 = use modulation to maximize objective quality (default)
* - 2 = use modulation to maximize perceptual quality
* - 2 = use modulation for local test
* - 3 = use modulation for key frame perceptual quality optimization
* - 4 = use modulation for user rating based perceptual quality optimization
*/
AV1E_SET_DELTAQ_MODE = 107,
@ -1143,7 +1203,7 @@ enum aome_enc_control_id {
/*!\brief Control to select maximum height for the GF group pyramid structure,
* unsigned int parameter
*
* Valid range: 0..4
* Valid range: 0..5
*/
AV1E_SET_GF_MAX_PYRAMID_HEIGHT = 123,
@ -1158,9 +1218,6 @@ enum aome_enc_control_id {
parameter */
AV1E_SET_REDUCED_REFERENCE_SET = 125,
/* NOTE: enums 126-139 unused */
/* NOTE: Need a gap in enum values to avoud conflict with 128, 129, 130 */
/*!\brief Control to set frequency of the cost updates for coefficients,
* unsigned int parameter
*
@ -1169,7 +1226,7 @@ enum aome_enc_control_id {
* - 2 = update at tile level
* - 3 = turn off
*/
AV1E_SET_COEFF_COST_UPD_FREQ = 140,
AV1E_SET_COEFF_COST_UPD_FREQ = 126,
/*!\brief Control to set frequency of the cost updates for mode, unsigned int
* parameter
@ -1179,7 +1236,7 @@ enum aome_enc_control_id {
* - 2 = update at tile level
* - 3 = turn off
*/
AV1E_SET_MODE_COST_UPD_FREQ = 141,
AV1E_SET_MODE_COST_UPD_FREQ = 127,
/*!\brief Control to set frequency of the cost updates for motion vectors,
* unsigned int parameter
@ -1189,7 +1246,7 @@ enum aome_enc_control_id {
* - 2 = update at tile level
* - 3 = turn off
*/
AV1E_SET_MV_COST_UPD_FREQ = 142,
AV1E_SET_MV_COST_UPD_FREQ = 128,
/*!\brief Control to set bit mask that specifies which tier each of the 32
* possible operating points conforms to, unsigned int parameter
@ -1197,37 +1254,37 @@ enum aome_enc_control_id {
* - 0 = main tier (default)
* - 1 = high tier
*/
AV1E_SET_TIER_MASK = 143,
AV1E_SET_TIER_MASK = 129,
/*!\brief Control to set minimum compression ratio, unsigned int parameter
* Take integer values. If non-zero, encoder will try to keep the compression
* ratio of each frame to be higher than the given value divided by 100.
* E.g. 850 means minimum compression ratio of 8.5.
*/
AV1E_SET_MIN_CR = 144,
AV1E_SET_MIN_CR = 130,
/* NOTE: enums 145-149 unused */
/*!\brief Codec control function to set the layer id, aom_svc_layer_id_t*
* parameter
*/
AV1E_SET_SVC_LAYER_ID = 150,
AV1E_SET_SVC_LAYER_ID = 131,
/*!\brief Codec control function to set SVC paramaeters, aom_svc_params_t*
* parameter
*/
AV1E_SET_SVC_PARAMS = 151,
AV1E_SET_SVC_PARAMS = 132,
/*!\brief Codec control function to set reference frame config:
* the ref_idx and the refresh flags for each buffer slot.
* aom_svc_ref_frame_config_t* parameter
*/
AV1E_SET_SVC_REF_FRAME_CONFIG = 152,
AV1E_SET_SVC_REF_FRAME_CONFIG = 133,
/*!\brief Codec control function to set the path to the VMAF model used when
* tuning the encoder for VMAF, const char* parameter
*/
AV1E_SET_VMAF_MODEL_PATH = 153,
AV1E_SET_VMAF_MODEL_PATH = 134,
/*!\brief Codec control function to enable EXT_TILE_DEBUG in AV1 encoder,
* unsigned int parameter
@ -1237,7 +1294,7 @@ enum aome_enc_control_id {
*
* \note This is only used in lightfield example test.
*/
AV1E_ENABLE_EXT_TILE_DEBUG = 154,
AV1E_ENABLE_EXT_TILE_DEBUG = 135,
/*!\brief Codec control function to enable the superblock multipass unit test
* in AV1 to ensure that the encoder does not leak state between different
@ -1248,14 +1305,150 @@ enum aome_enc_control_id {
*
* \note This is only used in sb_multipass unit test.
*/
AV1E_ENABLE_SB_MULTIPASS_UNIT_TEST = 155,
AV1E_ENABLE_SB_MULTIPASS_UNIT_TEST = 136,
/*!\brief Control to select minimum height for the GF group pyramid structure,
* unsigned int parameter
*
* Valid values: 0..4
* Valid values: 0..5
*/
AV1E_SET_GF_MIN_PYRAMID_HEIGHT = 156,
AV1E_SET_GF_MIN_PYRAMID_HEIGHT = 137,
/*!\brief Control to set average complexity of the corpus in the case of
* single pass vbr based on LAP, unsigned int parameter
*/
AV1E_SET_VBR_CORPUS_COMPLEXITY_LAP = 138,
/*!\brief Control to get baseline gf interval
*/
AV1E_GET_BASELINE_GF_INTERVAL = 139,
/*\brief Control to set encoding the denoised frame from denoise-noise-level
*
* - 0 = disabled/encode the original frame
* - 1 = enabled/encode the denoised frame (default)
*/
AV1E_SET_ENABLE_DNL_DENOISING = 140,
/*!\brief Codec control function to turn on / off D45 to D203 intra mode
* usage, int parameter
*
* This will enable or disable usage of D45 to D203 intra modes, which are a
* subset of directional modes. This control has no effect if directional
* modes are disabled (AV1E_SET_ENABLE_DIRECTIONAL_INTRA set to 0).
*
* - 0 = disable
* - 1 = enable (default)
*/
AV1E_SET_ENABLE_DIAGONAL_INTRA = 141,
/*!\brief Control to set frequency of the cost updates for intrabc motion
* vectors, unsigned int parameter
*
* - 0 = update at SB level (default)
* - 1 = update at SB row level in tile
* - 2 = update at tile level
* - 3 = turn off
*/
AV1E_SET_DV_COST_UPD_FREQ = 142,
/*!\brief Codec control to set the path for partition stats read and write.
* const char * parameter.
*/
AV1E_SET_PARTITION_INFO_PATH = 143,
/*!\brief Codec control to use an external partition model
* A set of callback functions is passed through this control
* to let the encoder encode with given partitions.
*/
AV1E_SET_EXTERNAL_PARTITION = 144,
/*!\brief Codec control function to turn on / off directional intra mode
* usage, int parameter
*
* - 0 = disable
* - 1 = enable (default)
*/
AV1E_SET_ENABLE_DIRECTIONAL_INTRA = 145,
/*!\brief Control to turn on / off transform size search.
*
* - 0 = disable, transforms always have the largest possible size
* - 1 = enable, search for the best transform size for each block (default)
*/
AV1E_SET_ENABLE_TX_SIZE_SEARCH = 146,
/*!\brief Codec control function to set reference frame compound prediction.
* aom_svc_ref_frame_comp_pred_t* parameter
*/
AV1E_SET_SVC_REF_FRAME_COMP_PRED = 147,
/*!\brief Set --deltaq-mode strength.
*
* Valid range: [0, 1000]
*/
AV1E_SET_DELTAQ_STRENGTH = 148,
/*!\brief Codec control to control loop filter
*
* - 0 = Loop filter is disabled for all frames
* - 1 = Loop filter is enabled for all frames
* - 2 = Loop filter is disabled for non-reference frames
* - 3 = Loop filter is disabled for the frames with low motion
*/
AV1E_SET_LOOPFILTER_CONTROL = 149,
/*!\brief Codec control function to get the loopfilter chosen by the encoder,
* int* parameter
*/
AOME_GET_LOOPFILTER_LEVEL = 150,
/*!\brief Codec control to automatically turn off several intra coding tools,
* unsigned int parameter
* - 0 = do not use the feature
* - 1 = enable the automatic decision to turn off several intra tools
*/
AV1E_SET_AUTO_INTRA_TOOLS_OFF = 151,
/*!\brief Codec control function to set flag for rate control used by external
* encoders.
* - 1 = Enable rate control for external encoders. This will disable content
* dependency in rate control and cyclic refresh.
* - 0 = Default. Disable rate control for external encoders.
*/
AV1E_SET_RTC_EXTERNAL_RC = 152,
/*!\brief Codec control function to enable frame parallel multi-threading
* of the encoder, unsigned int parameter
*
* - 0 = disable (default)
* - 1 = enable
*/
AV1E_SET_FP_MT = 153,
/*!\brief Codec control to enable actual frame parallel encode or
* simulation of frame parallel encode in FPMT unit test, unsigned int
* parameter
*
* - 0 = simulate frame parallel encode
* - 1 = actual frame parallel encode (default)
*
* \note This is only used in FPMT unit test.
*/
AV1E_SET_FP_MT_UNIT_TEST = 154,
/*!\brief Codec control function to get the target sequence level index for
* each operating point. int* parameter. There can be at most 32 operating
* points. The results will be written into a provided integer array of
* sufficient size. If a target level is not set, the result will be 31.
* Please refer to https://aomediacodec.github.io/av1-spec/#levels for more
* details on level definitions and indices.
*/
AV1E_GET_TARGET_SEQ_LEVEL_IDX = 155,
// Any new encoder control IDs should be added above.
// Maximum allowed encoder control ID is 229.
// No encoder control ID should be added below.
};
/*!\brief aom 1-D scaling mode
@ -1266,7 +1459,10 @@ typedef enum aom_scaling_mode_1d {
AOME_NORMAL = 0,
AOME_FOURFIVE = 1,
AOME_THREEFIVE = 2,
AOME_ONETWO = 3
AOME_THREEFOUR = 3,
AOME_ONEFOUR = 4,
AOME_ONEEIGHT = 5,
AOME_ONETWO = 6
} AOM_SCALING_MODE;
/*!\brief Max number of segments
@ -1323,6 +1519,7 @@ typedef struct aom_scaling_mode {
typedef enum {
AOM_CONTENT_DEFAULT,
AOM_CONTENT_SCREEN,
AOM_CONTENT_FILM,
AOM_CONTENT_INVALID
} aom_tune_content;
@ -1344,9 +1541,28 @@ typedef enum {
/* NOTE: enums 2 and 3 unused */
AOM_TUNE_VMAF_WITH_PREPROCESSING = 4,
AOM_TUNE_VMAF_WITHOUT_PREPROCESSING = 5,
AOM_TUNE_VMAF_MAX_GAIN = 6
AOM_TUNE_VMAF_MAX_GAIN = 6,
AOM_TUNE_VMAF_NEG_MAX_GAIN = 7,
AOM_TUNE_BUTTERAUGLI = 8,
} aom_tune_metric;
/*!\brief Distortion metric to use for RD optimization.
*
* Changes the encoder to use a different distortion metric for RD search. Note
* that this value operates on a "lower level" compared to aom_tune_metric - it
* affects the distortion metric inside a block, while aom_tune_metric only
* affects RD across blocks.
*
*/
typedef enum {
// Use PSNR for in-block rate-distortion optimization.
AOM_DIST_METRIC_PSNR,
// Use quantization matrix-weighted PSNR for in-block rate-distortion
// optimization. If --enable-qm=1 is not specified, this falls back to
// behaving in the same way as AOM_DIST_METRIC_PSNR.
AOM_DIST_METRIC_QM_PSNR,
} aom_dist_metric;
#define AOM_MAX_LAYERS 32 /**< Max number of layers */
#define AOM_MAX_SS_LAYERS 4 /**< Max number of spatial layers */
#define AOM_MAX_TS_LAYERS 8 /**< Max number of temporal layers */
@ -1381,6 +1597,13 @@ typedef struct aom_svc_ref_frame_config {
int refresh[8]; /**< Refresh flag for each of the 8 slots. */
} aom_svc_ref_frame_config_t;
/*!brief Parameters for setting ref frame compound prediction */
typedef struct aom_svc_ref_frame_comp_pred {
// Use compound prediction for the ref_frame pairs GOLDEN_LAST (0),
// LAST2_LAST (1), and ALTREF_LAST (2).
int use_comp_pred[3]; /**<Compound reference flag. */
} aom_svc_ref_frame_comp_pred_t;
/*!\cond */
/*!\brief Encoder control function parameter type
*
@ -1414,15 +1637,18 @@ AOM_CTRL_USE_TYPE(AOME_SET_CPUUSED, int)
AOM_CTRL_USE_TYPE(AOME_SET_ENABLEAUTOALTREF, unsigned int)
#define AOM_CTRL_AOME_SET_ENABLEAUTOALTREF
AOM_CTRL_USE_TYPE(AOME_SET_ENABLEAUTOBWDREF, unsigned int)
#define AOM_CTRL_AOME_SET_ENABLEAUTOBWDREF
AOM_CTRL_USE_TYPE(AOME_SET_SHARPNESS, unsigned int)
#define AOM_CTRL_AOME_SET_SHARPNESS
AOM_CTRL_USE_TYPE(AOME_SET_STATIC_THRESHOLD, unsigned int)
#define AOM_CTRL_AOME_SET_STATIC_THRESHOLD
AOM_CTRL_USE_TYPE(AOME_GET_LAST_QUANTIZER, int *)
#define AOM_CTRL_AOME_GET_LAST_QUANTIZER
AOM_CTRL_USE_TYPE(AOME_GET_LAST_QUANTIZER_64, int *)
#define AOM_CTRL_AOME_GET_LAST_QUANTIZER_64
AOM_CTRL_USE_TYPE(AOME_SET_ARNR_MAXFRAMES, unsigned int)
#define AOM_CTRL_AOME_SET_ARNR_MAXFRAMES
@ -1435,6 +1661,25 @@ AOM_CTRL_USE_TYPE(AOME_SET_TUNING, int) /* aom_tune_metric */
AOM_CTRL_USE_TYPE(AOME_SET_CQ_LEVEL, unsigned int)
#define AOM_CTRL_AOME_SET_CQ_LEVEL
AOM_CTRL_USE_TYPE(AOME_SET_MAX_INTRA_BITRATE_PCT, unsigned int)
#define AOM_CTRL_AOME_SET_MAX_INTRA_BITRATE_PCT
AOM_CTRL_USE_TYPE(AOME_SET_NUMBER_SPATIAL_LAYERS, int)
#define AOM_CTRL_AOME_SET_NUMBER_SPATIAL_LAYERS
// TODO(aomedia:3231): Deprecated. Remove it.
#define AOME_CTRL_AOME_SET_NUMBER_SPATIAL_LAYERS
AOM_CTRL_USE_TYPE(AOME_SET_MAX_INTER_BITRATE_PCT, unsigned int)
#define AOM_CTRL_AV1E_SET_MAX_INTER_BITRATE_PCT
// TODO(aomedia:3231): Deprecated. Remove it.
#define AOM_CTRL_AOME_SET_MAX_INTER_BITRATE_PCT
AOM_CTRL_USE_TYPE(AV1E_SET_GF_CBR_BOOST_PCT, unsigned int)
#define AOM_CTRL_AV1E_SET_GF_CBR_BOOST_PCT
AOM_CTRL_USE_TYPE(AV1E_SET_LOSSLESS, unsigned int)
#define AOM_CTRL_AV1E_SET_LOSSLESS
AOM_CTRL_USE_TYPE(AV1E_SET_ROW_MT, unsigned int)
#define AOM_CTRL_AV1E_SET_ROW_MT
@ -1450,26 +1695,68 @@ AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_TPL_MODEL, unsigned int)
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_KEYFRAME_FILTERING, unsigned int)
#define AOM_CTRL_AV1E_SET_ENABLE_KEYFRAME_FILTERING
AOM_CTRL_USE_TYPE(AOME_GET_LAST_QUANTIZER, int *)
#define AOM_CTRL_AOME_GET_LAST_QUANTIZER
AOM_CTRL_USE_TYPE(AV1E_SET_FRAME_PARALLEL_DECODING, unsigned int)
#define AOM_CTRL_AV1E_SET_FRAME_PARALLEL_DECODING
AOM_CTRL_USE_TYPE(AOME_GET_LAST_QUANTIZER_64, int *)
#define AOM_CTRL_AOME_GET_LAST_QUANTIZER_64
AOM_CTRL_USE_TYPE(AV1E_SET_ERROR_RESILIENT_MODE, int)
#define AOM_CTRL_AV1E_SET_ERROR_RESILIENT_MODE
AOM_CTRL_USE_TYPE(AOME_SET_MAX_INTRA_BITRATE_PCT, unsigned int)
#define AOM_CTRL_AOME_SET_MAX_INTRA_BITRATE_PCT
AOM_CTRL_USE_TYPE(AV1E_SET_S_FRAME_MODE, int)
#define AOM_CTRL_AV1E_SET_S_FRAME_MODE
AOM_CTRL_USE_TYPE(AOME_SET_MAX_INTER_BITRATE_PCT, unsigned int)
#define AOM_CTRL_AOME_SET_MAX_INTER_BITRATE_PCT
AOM_CTRL_USE_TYPE(AV1E_SET_AQ_MODE, unsigned int)
#define AOM_CTRL_AV1E_SET_AQ_MODE
AOM_CTRL_USE_TYPE(AOME_SET_NUMBER_SPATIAL_LAYERS, int)
#define AOME_CTRL_AOME_SET_NUMBER_SPATIAL_LAYERS
AOM_CTRL_USE_TYPE(AV1E_SET_FRAME_PERIODIC_BOOST, unsigned int)
#define AOM_CTRL_AV1E_SET_FRAME_PERIODIC_BOOST
AOM_CTRL_USE_TYPE(AV1E_SET_GF_CBR_BOOST_PCT, unsigned int)
#define AOM_CTRL_AV1E_SET_GF_CBR_BOOST_PCT
AOM_CTRL_USE_TYPE(AV1E_SET_NOISE_SENSITIVITY, unsigned int)
#define AOM_CTRL_AV1E_SET_NOISE_SENSITIVITY
AOM_CTRL_USE_TYPE(AV1E_SET_LOSSLESS, unsigned int)
#define AOM_CTRL_AV1E_SET_LOSSLESS
AOM_CTRL_USE_TYPE(AV1E_SET_TUNE_CONTENT, int) /* aom_tune_content */
#define AOM_CTRL_AV1E_SET_TUNE_CONTENT
AOM_CTRL_USE_TYPE(AV1E_SET_CDF_UPDATE_MODE, unsigned int)
#define AOM_CTRL_AV1E_SET_CDF_UPDATE_MODE
AOM_CTRL_USE_TYPE(AV1E_SET_COLOR_PRIMARIES, int)
#define AOM_CTRL_AV1E_SET_COLOR_PRIMARIES
AOM_CTRL_USE_TYPE(AV1E_SET_TRANSFER_CHARACTERISTICS, int)
#define AOM_CTRL_AV1E_SET_TRANSFER_CHARACTERISTICS
AOM_CTRL_USE_TYPE(AV1E_SET_MATRIX_COEFFICIENTS, int)
#define AOM_CTRL_AV1E_SET_MATRIX_COEFFICIENTS
AOM_CTRL_USE_TYPE(AV1E_SET_CHROMA_SAMPLE_POSITION, int)
#define AOM_CTRL_AV1E_SET_CHROMA_SAMPLE_POSITION
AOM_CTRL_USE_TYPE(AV1E_SET_MIN_GF_INTERVAL, unsigned int)
#define AOM_CTRL_AV1E_SET_MIN_GF_INTERVAL
AOM_CTRL_USE_TYPE(AV1E_SET_MAX_GF_INTERVAL, unsigned int)
#define AOM_CTRL_AV1E_SET_MAX_GF_INTERVAL
AOM_CTRL_USE_TYPE(AV1E_GET_ACTIVEMAP, aom_active_map_t *)
#define AOM_CTRL_AV1E_GET_ACTIVEMAP
AOM_CTRL_USE_TYPE(AV1E_SET_COLOR_RANGE, int)
#define AOM_CTRL_AV1E_SET_COLOR_RANGE
AOM_CTRL_USE_TYPE(AV1E_SET_RENDER_SIZE, int *)
#define AOM_CTRL_AV1E_SET_RENDER_SIZE
AOM_CTRL_USE_TYPE(AV1E_SET_TARGET_SEQ_LEVEL_IDX, int)
#define AOM_CTRL_AV1E_SET_TARGET_SEQ_LEVEL_IDX
AOM_CTRL_USE_TYPE(AV1E_GET_SEQ_LEVEL_IDX, int *)
#define AOM_CTRL_AV1E_GET_SEQ_LEVEL_IDX
AOM_CTRL_USE_TYPE(AV1E_SET_SUPERBLOCK_SIZE, unsigned int)
#define AOM_CTRL_AV1E_SET_SUPERBLOCK_SIZE
AOM_CTRL_USE_TYPE(AOME_SET_ENABLEAUTOBWDREF, unsigned int)
#define AOM_CTRL_AOME_SET_ENABLEAUTOBWDREF
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_CDEF, unsigned int)
#define AOM_CTRL_AV1E_SET_ENABLE_CDEF
@ -1489,6 +1776,7 @@ AOM_CTRL_USE_TYPE(AV1E_SET_DISABLE_TRELLIS_QUANT, unsigned int)
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_QM, unsigned int)
#define AOM_CTRL_AV1E_SET_ENABLE_QM
// TODO(aomedia:3231): Remove these two lines.
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_DIST_8X8, unsigned int)
#define AOM_CTRL_AV1E_SET_ENABLE_DIST_8X8
@ -1513,9 +1801,6 @@ AOM_CTRL_USE_TYPE(AV1E_SET_NUM_TG, unsigned int)
AOM_CTRL_USE_TYPE(AV1E_SET_MTU, unsigned int)
#define AOM_CTRL_AV1E_SET_MTU
AOM_CTRL_USE_TYPE(AV1E_SET_TIMING_INFO_TYPE, int) /* aom_timing_info_type_t */
#define AOM_CTRL_AV1E_SET_TIMING_INFO_TYPE
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_RECT_PARTITIONS, int)
#define AOM_CTRL_AV1E_SET_ENABLE_RECT_PARTITIONS
@ -1543,6 +1828,9 @@ AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_TX64, int)
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_FLIP_IDTX, int)
#define AOM_CTRL_AV1E_SET_ENABLE_FLIP_IDTX
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_RECT_TX, int)
#define AOM_CTRL_AV1E_SET_ENABLE_RECT_TX
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_DIST_WTD_COMP, int)
#define AOM_CTRL_AV1E_SET_ENABLE_DIST_WTD_COMP
@ -1615,77 +1903,20 @@ AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_INTRABC, int)
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_ANGLE_DELTA, int)
#define AOM_CTRL_AV1E_SET_ENABLE_ANGLE_DELTA
AOM_CTRL_USE_TYPE(AV1E_SET_FRAME_PARALLEL_DECODING, unsigned int)
#define AOM_CTRL_AV1E_SET_FRAME_PARALLEL_DECODING
AOM_CTRL_USE_TYPE(AV1E_SET_ERROR_RESILIENT_MODE, int)
#define AOM_CTRL_AV1E_SET_ERROR_RESILIENT_MODE
AOM_CTRL_USE_TYPE(AV1E_SET_S_FRAME_MODE, int)
#define AOM_CTRL_AV1E_SET_S_FRAME_MODE
AOM_CTRL_USE_TYPE(AV1E_SET_AQ_MODE, unsigned int)
#define AOM_CTRL_AV1E_SET_AQ_MODE
AOM_CTRL_USE_TYPE(AV1E_SET_DELTAQ_MODE, unsigned int)
#define AOM_CTRL_AV1E_SET_DELTAQ_MODE
AOM_CTRL_USE_TYPE(AV1E_SET_DELTALF_MODE, unsigned int)
#define AOM_CTRL_AV1E_SET_DELTALF_MODE
AOM_CTRL_USE_TYPE(AV1E_SET_FRAME_PERIODIC_BOOST, unsigned int)
#define AOM_CTRL_AV1E_SET_FRAME_PERIODIC_BOOST
AOM_CTRL_USE_TYPE(AV1E_SET_NOISE_SENSITIVITY, unsigned int)
#define AOM_CTRL_AV1E_SET_NOISE_SENSITIVITY
AOM_CTRL_USE_TYPE(AV1E_SET_TUNE_CONTENT, int) /* aom_tune_content */
#define AOM_CTRL_AV1E_SET_TUNE_CONTENT
AOM_CTRL_USE_TYPE(AV1E_SET_COLOR_PRIMARIES, int)
#define AOM_CTRL_AV1E_SET_COLOR_PRIMARIES
AOM_CTRL_USE_TYPE(AV1E_SET_TRANSFER_CHARACTERISTICS, int)
#define AOM_CTRL_AV1E_SET_TRANSFER_CHARACTERISTICS
AOM_CTRL_USE_TYPE(AV1E_SET_MATRIX_COEFFICIENTS, int)
#define AOM_CTRL_AV1E_SET_MATRIX_COEFFICIENTS
AOM_CTRL_USE_TYPE(AV1E_SET_CHROMA_SAMPLE_POSITION, int)
#define AOM_CTRL_AV1E_SET_CHROMA_SAMPLE_POSITION
AOM_CTRL_USE_TYPE(AV1E_SET_MIN_GF_INTERVAL, unsigned int)
#define AOM_CTRL_AV1E_SET_MIN_GF_INTERVAL
AOM_CTRL_USE_TYPE(AV1E_SET_MAX_GF_INTERVAL, unsigned int)
#define AOM_CTRL_AV1E_SET_MAX_GF_INTERVAL
AOM_CTRL_USE_TYPE(AV1E_GET_ACTIVEMAP, aom_active_map_t *)
#define AOM_CTRL_AV1E_GET_ACTIVEMAP
AOM_CTRL_USE_TYPE(AV1E_SET_COLOR_RANGE, int)
#define AOM_CTRL_AV1E_SET_COLOR_RANGE
#define AOM_CTRL_AV1E_SET_RENDER_SIZE
AOM_CTRL_USE_TYPE(AV1E_SET_RENDER_SIZE, int *)
AOM_CTRL_USE_TYPE(AV1E_SET_SUPERBLOCK_SIZE, unsigned int)
#define AOM_CTRL_AV1E_SET_SUPERBLOCK_SIZE
AOM_CTRL_USE_TYPE(AV1E_GET_SEQ_LEVEL_IDX, int *)
#define AOM_CTRL_AV1E_GET_SEQ_LEVEL_IDX
AOM_CTRL_USE_TYPE(AV1E_SET_SINGLE_TILE_DECODING, unsigned int)
#define AOM_CTRL_AV1E_SET_SINGLE_TILE_DECODING
AOM_CTRL_USE_TYPE(AV1E_ENABLE_MOTION_VECTOR_UNIT_TEST, unsigned int)
#define AOM_CTRL_AV1E_ENABLE_MOTION_VECTOR_UNIT_TEST
AOM_CTRL_USE_TYPE(AV1E_ENABLE_EXT_TILE_DEBUG, unsigned int)
#define AOM_CTRL_AV1E_ENABLE_EXT_TILE_DEBUG
AOM_CTRL_USE_TYPE(AV1E_SET_VMAF_MODEL_PATH, const char *)
#define AOM_CTRL_AV1E_SET_VMAF_MODEL_PATH
AOM_CTRL_USE_TYPE(AV1E_SET_TIMING_INFO_TYPE, int) /* aom_timing_info_type_t */
#define AOM_CTRL_AV1E_SET_TIMING_INFO_TYPE
AOM_CTRL_USE_TYPE(AV1E_SET_FILM_GRAIN_TEST_VECTOR, int)
#define AOM_CTRL_AV1E_SET_FILM_GRAIN_TEST_VECTOR
@ -1693,9 +1924,6 @@ AOM_CTRL_USE_TYPE(AV1E_SET_FILM_GRAIN_TEST_VECTOR, int)
AOM_CTRL_USE_TYPE(AV1E_SET_FILM_GRAIN_TABLE, const char *)
#define AOM_CTRL_AV1E_SET_FILM_GRAIN_TABLE
AOM_CTRL_USE_TYPE(AV1E_SET_CDF_UPDATE_MODE, unsigned int)
#define AOM_CTRL_AV1E_SET_CDF_UPDATE_MODE
AOM_CTRL_USE_TYPE(AV1E_SET_DENOISE_NOISE_LEVEL, int)
#define AOM_CTRL_AV1E_SET_DENOISE_NOISE_LEVEL
@ -1723,9 +1951,6 @@ AOM_CTRL_USE_TYPE(AV1E_SET_INTRA_DEFAULT_TX_ONLY, int)
AOM_CTRL_USE_TYPE(AV1E_SET_QUANT_B_ADAPT, int)
#define AOM_CTRL_AV1E_SET_QUANT_B_ADAPT
AOM_CTRL_USE_TYPE(AV1E_SET_GF_MIN_PYRAMID_HEIGHT, unsigned int)
#define AOM_CTRL_AV1E_SET_GF_MIN_PYRAMID_HEIGHT
AOM_CTRL_USE_TYPE(AV1E_SET_GF_MAX_PYRAMID_HEIGHT, unsigned int)
#define AOM_CTRL_AV1E_SET_GF_MAX_PYRAMID_HEIGHT
@ -1744,9 +1969,6 @@ AOM_CTRL_USE_TYPE(AV1E_SET_MODE_COST_UPD_FREQ, unsigned int)
AOM_CTRL_USE_TYPE(AV1E_SET_MV_COST_UPD_FREQ, unsigned int)
#define AOM_CTRL_AV1E_SET_MV_COST_UPD_FREQ
AOM_CTRL_USE_TYPE(AV1E_SET_TARGET_SEQ_LEVEL_IDX, int)
#define AOM_CTRL_AV1E_SET_TARGET_SEQ_LEVEL_IDX
AOM_CTRL_USE_TYPE(AV1E_SET_TIER_MASK, unsigned int)
#define AOM_CTRL_AV1E_SET_TIER_MASK
@ -1754,17 +1976,89 @@ AOM_CTRL_USE_TYPE(AV1E_SET_MIN_CR, unsigned int)
#define AOM_CTRL_AV1E_SET_MIN_CR
AOM_CTRL_USE_TYPE(AV1E_SET_SVC_LAYER_ID, aom_svc_layer_id_t *)
#define AOM_CTRL_AV1E_SET_SVC_LAYER_ID
// TODO(aomedia:3231): Deprecated. Remove it.
#define AOME_CTRL_AV1E_SET_SVC_LAYER_ID
AOM_CTRL_USE_TYPE(AV1E_SET_SVC_PARAMS, aom_svc_params_t *)
#define AOM_CTRL_AV1E_SET_SVC_PARAMS
// TODO(aomedia:3231): Deprecated. Remove it.
#define AOME_CTRL_AV1E_SET_SVC_PARAMS
AOM_CTRL_USE_TYPE(AV1E_SET_SVC_REF_FRAME_CONFIG, aom_svc_ref_frame_config_t *)
#define AOM_CTRL_AV1E_SET_SVC_REF_FRAME_CONFIG
// TODO(aomedia:3231): Deprecated. Remove it.
#define AOME_CTRL_AV1E_SET_SVC_REF_FRAME_CONFIG
AOM_CTRL_USE_TYPE(AV1E_SET_VMAF_MODEL_PATH, const char *)
#define AOM_CTRL_AV1E_SET_VMAF_MODEL_PATH
AOM_CTRL_USE_TYPE(AV1E_ENABLE_EXT_TILE_DEBUG, unsigned int)
#define AOM_CTRL_AV1E_ENABLE_EXT_TILE_DEBUG
AOM_CTRL_USE_TYPE(AV1E_ENABLE_SB_MULTIPASS_UNIT_TEST, unsigned int)
#define AOM_CTRL_AV1E_ENABLE_SB_MULTIPASS_UNIT_TEST
AOM_CTRL_USE_TYPE(AV1E_SET_GF_MIN_PYRAMID_HEIGHT, unsigned int)
#define AOM_CTRL_AV1E_SET_GF_MIN_PYRAMID_HEIGHT
AOM_CTRL_USE_TYPE(AV1E_SET_VBR_CORPUS_COMPLEXITY_LAP, unsigned int)
#define AOM_CTRL_AV1E_SET_VBR_CORPUS_COMPLEXITY_LAP
AOM_CTRL_USE_TYPE(AV1E_GET_BASELINE_GF_INTERVAL, int *)
#define AOM_CTRL_AV1E_GET_BASELINE_GF_INTERVAL
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_DNL_DENOISING, int)
#define AOM_CTRL_AV1E_SET_ENABLE_DNL_DENOISING
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_DIAGONAL_INTRA, int)
#define AOM_CTRL_AV1E_SET_ENABLE_DIAGONAL_INTRA
AOM_CTRL_USE_TYPE(AV1E_SET_DV_COST_UPD_FREQ, unsigned int)
#define AOM_CTRL_AV1E_SET_DV_COST_UPD_FREQ
AOM_CTRL_USE_TYPE(AV1E_SET_PARTITION_INFO_PATH, const char *)
#define AOM_CTRL_AV1E_SET_PARTITION_INFO_PATH
AOM_CTRL_USE_TYPE(AV1E_SET_EXTERNAL_PARTITION, aom_ext_part_funcs_t *)
#define AOM_CTRL_AV1E_SET_EXTERNAL_PARTITION
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_DIRECTIONAL_INTRA, int)
#define AOM_CTRL_AV1E_SET_ENABLE_DIRECTIONAL_INTRA
AOM_CTRL_USE_TYPE(AV1E_SET_ENABLE_TX_SIZE_SEARCH, int)
#define AOM_CTRL_AV1E_SET_ENABLE_TX_SIZE_SEARCH
AOM_CTRL_USE_TYPE(AV1E_SET_SVC_REF_FRAME_COMP_PRED,
aom_svc_ref_frame_comp_pred_t *)
#define AOM_CTRL_AV1E_SET_SVC_REF_FRAME_COMP_PRED
// TODO(aomedia:3231): Deprecated. Remove it.
#define AOME_CTRL_AV1E_SET_SVC_REF_FRAME_COMP_PRED
AOM_CTRL_USE_TYPE(AV1E_SET_DELTAQ_STRENGTH, unsigned int)
#define AOM_CTRL_AV1E_SET_DELTAQ_STRENGTH
AOM_CTRL_USE_TYPE(AV1E_SET_LOOPFILTER_CONTROL, int)
#define AOM_CTRL_AV1E_SET_LOOPFILTER_CONTROL
AOM_CTRL_USE_TYPE(AOME_GET_LOOPFILTER_LEVEL, int *)
#define AOM_CTRL_AOME_GET_LOOPFILTER_LEVEL
AOM_CTRL_USE_TYPE(AV1E_SET_AUTO_INTRA_TOOLS_OFF, unsigned int)
#define AOM_CTRL_AV1E_SET_AUTO_INTRA_TOOLS_OFF
AOM_CTRL_USE_TYPE(AV1E_SET_RTC_EXTERNAL_RC, int)
#define AOM_CTRL_AV1E_SET_RTC_EXTERNAL_RC
AOM_CTRL_USE_TYPE(AV1E_SET_FP_MT, unsigned int)
#define AOM_CTRL_AV1E_SET_FP_MT
AOM_CTRL_USE_TYPE(AV1E_SET_FP_MT_UNIT_TEST, unsigned int)
#define AOM_CTRL_AV1E_SET_FP_MT_UNIT_TEST
AOM_CTRL_USE_TYPE(AV1E_GET_TARGET_SEQ_LEVEL_IDX, int *)
#define AOM_CTRL_AV1E_GET_TARGET_SEQ_LEVEL_IDX
/*!\endcond */
/*! @} - end defgroup aom_encoder */
#ifdef __cplusplus

View file

@ -33,9 +33,17 @@ extern "C" {
* This interface provides the capability to decode AV1 streams.
* @{
*/
/*!\brief A single instance of the AV1 decoder.
*\deprecated This access mechanism is provided for backwards compatibility;
* prefer aom_codec_av1_dx().
*/
extern aom_codec_iface_t aom_codec_av1_dx_algo;
/*!\brief The interface to the AV1 decoder.
*/
extern aom_codec_iface_t *aom_codec_av1_dx(void);
/*!@} - end algorithm interface member group*/
/*!@} - end algorithm interface member group */
/** Data structure that stores bit accounting for debug
*/
@ -89,6 +97,81 @@ typedef struct aom_tile_data {
size_t extra_size;
} aom_tile_data;
/*!\brief Max number of tile columns
*
* This is the limit of number of tile columns allowed within a frame.
*
* Currently same as "MAX_TILE_COLS" in AV1, the maximum that AV1 supports.
*
*/
#define AOM_MAX_TILE_COLS 64
/*!\brief Max number of tile rows
*
* This is the limit of number of tile rows allowed within a frame.
*
* Currently same as "MAX_TILE_ROWS" in AV1, the maximum that AV1 supports.
*
*/
#define AOM_MAX_TILE_ROWS 64
/*!\brief Structure to hold information about tiles in a frame.
*
* Defines a structure to hold a frame's tile information, namely
* number of tile columns, number of tile_rows, and the width and
* height of each tile.
*/
typedef struct aom_tile_info {
/*! Indicates the number of tile columns. */
int tile_columns;
/*! Indicates the number of tile rows. */
int tile_rows;
/*! Indicates the tile widths in units of SB. */
int tile_widths[AOM_MAX_TILE_COLS];
/*! Indicates the tile heights in units of SB. */
int tile_heights[AOM_MAX_TILE_ROWS];
/*! Indicates the number of tile groups present in a frame. */
int num_tile_groups;
} aom_tile_info;
/*!\brief Structure to hold information about still image coding.
*
* Defines a structure to hold a information regarding still picture
* and its header type.
*/
typedef struct aom_still_picture_info {
/*! Video is a single frame still picture */
int is_still_picture;
/*! Use full header for still picture */
int is_reduced_still_picture_hdr;
} aom_still_picture_info;
/*!\brief Structure to hold information about S_FRAME.
*
* Defines a structure to hold a information regarding S_FRAME
* and its position.
*/
typedef struct aom_s_frame_info {
/*! Indicates if current frame is S_FRAME */
int is_s_frame;
/*! Indicates if current S_FRAME is present at ALTREF frame*/
int is_s_frame_at_altref;
} aom_s_frame_info;
/*!\brief Structure to hold information about screen content tools.
*
* Defines a structure to hold information about screen content
* tools, namely: allow_screen_content_tools, allow_intrabc, and
* force_integer_mv.
*/
typedef struct aom_screen_content_tools_info {
/*! Are screen content tools allowed */
int allow_screen_content_tools;
/*! Is intrabc allowed */
int allow_intrabc;
/*! Is integer mv forced */
int force_integer_mv;
} aom_screen_content_tools_info;
/*!\brief Structure to hold the external reference frame pointer.
*
* Define a structure to hold the external reference frame pointer.
@ -105,6 +188,7 @@ typedef struct av1_ext_ref_frame {
*
* This set of macros define the control functions available for the AOM
* decoder interface.
* The range for decoder control ID is >= 256.
*
* \sa #aom_codec_control(aom_codec_ctx_t *ctx, int ctrl_id, ...)
*/
@ -125,14 +209,16 @@ enum aom_dec_control_id {
AOMD_GET_LAST_REF_USED,
/*!\brief Codec control function to get the dimensions that the current
* frame is decoded at, int* parameter. This may be different to the
* intended display size for the frame as specified in the wrapper or frame
* header (see AV1D_GET_DISPLAY_SIZE).
* frame is decoded at, int* parameter
*
* This may be different to the intended display size for the frame as
* specified in the wrapper or frame header (see AV1D_GET_DISPLAY_SIZE).
*/
AV1D_GET_FRAME_SIZE,
/*!\brief Codec control function to get the current frame's intended display
* dimensions (as specified in the wrapper or frame header), int* parameter.
* dimensions (as specified in the wrapper or frame header), int* parameter
*
* This may be different to the decoded dimensions of this frame (see
* AV1D_GET_FRAME_SIZE).
*/
@ -148,12 +234,13 @@ enum aom_dec_control_id {
*/
AV1D_GET_IMG_FORMAT,
/*!\brief Codec control function to get the size of the tile, unsigned int
parameter */
/*!\brief Codec control function to get the size of the tile, unsigned int*
* parameter
*/
AV1D_GET_TILE_SIZE,
/*!\brief Codec control function to get the tile count in a tile list, int*
* parameter
/*!\brief Codec control function to get the tile count in a tile list,
* unsigned int* parameter
*/
AV1D_GET_TILE_COUNT,
@ -194,8 +281,8 @@ enum aom_dec_control_id {
* The caller should ensure that AOM_CODEC_OK is returned before attempting
* to dereference the Accounting pointer.
*
* \attention When compiled without --enable-accounting, this returns
* AOM_CODEC_INCAPABLE.
* \attention When configured with -DCONFIG_ACCOUNTING=0, the default, this
* returns AOM_CODEC_INCAPABLE.
*/
AV1_GET_ACCOUNTING,
@ -217,7 +304,8 @@ enum aom_dec_control_id {
AV1_SET_DECODE_TILE_ROW,
AV1_SET_DECODE_TILE_COL,
/*!\brief Codec control function to set the tile coding mode, int parameter
/*!\brief Codec control function to set the tile coding mode, unsigned int
* parameter
*
* - 0 = tiles are coded in normal tile mode
* - 1 = tiles are coded in large-scale tile mode
@ -225,7 +313,7 @@ enum aom_dec_control_id {
AV1_SET_TILE_MODE,
/*!\brief Codec control function to get the frame header information of an
* encoded frame, unsigned int* parameter
* encoded frame, aom_tile_data* parameter
*/
AV1D_GET_FRAME_HEADER_INFO,
@ -271,7 +359,7 @@ enum aom_dec_control_id {
AV1D_SET_OPERATING_POINT,
/*!\brief Codec control function to indicate whether to output one frame per
* temporal unit (the default), or one frame per spatial layer. int parameter
* temporal unit (the default), or one frame per spatial layer, int parameter
*
* In a scalable stream, each temporal unit corresponds to a single "frame"
* of video, and within a temporal unit there may be multiple spatial layers
@ -285,7 +373,7 @@ enum aom_dec_control_id {
/*!\brief Codec control function to set an aom_inspect_cb callback that is
* invoked each time a frame is decoded, aom_inspect_init* parameter
*
* \attention When compiled without --enable-inspection, this
* \attention When configured with -DCONFIG_INSPECTION=0, the default, this
* returns AOM_CODEC_INCAPABLE.
*/
AV1_SET_INSPECTION_CALLBACK,
@ -298,7 +386,83 @@ enum aom_dec_control_id {
*/
AV1D_SET_SKIP_FILM_GRAIN,
AOM_DECODER_CTRL_ID_MAX,
/*!\brief Codec control function to check the presence of forward key frames,
* int* parameter
*/
AOMD_GET_FWD_KF_PRESENT,
/*!\brief Codec control function to get the frame flags of the previous frame
* decoded, int* parameter
*
* This will return a flag of type aom_codec_frame_flags_t.
*/
AOMD_GET_FRAME_FLAGS,
/*!\brief Codec control function to check the presence of altref frames, int*
* parameter
*/
AOMD_GET_ALTREF_PRESENT,
/*!\brief Codec control function to get tile information of the previous frame
* decoded, aom_tile_info* parameter
*
* This will return a struct of type aom_tile_info.
*/
AOMD_GET_TILE_INFO,
/*!\brief Codec control function to get screen content tools information,
* aom_screen_content_tools_info* parameter
*
* It returns a struct of type aom_screen_content_tools_info, which contains
* the header flags allow_screen_content_tools, allow_intrabc, and
* force_integer_mv.
*/
AOMD_GET_SCREEN_CONTENT_TOOLS_INFO,
/*!\brief Codec control function to get the still picture coding information,
* aom_still_picture_info* parameter
*/
AOMD_GET_STILL_PICTURE,
/*!\brief Codec control function to get superblock size,
* aom_superblock_size_t* parameter
*
* It returns an enum, indicating the superblock size read from the sequence
* header(0 for BLOCK_64X64 and 1 for BLOCK_128X128)
*/
AOMD_GET_SB_SIZE,
/*!\brief Codec control function to check if the previous frame
* decoded has show existing frame flag set, int* parameter
*/
AOMD_GET_SHOW_EXISTING_FRAME_FLAG,
/*!\brief Codec control function to get the S_FRAME coding information,
* aom_s_frame_info* parameter
*/
AOMD_GET_S_FRAME_INFO,
/*!\brief Codec control function to get the show frame flag, int* parameter
*/
AOMD_GET_SHOW_FRAME_FLAG,
/*!\brief Codec control function to get the base q index of a frame, int*
* parameter
*/
AOMD_GET_BASE_Q_IDX,
/*!\brief Codec control function to get the order hint of a frame, unsigned
* int* parameter
*/
AOMD_GET_ORDER_HINT,
/*!\brief Codec control function to get the info of a 4x4 block.
* Parameters: int mi_row, int mi_col, and MB_MODE_INFO*.
*
* \note This only returns a shallow copy, so all pointer members should not
* be used.
*/
AV1D_GET_MI_INFO,
};
/*!\cond */
@ -322,8 +486,8 @@ AOM_CTRL_USE_TYPE(AOMD_GET_FRAME_CORRUPTED, int *)
AOM_CTRL_USE_TYPE(AOMD_GET_LAST_REF_USED, int *)
#define AOM_CTRL_AOMD_GET_LAST_REF_USED
AOM_CTRL_USE_TYPE(AOMD_GET_LAST_QUANTIZER, int *)
#define AOM_CTRL_AOMD_GET_LAST_QUANTIZER
AOM_CTRL_USE_TYPE(AV1D_GET_FRAME_SIZE, int *)
#define AOM_CTRL_AV1D_GET_FRAME_SIZE
AOM_CTRL_USE_TYPE(AV1D_GET_DISPLAY_SIZE, int *)
#define AOM_CTRL_AV1D_GET_DISPLAY_SIZE
@ -340,15 +504,18 @@ AOM_CTRL_USE_TYPE(AV1D_GET_TILE_SIZE, unsigned int *)
AOM_CTRL_USE_TYPE(AV1D_GET_TILE_COUNT, unsigned int *)
#define AOM_CTRL_AV1D_GET_TILE_COUNT
AOM_CTRL_USE_TYPE(AV1D_GET_FRAME_SIZE, int *)
#define AOM_CTRL_AV1D_GET_FRAME_SIZE
AOM_CTRL_USE_TYPE(AV1_INVERT_TILE_DECODE_ORDER, int)
#define AOM_CTRL_AV1_INVERT_TILE_DECODE_ORDER
AOM_CTRL_USE_TYPE(AV1_SET_SKIP_LOOP_FILTER, int)
#define AOM_CTRL_AV1_SET_SKIP_LOOP_FILTER
AOM_CTRL_USE_TYPE(AV1_GET_ACCOUNTING, Accounting **)
#define AOM_CTRL_AV1_GET_ACCOUNTING
AOM_CTRL_USE_TYPE(AOMD_GET_LAST_QUANTIZER, int *)
#define AOM_CTRL_AOMD_GET_LAST_QUANTIZER
AOM_CTRL_USE_TYPE(AV1_SET_DECODE_TILE_ROW, int)
#define AOM_CTRL_AV1_SET_DECODE_TILE_ROW
@ -373,9 +540,6 @@ AOM_CTRL_USE_TYPE(AV1D_EXT_TILE_DEBUG, unsigned int)
AOM_CTRL_USE_TYPE(AV1D_SET_ROW_MT, unsigned int)
#define AOM_CTRL_AV1D_SET_ROW_MT
AOM_CTRL_USE_TYPE(AV1D_SET_SKIP_FILM_GRAIN, int)
#define AOM_CTRL_AV1D_SET_SKIP_FILM_GRAIN
AOM_CTRL_USE_TYPE(AV1D_SET_IS_ANNEXB, unsigned int)
#define AOM_CTRL_AV1D_SET_IS_ANNEXB
@ -387,9 +551,52 @@ AOM_CTRL_USE_TYPE(AV1D_SET_OUTPUT_ALL_LAYERS, int)
AOM_CTRL_USE_TYPE(AV1_SET_INSPECTION_CALLBACK, aom_inspect_init *)
#define AOM_CTRL_AV1_SET_INSPECTION_CALLBACK
AOM_CTRL_USE_TYPE(AV1D_SET_SKIP_FILM_GRAIN, int)
#define AOM_CTRL_AV1D_SET_SKIP_FILM_GRAIN
AOM_CTRL_USE_TYPE(AOMD_GET_FWD_KF_PRESENT, int *)
#define AOM_CTRL_AOMD_GET_FWD_KF_PRESENT
AOM_CTRL_USE_TYPE(AOMD_GET_FRAME_FLAGS, int *)
#define AOM_CTRL_AOMD_GET_FRAME_FLAGS
AOM_CTRL_USE_TYPE(AOMD_GET_ALTREF_PRESENT, int *)
#define AOM_CTRL_AOMD_GET_ALTREF_PRESENT
AOM_CTRL_USE_TYPE(AOMD_GET_TILE_INFO, aom_tile_info *)
#define AOM_CTRL_AOMD_GET_TILE_INFO
AOM_CTRL_USE_TYPE(AOMD_GET_SCREEN_CONTENT_TOOLS_INFO,
aom_screen_content_tools_info *)
#define AOM_CTRL_AOMD_GET_SCREEN_CONTENT_TOOLS_INFO
AOM_CTRL_USE_TYPE(AOMD_GET_STILL_PICTURE, aom_still_picture_info *)
#define AOM_CTRL_AOMD_GET_STILL_PICTURE
AOM_CTRL_USE_TYPE(AOMD_GET_SB_SIZE, aom_superblock_size_t *)
#define AOMD_CTRL_AOMD_GET_SB_SIZE
AOM_CTRL_USE_TYPE(AOMD_GET_SHOW_EXISTING_FRAME_FLAG, int *)
#define AOMD_CTRL_AOMD_GET_SHOW_EXISTING_FRAME_FLAG
AOM_CTRL_USE_TYPE(AOMD_GET_S_FRAME_INFO, aom_s_frame_info *)
#define AOMD_CTRL_AOMD_GET_S_FRAME_INFO
AOM_CTRL_USE_TYPE(AOMD_GET_SHOW_FRAME_FLAG, int *)
#define AOM_CTRL_AOMD_GET_SHOW_FRAME_FLAG
AOM_CTRL_USE_TYPE(AOMD_GET_BASE_Q_IDX, int *)
#define AOM_CTRL_AOMD_GET_BASE_Q_IDX
AOM_CTRL_USE_TYPE(AOMD_GET_ORDER_HINT, unsigned int *)
#define AOM_CTRL_AOMD_GET_ORDER_HINT
// The AOM_CTRL_USE_TYPE macro can't be used with AV1D_GET_MI_INFO because
// AV1D_GET_MI_INFO takes more than one parameter.
#define AOM_CTRL_AV1D_GET_MI_INFO
/*!\endcond */
/*! @} - end defgroup aom_decoder */
#ifdef __cplusplus
} // extern "C"
#endif

View file

@ -6,6 +6,7 @@ text aom_codec_error
text aom_codec_error_detail
text aom_codec_get_caps
text aom_codec_iface_name
text aom_codec_set_option
text aom_codec_version
text aom_codec_version_extra_str
text aom_codec_version_str

View file

@ -28,13 +28,15 @@
* </pre>
*
* An application instantiates a specific decoder instance by using
* aom_codec_init() and a pointer to the algorithm's interface structure:
* aom_codec_dec_init() and a pointer to the algorithm's interface structure:
* <pre>
* my_app.c:
* extern aom_codec_iface_t my_codec;
* {
* aom_codec_ctx_t algo;
* res = aom_codec_init(&algo, &my_codec);
* int threads = 4;
* aom_codec_dec_cfg_t cfg = { threads, 0, 0, 1 };
* res = aom_codec_dec_init(&algo, &my_codec, &cfg, 0);
* }
* </pre>
*
@ -45,6 +47,7 @@
#define AOM_AOM_INTERNAL_AOM_CODEC_INTERNAL_H_
#include "../aom_decoder.h"
#include "../aom_encoder.h"
#include "common/args_helper.h"
#include <stdarg.h>
#ifdef __cplusplus
@ -66,7 +69,7 @@ typedef struct aom_codec_alg_priv aom_codec_alg_priv_t;
/*!\brief init function pointer prototype
*
* Performs algorithm-specific initialization of the decoder context. This
* function is called by the generic aom_codec_init() wrapper function, so
* function is called by aom_codec_dec_init() and aom_codec_enc_init(), so
* plugins implementing this interface may trust the input parameters to be
* properly initialized.
*
@ -151,22 +154,45 @@ typedef aom_codec_err_t (*aom_codec_get_si_fn_t)(aom_codec_alg_priv_t *ctx,
typedef aom_codec_err_t (*aom_codec_control_fn_t)(aom_codec_alg_priv_t *ctx,
va_list ap);
/*!\brief codec option setter function pointer prototype
* This function is used to set a codec option using a key (option name) & value
* pair.
*
* \param[in] ctx Pointer to this instance's context
* \param[in] name A string of the option's name (key)
* \param[in] value A string of the value to be set to
*
* \retval #AOM_CODEC_OK
* The option is successfully set to the value
* \retval #AOM_CODEC_INVALID_PARAM
* The data was not valid.
*/
typedef aom_codec_err_t (*aom_codec_set_option_fn_t)(aom_codec_alg_priv_t *ctx,
const char *name,
const char *value);
/*!\brief control function pointer mapping
*
* This structure stores the mapping between control identifiers and
* implementing functions. Each algorithm provides a list of these
* mappings. This list is searched by the aom_codec_control() wrapper
* mappings. This list is searched by the aom_codec_control()
* function to determine which function to invoke. The special
* value {0, NULL} is used to indicate end-of-list, and must be
* present. The special value {0, <non-null>} can be used as a catch-all
* mapping. This implies that ctrl_id values chosen by the algorithm
* \ref MUST be non-zero.
* value defined by CTRL_MAP_END is used to indicate end-of-list, and must be
* present. It can be tested with the at_ctrl_map_end function. Note that
* ctrl_id values \ref MUST be non-zero.
*/
typedef const struct aom_codec_ctrl_fn_map {
int ctrl_id;
aom_codec_control_fn_t fn;
} aom_codec_ctrl_fn_map_t;
#define CTRL_MAP_END \
{ 0, NULL }
static AOM_INLINE int at_ctrl_map_end(aom_codec_ctrl_fn_map_t *e) {
return e->ctrl_id == 0 && e->fn == NULL;
}
/*!\brief decode data function pointer prototype
*
* Processes a buffer of coded data. This function is called by the generic
@ -252,7 +278,7 @@ typedef aom_fixed_buf_t *(*aom_codec_get_global_headers_fn_t)(
typedef aom_image_t *(*aom_codec_get_preview_frame_fn_t)(
aom_codec_alg_priv_t *ctx);
/*!\brief Decoder algorithm interface interface
/*!\brief Decoder algorithm interface
*
* All decoders \ref MUST expose a variable of this type.
*/
@ -284,6 +310,7 @@ struct aom_codec_iface {
aom_codec_get_preview_frame_fn_t
get_preview; /**< \copydoc ::aom_codec_get_preview_frame_fn_t */
} enc;
aom_codec_set_option_fn_t set_option;
};
/*!\brief Instance private storage
@ -307,19 +334,6 @@ struct aom_codec_priv {
#define CAST(id, arg) va_arg((arg), aom_codec_control_type_##id)
/* CODEC_INTERFACE convenience macro
*
* By convention, each codec interface is a struct with extern linkage, where
* the symbol is suffixed with _algo. A getter function is also defined to
* return a pointer to the struct, since in some cases it's easier to work
* with text symbols than data symbols (see issue #169). This function has
* the same name as the struct, less the _algo suffix. The CODEC_INTERFACE
* macro is provided to define this getter function automatically.
*/
#define CODEC_INTERFACE(id) \
aom_codec_iface_t *id(void) { return &id##_algo; } \
aom_codec_iface_t id##_algo
/* Internal Utility Functions
*
* The following functions are intended to be used inside algorithms as
@ -356,7 +370,7 @@ const aom_codec_cx_pkt_t *aom_codec_pkt_list_get(
struct aom_internal_error_info {
aom_codec_err_t error_code;
int has_detail;
char detail[80];
char detail[ARG_ERR_MSG_MAX_LEN];
int setjmp; // Boolean: whether 'jmp' is valid.
jmp_buf jmp;
};
@ -369,9 +383,21 @@ struct aom_internal_error_info {
#endif
#endif
// Tells the compiler to perform `printf` format string checking if the
// compiler supports it; see the 'format' attribute in
// <https://gcc.gnu.org/onlinedocs/gcc/Common-Function-Attributes.html>.
#define LIBAOM_FORMAT_PRINTF(string_index, first_to_check)
#if defined(__has_attribute)
#if __has_attribute(format)
#undef LIBAOM_FORMAT_PRINTF
#define LIBAOM_FORMAT_PRINTF(string_index, first_to_check) \
__attribute__((__format__(__printf__, string_index, first_to_check)))
#endif
#endif
void aom_internal_error(struct aom_internal_error_info *info,
aom_codec_err_t error, const char *fmt,
...) CLANG_ANALYZER_NORETURN;
aom_codec_err_t error, const char *fmt, ...)
LIBAOM_FORMAT_PRINTF(3, 4) CLANG_ANALYZER_NORETURN;
void aom_merge_corrupted_flag(int *corrupted, int value);
#ifdef __cplusplus

View file

@ -32,8 +32,8 @@ struct aom_metadata_array {
/*!\brief Alloc memory for aom_metadata_array struct.
*
* Allocate memory for aom_metadata_array struct.
* If sz is 0 the aom_metadata_array structs internal buffer list will be NULL,
* but the aom_metadata_array struct itself will still be allocated.
* If sz is 0 the aom_metadata_array struct's internal buffer list will be
* NULL, but the aom_metadata_array struct itself will still be allocated.
* Returns a pointer to the allocated struct or NULL on failure.
*
* \param[in] sz Size of internal metadata list buffer

View file

@ -22,8 +22,6 @@
#include "aom/aom_integer.h"
#include "aom/internal/aom_codec_internal.h"
#define SAVE_STATUS(ctx, var) (ctx ? (ctx->err = var) : var)
int aom_codec_version(void) { return VERSION_PACKED; }
const char *aom_codec_version_str(void) { return VERSION_STRING_NOSP; }
@ -67,22 +65,19 @@ const char *aom_codec_error_detail(aom_codec_ctx_t *ctx) {
}
aom_codec_err_t aom_codec_destroy(aom_codec_ctx_t *ctx) {
aom_codec_err_t res;
if (!ctx)
res = AOM_CODEC_INVALID_PARAM;
else if (!ctx->iface || !ctx->priv)
res = AOM_CODEC_ERROR;
else {
ctx->iface->destroy((aom_codec_alg_priv_t *)ctx->priv);
ctx->iface = NULL;
ctx->name = NULL;
ctx->priv = NULL;
res = AOM_CODEC_OK;
if (!ctx) {
return AOM_CODEC_INVALID_PARAM;
}
return SAVE_STATUS(ctx, res);
if (!ctx->iface || !ctx->priv) {
ctx->err = AOM_CODEC_ERROR;
return AOM_CODEC_ERROR;
}
ctx->iface->destroy((aom_codec_alg_priv_t *)ctx->priv);
ctx->iface = NULL;
ctx->name = NULL;
ctx->priv = NULL;
ctx->err = AOM_CODEC_OK;
return AOM_CODEC_OK;
}
aom_codec_caps_t aom_codec_get_caps(aom_codec_iface_t *iface) {
@ -90,30 +85,48 @@ aom_codec_caps_t aom_codec_get_caps(aom_codec_iface_t *iface) {
}
aom_codec_err_t aom_codec_control(aom_codec_ctx_t *ctx, int ctrl_id, ...) {
aom_codec_err_t res;
if (!ctx || !ctrl_id)
res = AOM_CODEC_INVALID_PARAM;
else if (!ctx->iface || !ctx->priv || !ctx->iface->ctrl_maps)
res = AOM_CODEC_ERROR;
else {
aom_codec_ctrl_fn_map_t *entry;
res = AOM_CODEC_ERROR;
for (entry = ctx->iface->ctrl_maps; entry && entry->fn; entry++) {
if (!entry->ctrl_id || entry->ctrl_id == ctrl_id) {
va_list ap;
va_start(ap, ctrl_id);
res = entry->fn((aom_codec_alg_priv_t *)ctx->priv, ap);
va_end(ap);
break;
}
}
if (!ctx) {
return AOM_CODEC_INVALID_PARAM;
}
// Control ID must be non-zero.
if (!ctrl_id) {
ctx->err = AOM_CODEC_INVALID_PARAM;
return AOM_CODEC_INVALID_PARAM;
}
if (!ctx->iface || !ctx->priv || !ctx->iface->ctrl_maps) {
ctx->err = AOM_CODEC_ERROR;
return AOM_CODEC_ERROR;
}
return SAVE_STATUS(ctx, res);
// "ctrl_maps" is an array of (control ID, function pointer) elements,
// with CTRL_MAP_END as a sentinel.
for (aom_codec_ctrl_fn_map_t *entry = ctx->iface->ctrl_maps;
!at_ctrl_map_end(entry); ++entry) {
if (entry->ctrl_id == ctrl_id) {
va_list ap;
va_start(ap, ctrl_id);
ctx->err = entry->fn((aom_codec_alg_priv_t *)ctx->priv, ap);
va_end(ap);
return ctx->err;
}
}
ctx->err = AOM_CODEC_ERROR;
ctx->priv->err_detail = "Invalid control ID";
return AOM_CODEC_ERROR;
}
aom_codec_err_t aom_codec_set_option(aom_codec_ctx_t *ctx, const char *name,
const char *value) {
if (!ctx) {
return AOM_CODEC_INVALID_PARAM;
}
if (!ctx->iface || !ctx->priv || !ctx->iface->set_option) {
ctx->err = AOM_CODEC_ERROR;
return AOM_CODEC_ERROR;
}
ctx->err =
ctx->iface->set_option((aom_codec_alg_priv_t *)ctx->priv, name, value);
return ctx->err;
}
void aom_internal_error(struct aom_internal_error_info *info,

View file

@ -39,8 +39,25 @@ aom_codec_err_t aom_codec_enc_init_ver(aom_codec_ctx_t *ctx,
const aom_codec_enc_cfg_t *cfg,
aom_codec_flags_t flags, int ver) {
aom_codec_err_t res;
// The value of AOM_ENCODER_ABI_VERSION in libaom v3.0.0 and v3.1.0 - v3.1.3.
//
// We are compatible with these older libaom releases. AOM_ENCODER_ABI_VERSION
// was incremented after these releases for two reasons:
// 1. AOM_ENCODER_ABI_VERSION takes contribution from
// AOM_EXT_PART_ABI_VERSION. The external partition API is still
// experimental, so it should not be considered as part of the stable ABI.
// fd9ed8366 External partition: Define APIs
// https://aomedia-review.googlesource.com/c/aom/+/135663
// 2. As a way to detect the presence of speeds 7-9 in all-intra mode. I (wtc)
// suggested this change because I misunderstood how
// AOM_ENCODER_ABI_VERSION was used.
// bbdfa68d1 AllIntra: Redefine all-intra mode speed features for speed 7+
// https://aomedia-review.googlesource.com/c/aom/+/140624
const int aom_encoder_abi_version_25 = 25;
if (ver != AOM_ENCODER_ABI_VERSION)
// TODO(bug aomedia:3228): Remove the check for aom_encoder_abi_version_25 in
// libaom v4.0.0.
if (ver != AOM_ENCODER_ABI_VERSION && ver != aom_encoder_abi_version_25)
res = AOM_CODEC_ABI_MISMATCH;
else if (!ctx || !iface || !cfg)
res = AOM_CODEC_INVALID_PARAM;
@ -50,7 +67,11 @@ aom_codec_err_t aom_codec_enc_init_ver(aom_codec_ctx_t *ctx,
res = AOM_CODEC_INCAPABLE;
else if ((flags & AOM_CODEC_USE_PSNR) && !(iface->caps & AOM_CODEC_CAP_PSNR))
res = AOM_CODEC_INCAPABLE;
else {
else if (cfg->g_bit_depth > 8 && (flags & AOM_CODEC_USE_HIGHBITDEPTH) == 0) {
res = AOM_CODEC_INVALID_PARAM;
ctx->err_detail =
"High bit-depth used without the AOM_CODEC_USE_HIGHBITDEPTH flag.";
} else {
ctx->iface = iface;
ctx->name = iface->name;
ctx->priv = NULL;

View file

@ -9,6 +9,7 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include <limits.h>
#include <stdlib.h>
#include <string.h>
@ -38,6 +39,8 @@ static aom_image_t *img_alloc_helper(
unsigned int h, w, s, xcs, ycs, bps, bit_depth;
unsigned int stride_in_bytes;
if (img != NULL) memset(img, 0, sizeof(aom_image_t));
/* Treat align==0 like align==1 */
if (!buf_align) buf_align = 1;
@ -60,6 +63,7 @@ static aom_image_t *img_alloc_helper(
switch (fmt) {
case AOM_IMG_FMT_I420:
case AOM_IMG_FMT_YV12:
case AOM_IMG_FMT_NV12:
case AOM_IMG_FMT_AOMI420:
case AOM_IMG_FMT_AOMYV12: bps = 12; break;
case AOM_IMG_FMT_I422: bps = 16; break;
@ -77,6 +81,7 @@ static aom_image_t *img_alloc_helper(
switch (fmt) {
case AOM_IMG_FMT_I420:
case AOM_IMG_FMT_YV12:
case AOM_IMG_FMT_NV12:
case AOM_IMG_FMT_AOMI420:
case AOM_IMG_FMT_AOMYV12:
case AOM_IMG_FMT_I422:
@ -89,6 +94,7 @@ static aom_image_t *img_alloc_helper(
switch (fmt) {
case AOM_IMG_FMT_I420:
case AOM_IMG_FMT_YV12:
case AOM_IMG_FMT_NV12:
case AOM_IMG_FMT_AOMI420:
case AOM_IMG_FMT_AOMYV12:
case AOM_IMG_FMT_YV1216:
@ -111,8 +117,6 @@ static aom_image_t *img_alloc_helper(
if (!img) goto fail;
img->self_allocd = 1;
} else {
memset(img, 0, sizeof(aom_image_t));
}
img->img_data = img_data;
@ -154,6 +158,13 @@ static aom_image_t *img_alloc_helper(
img->stride[AOM_PLANE_Y] = stride_in_bytes;
img->stride[AOM_PLANE_U] = img->stride[AOM_PLANE_V] = stride_in_bytes >> xcs;
if (fmt == AOM_IMG_FMT_NV12) {
// Each row is a row of U and a row of V interleaved, so the stride is twice
// as long.
img->stride[AOM_PLANE_U] *= 2;
img->stride[AOM_PLANE_V] = 0;
}
/* Default viewport to entire image. (This aom_img_set_rect call always
* succeeds.) */
aom_img_set_rect(img, 0, 0, d_w, d_h, border);
@ -200,9 +211,8 @@ aom_image_t *aom_img_alloc_with_border(aom_image_t *img, aom_img_fmt_t fmt,
int aom_img_set_rect(aom_image_t *img, unsigned int x, unsigned int y,
unsigned int w, unsigned int h, unsigned int border) {
unsigned char *data;
if (x + w <= img->w && y + h <= img->h) {
if (x <= UINT_MAX - w && x + w <= img->w && y <= UINT_MAX - h &&
y + h <= img->h) {
img->d_w = w;
img->d_h = h;
@ -216,7 +226,7 @@ int aom_img_set_rect(aom_image_t *img, unsigned int x, unsigned int y,
} else {
const int bytes_per_sample =
(img->fmt & AOM_IMG_FMT_HIGHBITDEPTH) ? 2 : 1;
data = img->img_data;
unsigned char *data = img->img_data;
img->planes[AOM_PLANE_Y] =
data + x * bytes_per_sample + y * img->stride[AOM_PLANE_Y];
@ -225,7 +235,11 @@ int aom_img_set_rect(aom_image_t *img, unsigned int x, unsigned int y,
unsigned int uv_border_h = border >> img->y_chroma_shift;
unsigned int uv_x = x >> img->x_chroma_shift;
unsigned int uv_y = y >> img->y_chroma_shift;
if (!(img->fmt & AOM_IMG_FMT_UV_FLIP)) {
if (img->fmt == AOM_IMG_FMT_NV12) {
img->planes[AOM_PLANE_U] = data + uv_x * bytes_per_sample * 2 +
uv_y * img->stride[AOM_PLANE_U];
img->planes[AOM_PLANE_V] = NULL;
} else if (!(img->fmt & AOM_IMG_FMT_UV_FLIP)) {
img->planes[AOM_PLANE_U] =
data + uv_x * bytes_per_sample + uv_y * img->stride[AOM_PLANE_U];
data += ((img->h >> img->y_chroma_shift) + 2 * uv_border_h) *
@ -350,26 +364,18 @@ int aom_img_add_metadata(aom_image_t *img, uint32_t type, const uint8_t *data,
}
aom_metadata_t *metadata =
aom_img_metadata_alloc(type, data, sz, insert_flag);
if (!metadata) goto fail;
if (!img->metadata->metadata_array) {
img->metadata->metadata_array =
(aom_metadata_t **)calloc(1, sizeof(metadata));
if (!img->metadata->metadata_array || img->metadata->sz != 0) {
aom_img_metadata_free(metadata);
goto fail;
}
} else {
img->metadata->metadata_array =
(aom_metadata_t **)realloc(img->metadata->metadata_array,
(img->metadata->sz + 1) * sizeof(metadata));
if (!metadata) return -1;
aom_metadata_t **metadata_array =
(aom_metadata_t **)realloc(img->metadata->metadata_array,
(img->metadata->sz + 1) * sizeof(metadata));
if (!metadata_array) {
aom_img_metadata_free(metadata);
return -1;
}
img->metadata->metadata_array = metadata_array;
img->metadata->metadata_array[img->metadata->sz] = metadata;
img->metadata->sz++;
return 0;
fail:
aom_img_metadata_array_free(img->metadata);
img->metadata = NULL;
return -1;
}
void aom_img_remove_metadata(aom_image_t *img) {

View file

@ -111,19 +111,52 @@ void aom_convolve8_vert_c(const uint8_t *src, ptrdiff_t src_stride,
w, h);
}
void aom_convolve8_c(const uint8_t *src, ptrdiff_t src_stride, uint8_t *dst,
ptrdiff_t dst_stride, const InterpKernel *filter,
int x0_q4, int x_step_q4, int y0_q4, int y_step_q4, int w,
int h) {
// Note: Fixed size intermediate buffer, temp, places limits on parameters.
// 2d filtering proceeds in 2 steps:
// (1) Interpolate horizontally into an intermediate buffer, temp.
// (2) Interpolate temp vertically to derive the sub-pixel result.
// Deriving the maximum number of rows in the temp buffer (135):
// --Smallest scaling factor is x1/2 ==> y_step_q4 = 32 (Normative).
// --Largest block size is 64x64 pixels.
// --64 rows in the downscaled frame span a distance of (64 - 1) * 32 in the
// original frame (in 1/16th pixel units).
// --Must round-up because block may be located at sub-pixel position.
// --Require an additional SUBPEL_TAPS rows for the 8-tap filter tails.
// --((64 - 1) * 32 + 15) >> 4 + 8 = 135.
// When calling in frame scaling function, the smallest scaling factor is x1/4
// ==> y_step_q4 = 64. Since w and h are at most 16, the temp buffer is still
// big enough.
uint8_t temp[64 * 135];
const int intermediate_height =
(((h - 1) * y_step_q4 + y0_q4) >> SUBPEL_BITS) + SUBPEL_TAPS;
assert(w <= 64);
assert(h <= 64);
assert(y_step_q4 <= 32 || (y_step_q4 <= 64 && h <= 32));
assert(x_step_q4 <= 64);
convolve_horiz(src - src_stride * (SUBPEL_TAPS / 2 - 1), src_stride, temp, 64,
filter, x0_q4, x_step_q4, w, intermediate_height);
convolve_vert(temp + 64 * (SUBPEL_TAPS / 2 - 1), 64, dst, dst_stride, filter,
y0_q4, y_step_q4, w, h);
}
void aom_scaled_2d_c(const uint8_t *src, ptrdiff_t src_stride, uint8_t *dst,
ptrdiff_t dst_stride, const InterpKernel *filter,
int x0_q4, int x_step_q4, int y0_q4, int y_step_q4, int w,
int h) {
aom_convolve8_c(src, src_stride, dst, dst_stride, filter, x0_q4, x_step_q4,
y0_q4, y_step_q4, w, h);
}
void aom_convolve_copy_c(const uint8_t *src, ptrdiff_t src_stride, uint8_t *dst,
ptrdiff_t dst_stride, const int16_t *filter_x,
int filter_x_stride, const int16_t *filter_y,
int filter_y_stride, int w, int h) {
int r;
(void)filter_x;
(void)filter_x_stride;
(void)filter_y;
(void)filter_y_stride;
for (r = h; r > 0; --r) {
memcpy(dst, src, w);
ptrdiff_t dst_stride, int w, int h) {
for (int r = h; r > 0; --r) {
memmove(dst, src, w);
src += src_stride;
dst += dst_stride;
}
@ -216,22 +249,11 @@ void aom_highbd_convolve8_vert_c(const uint8_t *src, ptrdiff_t src_stride,
y_step_q4, w, h, bd);
}
void aom_highbd_convolve_copy_c(const uint8_t *src8, ptrdiff_t src_stride,
uint8_t *dst8, ptrdiff_t dst_stride,
const int16_t *filter_x, int filter_x_stride,
const int16_t *filter_y, int filter_y_stride,
int w, int h, int bd) {
int r;
uint16_t *src = CONVERT_TO_SHORTPTR(src8);
uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);
(void)filter_x;
(void)filter_y;
(void)filter_x_stride;
(void)filter_y_stride;
(void)bd;
for (r = h; r > 0; --r) {
memcpy(dst, src, w * sizeof(uint16_t));
void aom_highbd_convolve_copy_c(const uint16_t *src, ptrdiff_t src_stride,
uint16_t *dst, ptrdiff_t dst_stride, int w,
int h) {
for (int y = 0; y < h; ++y) {
memmove(dst, src, w * sizeof(src[0]));
src += src_stride;
dst += dst_stride;
}

View file

@ -31,9 +31,12 @@ list(APPEND AOM_DSP_COMMON_SOURCES
"${AOM_ROOT}/aom_dsp/entcode.h"
"${AOM_ROOT}/aom_dsp/fft.c"
"${AOM_ROOT}/aom_dsp/fft_common.h"
"${AOM_ROOT}/aom_dsp/grain_params.h"
"${AOM_ROOT}/aom_dsp/intrapred.c"
"${AOM_ROOT}/aom_dsp/intrapred_common.h"
"${AOM_ROOT}/aom_dsp/loopfilter.c"
"${AOM_ROOT}/aom_dsp/odintrin.c"
"${AOM_ROOT}/aom_dsp/odintrin.h"
"${AOM_ROOT}/aom_dsp/prob.h"
"${AOM_ROOT}/aom_dsp/recenter.h"
"${AOM_ROOT}/aom_dsp/simd/v128_intrinsics.h"
@ -44,11 +47,9 @@ list(APPEND AOM_DSP_COMMON_SOURCES
"${AOM_ROOT}/aom_dsp/simd/v64_intrinsics_c.h"
"${AOM_ROOT}/aom_dsp/subtract.c"
"${AOM_ROOT}/aom_dsp/txfm_common.h"
"${AOM_ROOT}/aom_dsp/x86/convolve_common_intrin.h"
"${AOM_ROOT}/aom_dsp/avg.c")
"${AOM_ROOT}/aom_dsp/x86/convolve_common_intrin.h")
list(APPEND AOM_DSP_COMMON_ASM_SSE2
"${AOM_ROOT}/aom_dsp/x86/aom_convolve_copy_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/aom_high_subpixel_8t_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/aom_high_subpixel_bilinear_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/aom_subpixel_8t_sse2.asm"
@ -58,14 +59,13 @@ list(APPEND AOM_DSP_COMMON_ASM_SSE2
"${AOM_ROOT}/aom_dsp/x86/inv_wht_sse2.asm")
list(APPEND AOM_DSP_COMMON_INTRIN_SSE2
"${AOM_ROOT}/aom_dsp/x86/aom_convolve_copy_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/aom_subpixel_8t_intrin_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/aom_asm_stubs.c"
"${AOM_ROOT}/aom_dsp/x86/convolve.h"
"${AOM_ROOT}/aom_dsp/x86/convolve_sse2.h"
"${AOM_ROOT}/aom_dsp/x86/fft_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_convolve_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_intrapred_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_loopfilter_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/intrapred_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/intrapred_x86.h"
"${AOM_ROOT}/aom_dsp/x86/loopfilter_sse2.c"
@ -74,67 +74,53 @@ list(APPEND AOM_DSP_COMMON_INTRIN_SSE2
"${AOM_ROOT}/aom_dsp/x86/transpose_sse2.h"
"${AOM_ROOT}/aom_dsp/x86/txfm_common_sse2.h"
"${AOM_ROOT}/aom_dsp/x86/sum_squares_sse2.h"
"${AOM_ROOT}/aom_dsp/x86/avg_intrin_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/bitdepth_conversion_sse2.h")
if(NOT CONFIG_AV1_HIGHBITDEPTH)
list(REMOVE_ITEM AOM_DSP_COMMON_INTRIN_SSE2
"${AOM_ROOT}/aom_dsp/x86/highbd_convolve_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_loopfilter_sse2.c")
endif()
list(APPEND AOM_DSP_COMMON_ASM_SSSE3
"${AOM_ROOT}/aom_dsp/x86/aom_subpixel_8t_ssse3.asm"
"${AOM_ROOT}/aom_dsp/x86/aom_subpixel_bilinear_ssse3.asm")
list(APPEND AOM_DSP_COMMON_INTRIN_SSSE3
"${AOM_ROOT}/aom_dsp/x86/aom_subpixel_8t_intrin_ssse3.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_convolve_ssse3.c"
"${AOM_ROOT}/aom_dsp/x86/convolve_ssse3.h"
"${AOM_ROOT}/aom_dsp/x86/intrapred_ssse3.c")
if(NOT CONFIG_AV1_HIGHBITDEPTH)
list(REMOVE_ITEM AOM_DSP_COMMON_INTRIN_SSSE3
"${AOM_ROOT}/aom_dsp/x86/highbd_convolve_ssse3.c")
endif()
list(APPEND AOM_DSP_COMMON_INTRIN_SSE4_1
"${AOM_ROOT}/aom_dsp/x86/blend_mask_sse4.h"
"${AOM_ROOT}/aom_dsp/x86/blend_a64_hmask_sse4.c"
"${AOM_ROOT}/aom_dsp/x86/blend_a64_mask_sse4.c"
"${AOM_ROOT}/aom_dsp/x86/blend_a64_vmask_sse4.c")
"${AOM_ROOT}/aom_dsp/x86/blend_a64_vmask_sse4.c"
"${AOM_ROOT}/aom_dsp/x86/intrapred_sse4.c"
"${AOM_ROOT}/aom_dsp/x86/intrapred_utils.h")
list(APPEND AOM_DSP_COMMON_INTRIN_AVX2
"${AOM_ROOT}/aom_dsp/x86/aom_convolve_copy_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/aom_subpixel_8t_intrin_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/common_avx2.h"
"${AOM_ROOT}/aom_dsp/x86/txfm_common_avx2.h"
"${AOM_ROOT}/aom_dsp/x86/convolve_avx2.h"
"${AOM_ROOT}/aom_dsp/x86/fft_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_convolve_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_loopfilter_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/intrapred_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/loopfilter_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/blend_a64_mask_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/avg_intrin_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/bitdepth_conversion_avx2.h")
if(NOT CONFIG_AV1_HIGHBITDEPTH)
list(REMOVE_ITEM AOM_DSP_COMMON_INTRIN_AVX2
"${AOM_ROOT}/aom_dsp/x86/highbd_convolve_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_loopfilter_avx2.c")
endif()
list(APPEND AOM_DSP_COMMON_INTRIN_NEON "${AOM_ROOT}/aom_dsp/arm/fwd_txfm_neon.c"
list(APPEND AOM_DSP_COMMON_INTRIN_NEON
"${AOM_ROOT}/aom_dsp/arm/aom_convolve_copy_neon.c"
"${AOM_ROOT}/aom_dsp/arm/fwd_txfm_neon.c"
"${AOM_ROOT}/aom_dsp/arm/loopfilter_neon.c"
"${AOM_ROOT}/aom_dsp/arm/highbd_intrapred_neon.c"
"${AOM_ROOT}/aom_dsp/arm/intrapred_neon.c"
"${AOM_ROOT}/aom_dsp/arm/subtract_neon.c"
"${AOM_ROOT}/aom_dsp/arm/blend_a64_mask_neon.c")
list(APPEND AOM_DSP_COMMON_INTRIN_DSPR2
"${AOM_ROOT}/aom_dsp/mips/aom_convolve_copy_dspr2.c"
"${AOM_ROOT}/aom_dsp/mips/common_dspr2.c"
"${AOM_ROOT}/aom_dsp/mips/common_dspr2.h"
"${AOM_ROOT}/aom_dsp/mips/convolve2_dspr2.c"
"${AOM_ROOT}/aom_dsp/mips/convolve2_horiz_dspr2.c"
"${AOM_ROOT}/aom_dsp/mips/convolve2_vert_dspr2.c"
"${AOM_ROOT}/aom_dsp/mips/convolve8_dspr2.c"
"${AOM_ROOT}/aom_dsp/mips/convolve8_horiz_dspr2.c"
"${AOM_ROOT}/aom_dsp/mips/convolve8_vert_dspr2.c"
"${AOM_ROOT}/aom_dsp/mips/convolve_common_dspr2.h"
@ -151,19 +137,34 @@ list(APPEND AOM_DSP_COMMON_INTRIN_MSA
"${AOM_ROOT}/aom_dsp/mips/intrapred_msa.c"
"${AOM_ROOT}/aom_dsp/mips/macros_msa.h")
if(CONFIG_AV1_HIGHBITDEPTH)
list(APPEND AOM_DSP_COMMON_INTRIN_SSE2
"${AOM_ROOT}/aom_dsp/x86/highbd_convolve_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_loopfilter_sse2.c")
list(APPEND AOM_DSP_COMMON_INTRIN_SSSE3
"${AOM_ROOT}/aom_dsp/x86/highbd_convolve_ssse3.c")
list(APPEND AOM_DSP_COMMON_INTRIN_AVX2
"${AOM_ROOT}/aom_dsp/x86/highbd_convolve_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_loopfilter_avx2.c")
list(APPEND AOM_DSP_COMMON_INTRIN_NEON
"${AOM_ROOT}/aom_dsp/arm/highbd_loopfilter_neon.c")
endif()
if(CONFIG_AV1_DECODER)
list(APPEND AOM_DSP_DECODER_SOURCES
"${AOM_ROOT}/aom_dsp/binary_codes_reader.c"
"${AOM_ROOT}/aom_dsp/binary_codes_reader.h"
"${AOM_ROOT}/aom_dsp/bitreader.c"
"${AOM_ROOT}/aom_dsp/bitreader.h" "${AOM_ROOT}/aom_dsp/entdec.c"
"${AOM_ROOT}/aom_dsp/entdec.h"
"${AOM_ROOT}/aom_dsp/grain_synthesis.c"
"${AOM_ROOT}/aom_dsp/grain_synthesis.h")
"${AOM_ROOT}/aom_dsp/entdec.h")
endif()
if(CONFIG_AV1_ENCODER)
list(APPEND AOM_DSP_ENCODER_SOURCES
"${AOM_ROOT}/aom_dsp/avg.c"
"${AOM_ROOT}/aom_dsp/binary_codes_writer.c"
"${AOM_ROOT}/aom_dsp/binary_codes_writer.h"
"${AOM_ROOT}/aom_dsp/bitwriter.c"
@ -183,18 +184,15 @@ if(CONFIG_AV1_ENCODER)
"${AOM_ROOT}/aom_dsp/quantize.c"
"${AOM_ROOT}/aom_dsp/quantize.h"
"${AOM_ROOT}/aom_dsp/sad.c"
"${AOM_ROOT}/aom_dsp/sse.c"
"${AOM_ROOT}/aom_dsp/sad_av1.c"
"${AOM_ROOT}/aom_dsp/sse.c"
"${AOM_ROOT}/aom_dsp/ssim.c"
"${AOM_ROOT}/aom_dsp/ssim.h"
"${AOM_ROOT}/aom_dsp/sum_squares.c"
"${AOM_ROOT}/aom_dsp/variance.c"
"${AOM_ROOT}/aom_dsp/variance.h")
list(APPEND AOM_DSP_ENCODER_ASM_SSE2
"${AOM_ROOT}/aom_dsp/x86/highbd_sad4d_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/highbd_sad_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/highbd_subpel_variance_impl_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/highbd_variance_impl_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/sad4d_sse2.asm"
list(APPEND AOM_DSP_ENCODER_ASM_SSE2 "${AOM_ROOT}/aom_dsp/x86/sad4d_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/sad_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/subpel_variance_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/subtract_sse2.asm")
@ -203,32 +201,23 @@ if(CONFIG_AV1_ENCODER)
"${AOM_ROOT}/aom_dsp/x86/ssim_sse2_x86_64.asm")
list(APPEND AOM_DSP_ENCODER_INTRIN_SSE2
"${AOM_ROOT}/aom_dsp/x86/avg_intrin_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/fwd_txfm_impl_sse2.h"
"${AOM_ROOT}/aom_dsp/x86/fwd_txfm_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/fwd_txfm_sse2.h"
"${AOM_ROOT}/aom_dsp/x86/highbd_quantize_intrin_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_subtract_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_variance_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/quantize_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/adaptive_quantize_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_adaptive_quantize_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/quantize_x86.h"
"${AOM_ROOT}/aom_dsp/x86/blk_sse_sum_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/sum_squares_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/variance_sse2.c")
if(NOT CONFIG_AV1_HIGHBITDEPTH)
list(REMOVE_ITEM AOM_DSP_ENCODER_INTRIN_SSE2
"${AOM_ROOT}/aom_dsp/x86/highbd_adaptive_quantize_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_quantize_intrin_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_subtract_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_variance_sse2.c")
endif()
list(APPEND AOM_DSP_ENCODER_ASM_SSSE3_X86_64
"${AOM_ROOT}/aom_dsp/x86/fwd_txfm_ssse3_x86_64.asm"
"${AOM_ROOT}/aom_dsp/x86/quantize_ssse3_x86_64.asm")
list(APPEND AOM_DSP_ENCODER_INTRIN_AVX2
"${AOM_ROOT}/aom_dsp/x86/avg_intrin_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/masked_sad_intrin_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/subtract_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_quantize_intrin_avx2.c"
@ -236,10 +225,9 @@ if(CONFIG_AV1_ENCODER)
"${AOM_ROOT}/aom_dsp/x86/highbd_adaptive_quantize_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/sad4d_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/sad_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/sad_highbd_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_sad_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/sad_impl_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/variance_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_variance_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/sse_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/variance_impl_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/obmc_sad_avx2.c"
@ -247,8 +235,8 @@ if(CONFIG_AV1_ENCODER)
"${AOM_ROOT}/aom_dsp/x86/blk_sse_sum_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/sum_squares_avx2.c")
list(APPEND AOM_DSP_ENCODER_AVX_ASM_X86_64
"${AOM_ROOT}/aom_dsp/x86/quantize_avx_x86_64.asm")
list(APPEND AOM_DSP_ENCODER_INTRIN_AVX
"${AOM_ROOT}/aom_dsp/x86/aom_quantize_avx.c")
list(APPEND AOM_DSP_ENCODER_INTRIN_SSSE3
"${AOM_ROOT}/aom_dsp/x86/masked_sad_intrin_ssse3.h"
@ -261,40 +249,72 @@ if(CONFIG_AV1_ENCODER)
"${AOM_ROOT}/aom_dsp/x86/jnt_variance_ssse3.c"
"${AOM_ROOT}/aom_dsp/x86/jnt_sad_ssse3.c")
list(APPEND AOM_DSP_ENCODER_INTRIN_SSE4_1
"${AOM_ROOT}/aom_dsp/x86/highbd_variance_sse4.c"
"${AOM_ROOT}/aom_dsp/x86/sse_sse4.c"
list(APPEND AOM_DSP_ENCODER_INTRIN_SSE4_1 "${AOM_ROOT}/aom_dsp/x86/sse_sse4.c"
"${AOM_ROOT}/aom_dsp/x86/obmc_sad_sse4.c"
"${AOM_ROOT}/aom_dsp/x86/obmc_variance_sse4.c")
if(NOT CONFIG_AV1_HIGHBITDEPTH)
list(REMOVE_ITEM AOM_DSP_ENCODER_INTRIN_SSE4_1
"${AOM_ROOT}/aom_dsp/x86/highbd_variance_sse4.c")
endif()
list(APPEND AOM_DSP_ENCODER_INTRIN_NEON "${AOM_ROOT}/aom_dsp/arm/sad4d_neon.c"
"${AOM_ROOT}/aom_dsp/arm/sad_neon.c"
"${AOM_ROOT}/aom_dsp/arm/subpel_variance_neon.c"
"${AOM_ROOT}/aom_dsp/arm/variance_neon.c"
"${AOM_ROOT}/aom_dsp/arm/hadamard_neon.c"
"${AOM_ROOT}/aom_dsp/arm/avg_neon.c"
"${AOM_ROOT}/aom_dsp/arm/sse_neon.c")
"${AOM_ROOT}/aom_dsp/arm/sse_neon.c"
"${AOM_ROOT}/aom_dsp/arm/sum_squares_neon.c")
list(APPEND AOM_DSP_ENCODER_INTRIN_MSA "${AOM_ROOT}/aom_dsp/mips/sad_msa.c"
"${AOM_ROOT}/aom_dsp/mips/subtract_msa.c"
"${AOM_ROOT}/aom_dsp/mips/variance_msa.c"
"${AOM_ROOT}/aom_dsp/mips/sub_pixel_variance_msa.c")
if(CONFIG_AV1_HIGHBITDEPTH)
list(APPEND AOM_DSP_ENCODER_ASM_SSE2
"${AOM_ROOT}/aom_dsp/x86/highbd_sad4d_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/highbd_sad_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/highbd_subpel_variance_impl_sse2.asm"
"${AOM_ROOT}/aom_dsp/x86/highbd_variance_impl_sse2.asm")
list(APPEND AOM_DSP_ENCODER_INTRIN_SSE2
"${AOM_ROOT}/aom_dsp/x86/highbd_adaptive_quantize_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_quantize_intrin_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_subtract_sse2.c"
"${AOM_ROOT}/aom_dsp/x86/highbd_variance_sse2.c")
list(APPEND AOM_DSP_ENCODER_INTRIN_AVX2
"${AOM_ROOT}/aom_dsp/x86/highbd_variance_avx2.c")
list(APPEND AOM_DSP_ENCODER_INTRIN_SSE4_1
"${AOM_ROOT}/aom_dsp/x86/highbd_variance_sse4.c")
list(APPEND AOM_DSP_ENCODER_INTRIN_NEON
"${AOM_ROOT}/aom_dsp/arm/highbd_quantize_neon.c"
"${AOM_ROOT}/aom_dsp/arm/highbd_variance_neon.c")
endif()
if(CONFIG_INTERNAL_STATS)
list(APPEND AOM_DSP_ENCODER_SOURCES "${AOM_ROOT}/aom_dsp/fastssim.c"
"${AOM_ROOT}/aom_dsp/psnrhvs.c" "${AOM_ROOT}/aom_dsp/ssim.c"
"${AOM_ROOT}/aom_dsp/ssim.h")
"${AOM_ROOT}/aom_dsp/psnrhvs.c")
endif()
if(CONFIG_TUNE_VMAF)
list(APPEND AOM_DSP_ENCODER_SOURCES "${AOM_ROOT}/aom_dsp/vmaf.c"
"${AOM_ROOT}/aom_dsp/vmaf.h")
endif()
if(CONFIG_TUNE_BUTTERAUGLI)
list(APPEND AOM_DSP_ENCODER_SOURCES "${AOM_ROOT}/aom_dsp/butteraugli.c"
"${AOM_ROOT}/aom_dsp/butteraugli.h")
endif()
if(CONFIG_REALTIME_ONLY)
list(REMOVE_ITEM AOM_DSP_ENCODER_INTRIN_AVX2
"${AOM_ROOT}/aom_dsp/x86/obmc_sad_avx2.c"
"${AOM_ROOT}/aom_dsp/x86/obmc_variance_avx2.c")
list(REMOVE_ITEM AOM_DSP_ENCODER_INTRIN_SSE4_1
"${AOM_ROOT}/aom_dsp/x86/obmc_sad_sse4.c"
"${AOM_ROOT}/aom_dsp/x86/obmc_variance_sse4.c")
endif()
endif()
# Creates aom_dsp build targets. Must not be called until after libaom target
@ -330,6 +350,9 @@ function(setup_aom_dsp_targets)
if(BUILD_SHARED_LIBS)
target_sources(aom_static PRIVATE $<TARGET_OBJECTS:aom_dsp_encoder>)
endif()
if(CONFIG_TUNE_VMAF)
target_include_directories(aom_dsp_encoder PRIVATE ${VMAF_INCLUDE_DIRS})
endif()
endif()
if(HAVE_SSE2)
@ -372,9 +395,10 @@ function(setup_aom_dsp_targets)
endif()
endif()
if(HAVE_AVX AND "${AOM_TARGET_CPU}" STREQUAL "x86_64")
if(HAVE_AVX)
if(CONFIG_AV1_ENCODER)
add_asm_library("aom_dsp_encoder_avx" "AOM_DSP_ENCODER_AVX_ASM_X86_64")
add_intrinsics_object_library("-mavx" "avx" "aom_dsp_encoder"
"AOM_DSP_ENCODER_INTRIN_AVX")
endif()
endif()

View file

@ -21,6 +21,8 @@
extern "C" {
#endif
#define PI 3.141592653589793238462643383279502884
#ifndef MAX_SB_SIZE
#define MAX_SB_SIZE 128
#endif // ndef MAX_SB_SIZE

1013
media/libaom/src/aom_dsp/aom_dsp_rtcd_defs.pl Normal file → Executable file

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,52 @@
/*
* Copyright (c) 2020, Alliance for Open Media. All Rights Reserved.
*
* Use of this source code is governed by a BSD-style license
* that can be found in the LICENSE file in the root of the source
* tree. An additional intellectual property rights grant can be found
* in the file PATENTS. All contributing project authors may
* be found in the AUTHORS file in the root of the source tree.
*/
#include <arm_neon.h>
#include "config/aom_dsp_rtcd.h"
void aom_convolve_copy_neon(const uint8_t *src, ptrdiff_t src_stride,
uint8_t *dst, ptrdiff_t dst_stride, int w, int h) {
const uint8_t *src1;
uint8_t *dst1;
int y;
if (!(w & 0x0F)) {
for (y = 0; y < h; ++y) {
src1 = src;
dst1 = dst;
for (int x = 0; x < (w >> 4); ++x) {
vst1q_u8(dst1, vld1q_u8(src1));
src1 += 16;
dst1 += 16;
}
src += src_stride;
dst += dst_stride;
}
} else if (!(w & 0x07)) {
for (y = 0; y < h; ++y) {
vst1_u8(dst, vld1_u8(src));
src += src_stride;
dst += dst_stride;
}
} else if (!(w & 0x03)) {
for (y = 0; y < h; ++y) {
vst1_lane_u32((uint32_t *)(dst), vreinterpret_u32_u8(vld1_u8(src)), 0);
src += src_stride;
dst += dst_stride;
}
} else if (!(w & 0x01)) {
for (y = 0; y < h; ++y) {
vst1_lane_u16((uint16_t *)(dst), vreinterpret_u16_u8(vld1_u8(src)), 0);
src += src_stride;
dst += dst_stride;
}
}
}

View file

@ -12,9 +12,10 @@
#include "config/aom_dsp_rtcd.h"
#include "aom/aom_integer.h"
#include "aom_dsp/arm/mem_neon.h"
#include "aom_dsp/arm/sum_neon.h"
#include "av1/common/arm/mem_neon.h"
#include "av1/common/arm/transpose_neon.h"
#include "aom_dsp/arm/transpose_neon.h"
#include "aom_ports/mem.h"
unsigned int aom_avg_4x4_neon(const uint8_t *a, int a_stride) {
const uint8x16_t b = load_unaligned_u8q(a, a_stride);
@ -48,6 +49,16 @@ unsigned int aom_avg_8x8_neon(const uint8_t *a, int a_stride) {
return vget_lane_u32(vrshr_n_u32(d, 6), 0);
}
void aom_avg_8x8_quad_neon(const uint8_t *s, int p, int x16_idx, int y16_idx,
int *avg) {
for (int k = 0; k < 4; k++) {
const int x8_idx = x16_idx + ((k & 1) << 3);
const int y8_idx = y16_idx + ((k >> 1) << 3);
const uint8_t *s_tmp = s + y8_idx * p + x8_idx;
avg[k] = aom_avg_8x8_neon(s_tmp, p);
}
}
int aom_satd_lp_neon(const int16_t *coeff, int length) {
const int16x4_t zero = vdup_n_s16(0);
int32x4_t accum = vdupq_n_s32(0);
@ -72,3 +83,142 @@ int aom_satd_lp_neon(const int16_t *coeff, int length) {
return satd;
}
}
void aom_int_pro_row_neon(int16_t hbuf[16], const uint8_t *ref,
const int ref_stride, const int height) {
int i;
const uint8_t *idx = ref;
uint16x8_t vec0 = vdupq_n_u16(0);
uint16x8_t vec1 = vec0;
uint8x16_t tmp;
for (i = 0; i < height; ++i) {
tmp = vld1q_u8(idx);
idx += ref_stride;
vec0 = vaddw_u8(vec0, vget_low_u8(tmp));
vec1 = vaddw_u8(vec1, vget_high_u8(tmp));
}
if (128 == height) {
vec0 = vshrq_n_u16(vec0, 6);
vec1 = vshrq_n_u16(vec1, 6);
} else if (64 == height) {
vec0 = vshrq_n_u16(vec0, 5);
vec1 = vshrq_n_u16(vec1, 5);
} else if (32 == height) {
vec0 = vshrq_n_u16(vec0, 4);
vec1 = vshrq_n_u16(vec1, 4);
} else if (16 == height) {
vec0 = vshrq_n_u16(vec0, 3);
vec1 = vshrq_n_u16(vec1, 3);
}
vst1q_s16(hbuf, vreinterpretq_s16_u16(vec0));
hbuf += 8;
vst1q_s16(hbuf, vreinterpretq_s16_u16(vec1));
}
int16_t aom_int_pro_col_neon(const uint8_t *ref, const int width) {
const uint8_t *idx;
uint16x8_t sum = vdupq_n_u16(0);
for (idx = ref; idx < (ref + width); idx += 16) {
uint8x16_t vec = vld1q_u8(idx);
sum = vaddq_u16(sum, vpaddlq_u8(vec));
}
#if defined(__aarch64__)
return (int16_t)vaddvq_u16(sum);
#else
const uint32x4_t a = vpaddlq_u16(sum);
const uint64x2_t b = vpaddlq_u32(a);
const uint32x2_t c = vadd_u32(vreinterpret_u32_u64(vget_low_u64(b)),
vreinterpret_u32_u64(vget_high_u64(b)));
return (int16_t)vget_lane_u32(c, 0);
#endif
}
// coeff: 16 bits, dynamic range [-32640, 32640].
// length: value range {16, 64, 256, 1024}.
int aom_satd_neon(const tran_low_t *coeff, int length) {
const int32x4_t zero = vdupq_n_s32(0);
int32x4_t accum = zero;
do {
const int32x4_t src0 = vld1q_s32(&coeff[0]);
const int32x4_t src8 = vld1q_s32(&coeff[4]);
const int32x4_t src16 = vld1q_s32(&coeff[8]);
const int32x4_t src24 = vld1q_s32(&coeff[12]);
accum = vabaq_s32(accum, src0, zero);
accum = vabaq_s32(accum, src8, zero);
accum = vabaq_s32(accum, src16, zero);
accum = vabaq_s32(accum, src24, zero);
length -= 16;
coeff += 16;
} while (length != 0);
// satd: 26 bits, dynamic range [-32640 * 1024, 32640 * 1024]
#ifdef __aarch64__
return vaddvq_s32(accum);
#else
return horizontal_add_s32x4(accum);
#endif // __aarch64__
}
int aom_vector_var_neon(const int16_t *ref, const int16_t *src, const int bwl) {
int32x4_t v_mean = vdupq_n_s32(0);
int32x4_t v_sse = v_mean;
int16x8_t v_ref, v_src;
int16x4_t v_low;
int i, width = 4 << bwl;
for (i = 0; i < width; i += 8) {
v_ref = vld1q_s16(&ref[i]);
v_src = vld1q_s16(&src[i]);
const int16x8_t diff = vsubq_s16(v_ref, v_src);
// diff: dynamic range [-510, 510], 10 bits.
v_mean = vpadalq_s16(v_mean, diff);
v_low = vget_low_s16(diff);
v_sse = vmlal_s16(v_sse, v_low, v_low);
#if defined(__aarch64__)
v_sse = vmlal_high_s16(v_sse, diff, diff);
#else
const int16x4_t v_high = vget_high_s16(diff);
v_sse = vmlal_s16(v_sse, v_high, v_high);
#endif
}
#if defined(__aarch64__)
int mean = vaddvq_s32(v_mean);
int sse = (int)vaddvq_s32(v_sse);
#else
int mean = horizontal_add_s32x4(v_mean);
int sse = horizontal_add_s32x4(v_sse);
#endif
// (mean * mean): dynamic range 31 bits.
int var = sse - ((mean * mean) >> (bwl + 2));
return var;
}
#if CONFIG_AV1_HIGHBITDEPTH
unsigned int aom_highbd_avg_4x4_neon(const uint8_t *s, int p) {
const uint16_t *src = CONVERT_TO_SHORTPTR(s);
const uint16x4_t r0 = vld1_u16(src);
src += p;
uint16x4_t r1, r2, r3;
r1 = vld1_u16(src);
src += p;
r2 = vld1_u16(src);
src += p;
r3 = vld1_u16(src);
const uint16x4_t s1 = vadd_u16(r0, r1);
const uint16x4_t s2 = vadd_u16(r2, r3);
const uint16x4_t s3 = vadd_u16(s1, s2);
#if defined(__aarch64__)
return (vaddv_u16(s3) + 8) >> 4;
#else
const uint16x4_t h1 = vpadd_u16(s3, s3);
const uint16x4_t h2 = vpadd_u16(h1, h1);
const uint16x4_t res = vrshr_n_u16(h2, 4);
return vget_lane_u16(res, 0);
#endif
}
#endif // CONFIG_AV1_HIGHBITDEPTH

View file

@ -15,8 +15,8 @@
#include "aom/aom_integer.h"
#include "aom_dsp/aom_dsp_common.h"
#include "aom_dsp/blend.h"
#include "aom_dsp/arm/mem_neon.h"
#include "aom_ports/mem.h"
#include "av1/common/arm/mem_neon.h"
#include "config/aom_dsp_rtcd.h"
static INLINE void blend8x1(int16x8_t mask, int16x8_t src_0, int16x8_t src_1,

View file

@ -14,8 +14,8 @@
#include "config/aom_config.h"
#include "aom_dsp/txfm_common.h"
#include "av1/common/arm/mem_neon.h"
#include "av1/common/arm/transpose_neon.h"
#include "aom_dsp/arm/mem_neon.h"
#include "aom_dsp/arm/transpose_neon.h"
static void aom_fdct4x4_helper(const int16_t *input, int stride,
int16x4_t *input_0, int16x4_t *input_1,

View file

@ -12,8 +12,8 @@
#include "config/aom_dsp_rtcd.h"
#include "aom/aom_integer.h"
#include "av1/common/arm/mem_neon.h"
#include "av1/common/arm/transpose_neon.h"
#include "aom_dsp/arm/mem_neon.h"
#include "aom_dsp/arm/transpose_neon.h"
static void hadamard8x8_one_pass(int16x8_t *a0, int16x8_t *a1, int16x8_t *a2,
int16x8_t *a3, int16x8_t *a4, int16x8_t *a5,
@ -104,6 +104,13 @@ void aom_hadamard_lp_8x8_neon(const int16_t *src_diff, ptrdiff_t src_stride,
vst1q_s16(coeff + 56, a7);
}
void aom_hadamard_8x8_dual_neon(const int16_t *src_diff, ptrdiff_t src_stride,
int16_t *coeff) {
for (int i = 0; i < 2; i++) {
aom_hadamard_lp_8x8_neon(src_diff + (i * 8), src_stride, coeff + (i * 64));
}
}
void aom_hadamard_lp_16x16_neon(const int16_t *src_diff, ptrdiff_t src_stride,
int16_t *coeff) {
/* Rearrange 16x16 to 8x32 and remove stride.

View file

@ -0,0 +1,835 @@
/*
* Copyright (c) 2022, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include <arm_neon.h>
#include "config/aom_config.h"
#include "config/aom_dsp_rtcd.h"
#include "aom/aom_integer.h"
#include "aom_dsp/intrapred_common.h"
// -----------------------------------------------------------------------------
// DC
static INLINE void highbd_dc_predictor(uint16_t *dst, ptrdiff_t stride, int bw,
const uint16_t *above,
const uint16_t *left) {
assert(bw >= 4);
assert(IS_POWER_OF_TWO(bw));
int expected_dc, sum = 0;
const int count = bw * 2;
uint32x4_t sum_q = vdupq_n_u32(0);
uint32x2_t sum_d;
uint16_t *dst_1;
if (bw >= 8) {
for (int i = 0; i < bw; i += 8) {
sum_q = vpadalq_u16(sum_q, vld1q_u16(above));
sum_q = vpadalq_u16(sum_q, vld1q_u16(left));
above += 8;
left += 8;
}
sum_d = vadd_u32(vget_low_u32(sum_q), vget_high_u32(sum_q));
sum = vget_lane_s32(vreinterpret_s32_u64(vpaddl_u32(sum_d)), 0);
expected_dc = (sum + (count >> 1)) / count;
const uint16x8_t dc = vdupq_n_u16((uint16_t)expected_dc);
for (int r = 0; r < bw; r++) {
dst_1 = dst;
for (int i = 0; i < bw; i += 8) {
vst1q_u16(dst_1, dc);
dst_1 += 8;
}
dst += stride;
}
} else { // 4x4
sum_q = vaddl_u16(vld1_u16(above), vld1_u16(left));
sum_d = vadd_u32(vget_low_u32(sum_q), vget_high_u32(sum_q));
sum = vget_lane_s32(vreinterpret_s32_u64(vpaddl_u32(sum_d)), 0);
expected_dc = (sum + (count >> 1)) / count;
const uint16x4_t dc = vdup_n_u16((uint16_t)expected_dc);
for (int r = 0; r < bw; r++) {
vst1_u16(dst, dc);
dst += stride;
}
}
}
#define INTRA_PRED_HIGHBD_SIZED_NEON(type, width) \
void aom_highbd_##type##_predictor_##width##x##width##_neon( \
uint16_t *dst, ptrdiff_t stride, const uint16_t *above, \
const uint16_t *left, int bd) { \
(void)bd; \
highbd_##type##_predictor(dst, stride, width, above, left); \
}
#define INTRA_PRED_SQUARE(type) \
INTRA_PRED_HIGHBD_SIZED_NEON(type, 4) \
INTRA_PRED_HIGHBD_SIZED_NEON(type, 8) \
INTRA_PRED_HIGHBD_SIZED_NEON(type, 16) \
INTRA_PRED_HIGHBD_SIZED_NEON(type, 32) \
INTRA_PRED_HIGHBD_SIZED_NEON(type, 64)
INTRA_PRED_SQUARE(dc)
#undef INTRA_PRED_SQUARE
// -----------------------------------------------------------------------------
// V_PRED
#define HIGHBD_V_NXM(W, H) \
void aom_highbd_v_predictor_##W##x##H##_neon( \
uint16_t *dst, ptrdiff_t stride, const uint16_t *above, \
const uint16_t *left, int bd) { \
(void)left; \
(void)bd; \
vertical##W##xh_neon(dst, stride, above, H); \
}
static INLINE uint16x8x2_t load_uint16x8x2(uint16_t const *ptr) {
uint16x8x2_t x;
// Clang/gcc uses ldp here.
x.val[0] = vld1q_u16(ptr);
x.val[1] = vld1q_u16(ptr + 8);
return x;
}
static INLINE void store_uint16x8x2(uint16_t *ptr, uint16x8x2_t x) {
vst1q_u16(ptr, x.val[0]);
vst1q_u16(ptr + 8, x.val[1]);
}
static INLINE void vertical4xh_neon(uint16_t *dst, ptrdiff_t stride,
const uint16_t *const above, int height) {
const uint16x4_t row = vld1_u16(above);
int y = height;
do {
vst1_u16(dst, row);
vst1_u16(dst + stride, row);
dst += stride << 1;
y -= 2;
} while (y != 0);
}
static INLINE void vertical8xh_neon(uint16_t *dst, ptrdiff_t stride,
const uint16_t *const above, int height) {
const uint16x8_t row = vld1q_u16(above);
int y = height;
do {
vst1q_u16(dst, row);
vst1q_u16(dst + stride, row);
dst += stride << 1;
y -= 2;
} while (y != 0);
}
static INLINE void vertical16xh_neon(uint16_t *dst, ptrdiff_t stride,
const uint16_t *const above, int height) {
const uint16x8x2_t row = load_uint16x8x2(above);
int y = height;
do {
store_uint16x8x2(dst, row);
store_uint16x8x2(dst + stride, row);
dst += stride << 1;
y -= 2;
} while (y != 0);
}
static INLINE uint16x8x4_t load_uint16x8x4(uint16_t const *ptr) {
uint16x8x4_t x;
// Clang/gcc uses ldp here.
x.val[0] = vld1q_u16(ptr);
x.val[1] = vld1q_u16(ptr + 8);
x.val[2] = vld1q_u16(ptr + 16);
x.val[3] = vld1q_u16(ptr + 24);
return x;
}
static INLINE void store_uint16x8x4(uint16_t *ptr, uint16x8x4_t x) {
vst1q_u16(ptr, x.val[0]);
vst1q_u16(ptr + 8, x.val[1]);
vst1q_u16(ptr + 16, x.val[2]);
vst1q_u16(ptr + 24, x.val[3]);
}
static INLINE void vertical32xh_neon(uint16_t *dst, ptrdiff_t stride,
const uint16_t *const above, int height) {
const uint16x8x4_t row = load_uint16x8x4(above);
int y = height;
do {
store_uint16x8x4(dst, row);
store_uint16x8x4(dst + stride, row);
dst += stride << 1;
y -= 2;
} while (y != 0);
}
static INLINE void vertical64xh_neon(uint16_t *dst, ptrdiff_t stride,
const uint16_t *const above, int height) {
uint16_t *dst32 = dst + 32;
const uint16x8x4_t row = load_uint16x8x4(above);
const uint16x8x4_t row32 = load_uint16x8x4(above + 32);
int y = height;
do {
store_uint16x8x4(dst, row);
store_uint16x8x4(dst32, row32);
store_uint16x8x4(dst + stride, row);
store_uint16x8x4(dst32 + stride, row32);
dst += stride << 1;
dst32 += stride << 1;
y -= 2;
} while (y != 0);
}
HIGHBD_V_NXM(4, 4)
HIGHBD_V_NXM(4, 8)
HIGHBD_V_NXM(4, 16)
HIGHBD_V_NXM(8, 4)
HIGHBD_V_NXM(8, 8)
HIGHBD_V_NXM(8, 16)
HIGHBD_V_NXM(8, 32)
HIGHBD_V_NXM(16, 4)
HIGHBD_V_NXM(16, 8)
HIGHBD_V_NXM(16, 16)
HIGHBD_V_NXM(16, 32)
HIGHBD_V_NXM(16, 64)
HIGHBD_V_NXM(32, 8)
HIGHBD_V_NXM(32, 16)
HIGHBD_V_NXM(32, 32)
HIGHBD_V_NXM(32, 64)
HIGHBD_V_NXM(64, 16)
HIGHBD_V_NXM(64, 32)
HIGHBD_V_NXM(64, 64)
// -----------------------------------------------------------------------------
// PAETH
static INLINE void highbd_paeth_4or8_x_h_neon(uint16_t *dest, ptrdiff_t stride,
const uint16_t *const top_row,
const uint16_t *const left_column,
int width, int height) {
const uint16x8_t top_left = vdupq_n_u16(top_row[-1]);
const uint16x8_t top_left_x2 = vdupq_n_u16(top_row[-1] + top_row[-1]);
uint16x8_t top;
if (width == 4) {
top = vcombine_u16(vld1_u16(top_row), vdup_n_u16(0));
} else { // width == 8
top = vld1q_u16(top_row);
}
for (int y = 0; y < height; ++y) {
const uint16x8_t left = vdupq_n_u16(left_column[y]);
const uint16x8_t left_dist = vabdq_u16(top, top_left);
const uint16x8_t top_dist = vabdq_u16(left, top_left);
const uint16x8_t top_left_dist =
vabdq_u16(vaddq_u16(top, left), top_left_x2);
const uint16x8_t left_le_top = vcleq_u16(left_dist, top_dist);
const uint16x8_t left_le_top_left = vcleq_u16(left_dist, top_left_dist);
const uint16x8_t top_le_top_left = vcleq_u16(top_dist, top_left_dist);
// if (left_dist <= top_dist && left_dist <= top_left_dist)
const uint16x8_t left_mask = vandq_u16(left_le_top, left_le_top_left);
// dest[x] = left_column[y];
// Fill all the unused spaces with 'top'. They will be overwritten when
// the positions for top_left are known.
uint16x8_t result = vbslq_u16(left_mask, left, top);
// else if (top_dist <= top_left_dist)
// dest[x] = top_row[x];
// Add these values to the mask. They were already set.
const uint16x8_t left_or_top_mask = vorrq_u16(left_mask, top_le_top_left);
// else
// dest[x] = top_left;
result = vbslq_u16(left_or_top_mask, result, top_left);
if (width == 4) {
vst1_u16(dest, vget_low_u16(result));
} else { // width == 8
vst1q_u16(dest, result);
}
dest += stride;
}
}
#define HIGHBD_PAETH_NXM(W, H) \
void aom_highbd_paeth_predictor_##W##x##H##_neon( \
uint16_t *dst, ptrdiff_t stride, const uint16_t *above, \
const uint16_t *left, int bd) { \
(void)bd; \
highbd_paeth_4or8_x_h_neon(dst, stride, above, left, W, H); \
}
HIGHBD_PAETH_NXM(4, 4)
HIGHBD_PAETH_NXM(4, 8)
HIGHBD_PAETH_NXM(4, 16)
HIGHBD_PAETH_NXM(8, 4)
HIGHBD_PAETH_NXM(8, 8)
HIGHBD_PAETH_NXM(8, 16)
HIGHBD_PAETH_NXM(8, 32)
// Select the closest values and collect them.
static INLINE uint16x8_t select_paeth(const uint16x8_t top,
const uint16x8_t left,
const uint16x8_t top_left,
const uint16x8_t left_le_top,
const uint16x8_t left_le_top_left,
const uint16x8_t top_le_top_left) {
// if (left_dist <= top_dist && left_dist <= top_left_dist)
const uint16x8_t left_mask = vandq_u16(left_le_top, left_le_top_left);
// dest[x] = left_column[y];
// Fill all the unused spaces with 'top'. They will be overwritten when
// the positions for top_left are known.
const uint16x8_t result = vbslq_u16(left_mask, left, top);
// else if (top_dist <= top_left_dist)
// dest[x] = top_row[x];
// Add these values to the mask. They were already set.
const uint16x8_t left_or_top_mask = vorrq_u16(left_mask, top_le_top_left);
// else
// dest[x] = top_left;
return vbslq_u16(left_or_top_mask, result, top_left);
}
#define PAETH_PREDICTOR(num) \
do { \
const uint16x8_t left_dist = vabdq_u16(top[num], top_left); \
const uint16x8_t top_left_dist = \
vabdq_u16(vaddq_u16(top[num], left), top_left_x2); \
const uint16x8_t left_le_top = vcleq_u16(left_dist, top_dist); \
const uint16x8_t left_le_top_left = vcleq_u16(left_dist, top_left_dist); \
const uint16x8_t top_le_top_left = vcleq_u16(top_dist, top_left_dist); \
const uint16x8_t result = \
select_paeth(top[num], left, top_left, left_le_top, left_le_top_left, \
top_le_top_left); \
vst1q_u16(dest + (num * 8), result); \
} while (0)
#define LOAD_TOP_ROW(num) vld1q_u16(top_row + (num * 8))
static INLINE void highbd_paeth16_plus_x_h_neon(
uint16_t *dest, ptrdiff_t stride, const uint16_t *const top_row,
const uint16_t *const left_column, int width, int height) {
const uint16x8_t top_left = vdupq_n_u16(top_row[-1]);
const uint16x8_t top_left_x2 = vdupq_n_u16(top_row[-1] + top_row[-1]);
uint16x8_t top[8];
top[0] = LOAD_TOP_ROW(0);
top[1] = LOAD_TOP_ROW(1);
if (width > 16) {
top[2] = LOAD_TOP_ROW(2);
top[3] = LOAD_TOP_ROW(3);
if (width == 64) {
top[4] = LOAD_TOP_ROW(4);
top[5] = LOAD_TOP_ROW(5);
top[6] = LOAD_TOP_ROW(6);
top[7] = LOAD_TOP_ROW(7);
}
}
for (int y = 0; y < height; ++y) {
const uint16x8_t left = vdupq_n_u16(left_column[y]);
const uint16x8_t top_dist = vabdq_u16(left, top_left);
PAETH_PREDICTOR(0);
PAETH_PREDICTOR(1);
if (width > 16) {
PAETH_PREDICTOR(2);
PAETH_PREDICTOR(3);
if (width == 64) {
PAETH_PREDICTOR(4);
PAETH_PREDICTOR(5);
PAETH_PREDICTOR(6);
PAETH_PREDICTOR(7);
}
}
dest += stride;
}
}
#define HIGHBD_PAETH_NXM_WIDE(W, H) \
void aom_highbd_paeth_predictor_##W##x##H##_neon( \
uint16_t *dst, ptrdiff_t stride, const uint16_t *above, \
const uint16_t *left, int bd) { \
(void)bd; \
highbd_paeth16_plus_x_h_neon(dst, stride, above, left, W, H); \
}
HIGHBD_PAETH_NXM_WIDE(16, 4)
HIGHBD_PAETH_NXM_WIDE(16, 8)
HIGHBD_PAETH_NXM_WIDE(16, 16)
HIGHBD_PAETH_NXM_WIDE(16, 32)
HIGHBD_PAETH_NXM_WIDE(16, 64)
HIGHBD_PAETH_NXM_WIDE(32, 8)
HIGHBD_PAETH_NXM_WIDE(32, 16)
HIGHBD_PAETH_NXM_WIDE(32, 32)
HIGHBD_PAETH_NXM_WIDE(32, 64)
HIGHBD_PAETH_NXM_WIDE(64, 16)
HIGHBD_PAETH_NXM_WIDE(64, 32)
HIGHBD_PAETH_NXM_WIDE(64, 64)
// -----------------------------------------------------------------------------
// SMOOTH
// 256 - v = vneg_s8(v)
static INLINE uint16x4_t negate_s8(const uint16x4_t v) {
return vreinterpret_u16_s8(vneg_s8(vreinterpret_s8_u16(v)));
}
static INLINE void highbd_smooth_4xh_neon(uint16_t *dst, ptrdiff_t stride,
const uint16_t *const top_row,
const uint16_t *const left_column,
const int height) {
const uint16_t top_right = top_row[3];
const uint16_t bottom_left = left_column[height - 1];
const uint16_t *const weights_y = smooth_weights_u16 + height - 4;
const uint16x4_t top_v = vld1_u16(top_row);
const uint16x4_t bottom_left_v = vdup_n_u16(bottom_left);
const uint16x4_t weights_x_v = vld1_u16(smooth_weights_u16);
const uint16x4_t scaled_weights_x = negate_s8(weights_x_v);
const uint32x4_t weighted_tr = vmull_n_u16(scaled_weights_x, top_right);
for (int y = 0; y < height; ++y) {
// Each variable in the running summation is named for the last item to be
// accumulated.
const uint32x4_t weighted_top =
vmlal_n_u16(weighted_tr, top_v, weights_y[y]);
const uint32x4_t weighted_left =
vmlal_n_u16(weighted_top, weights_x_v, left_column[y]);
const uint32x4_t weighted_bl =
vmlal_n_u16(weighted_left, bottom_left_v, 256 - weights_y[y]);
const uint16x4_t pred =
vrshrn_n_u32(weighted_bl, SMOOTH_WEIGHT_LOG2_SCALE + 1);
vst1_u16(dst, pred);
dst += stride;
}
}
// Common code between 8xH and [16|32|64]xH.
static INLINE void highbd_calculate_pred8(
uint16_t *dst, const uint32x4_t weighted_corners_low,
const uint32x4_t weighted_corners_high, const uint16x4x2_t top_vals,
const uint16x4x2_t weights_x, const uint16_t left_y,
const uint16_t weight_y) {
// Each variable in the running summation is named for the last item to be
// accumulated.
const uint32x4_t weighted_top_low =
vmlal_n_u16(weighted_corners_low, top_vals.val[0], weight_y);
const uint32x4_t weighted_edges_low =
vmlal_n_u16(weighted_top_low, weights_x.val[0], left_y);
const uint16x4_t pred_low =
vrshrn_n_u32(weighted_edges_low, SMOOTH_WEIGHT_LOG2_SCALE + 1);
vst1_u16(dst, pred_low);
const uint32x4_t weighted_top_high =
vmlal_n_u16(weighted_corners_high, top_vals.val[1], weight_y);
const uint32x4_t weighted_edges_high =
vmlal_n_u16(weighted_top_high, weights_x.val[1], left_y);
const uint16x4_t pred_high =
vrshrn_n_u32(weighted_edges_high, SMOOTH_WEIGHT_LOG2_SCALE + 1);
vst1_u16(dst + 4, pred_high);
}
static void highbd_smooth_8xh_neon(uint16_t *dst, ptrdiff_t stride,
const uint16_t *const top_row,
const uint16_t *const left_column,
const int height) {
const uint16_t top_right = top_row[7];
const uint16_t bottom_left = left_column[height - 1];
const uint16_t *const weights_y = smooth_weights_u16 + height - 4;
const uint16x4x2_t top_vals = { { vld1_u16(top_row),
vld1_u16(top_row + 4) } };
const uint16x4_t bottom_left_v = vdup_n_u16(bottom_left);
const uint16x4x2_t weights_x = { { vld1_u16(smooth_weights_u16 + 4),
vld1_u16(smooth_weights_u16 + 8) } };
const uint32x4_t weighted_tr_low =
vmull_n_u16(negate_s8(weights_x.val[0]), top_right);
const uint32x4_t weighted_tr_high =
vmull_n_u16(negate_s8(weights_x.val[1]), top_right);
for (int y = 0; y < height; ++y) {
const uint32x4_t weighted_bl =
vmull_n_u16(bottom_left_v, 256 - weights_y[y]);
const uint32x4_t weighted_corners_low =
vaddq_u32(weighted_bl, weighted_tr_low);
const uint32x4_t weighted_corners_high =
vaddq_u32(weighted_bl, weighted_tr_high);
highbd_calculate_pred8(dst, weighted_corners_low, weighted_corners_high,
top_vals, weights_x, left_column[y], weights_y[y]);
dst += stride;
}
}
#define HIGHBD_SMOOTH_NXM(W, H) \
void aom_highbd_smooth_predictor_##W##x##H##_neon( \
uint16_t *dst, ptrdiff_t y_stride, const uint16_t *above, \
const uint16_t *left, int bd) { \
(void)bd; \
highbd_smooth_##W##xh_neon(dst, y_stride, above, left, H); \
}
HIGHBD_SMOOTH_NXM(4, 4)
HIGHBD_SMOOTH_NXM(4, 8)
HIGHBD_SMOOTH_NXM(8, 4)
HIGHBD_SMOOTH_NXM(8, 8)
HIGHBD_SMOOTH_NXM(4, 16)
HIGHBD_SMOOTH_NXM(8, 16)
HIGHBD_SMOOTH_NXM(8, 32)
#undef HIGHBD_SMOOTH_NXM
// For width 16 and above.
#define HIGHBD_SMOOTH_PREDICTOR(W) \
static void highbd_smooth_##W##xh_neon( \
uint16_t *dst, ptrdiff_t stride, const uint16_t *const top_row, \
const uint16_t *const left_column, const int height) { \
const uint16_t top_right = top_row[(W)-1]; \
const uint16_t bottom_left = left_column[height - 1]; \
const uint16_t *const weights_y = smooth_weights_u16 + height - 4; \
\
/* Precompute weighted values that don't vary with |y|. */ \
uint32x4_t weighted_tr_low[(W) >> 3]; \
uint32x4_t weighted_tr_high[(W) >> 3]; \
for (int i = 0; i<(W)>> 3; ++i) { \
const int x = i << 3; \
const uint16x4_t weights_x_low = \
vld1_u16(smooth_weights_u16 + (W)-4 + x); \
weighted_tr_low[i] = vmull_n_u16(negate_s8(weights_x_low), top_right); \
const uint16x4_t weights_x_high = \
vld1_u16(smooth_weights_u16 + (W) + x); \
weighted_tr_high[i] = vmull_n_u16(negate_s8(weights_x_high), top_right); \
} \
\
const uint16x4_t bottom_left_v = vdup_n_u16(bottom_left); \
for (int y = 0; y < height; ++y) { \
const uint32x4_t weighted_bl = \
vmull_n_u16(bottom_left_v, 256 - weights_y[y]); \
uint16_t *dst_x = dst; \
for (int i = 0; i<(W)>> 3; ++i) { \
const int x = i << 3; \
const uint16x4x2_t top_vals = { { vld1_u16(top_row + x), \
vld1_u16(top_row + x + 4) } }; \
const uint32x4_t weighted_corners_low = \
vaddq_u32(weighted_bl, weighted_tr_low[i]); \
const uint32x4_t weighted_corners_high = \
vaddq_u32(weighted_bl, weighted_tr_high[i]); \
/* Accumulate weighted edge values and store. */ \
const uint16x4x2_t weights_x = { \
{ vld1_u16(smooth_weights_u16 + (W)-4 + x), \
vld1_u16(smooth_weights_u16 + (W) + x) } \
}; \
highbd_calculate_pred8(dst_x, weighted_corners_low, \
weighted_corners_high, top_vals, weights_x, \
left_column[y], weights_y[y]); \
dst_x += 8; \
} \
dst += stride; \
} \
}
HIGHBD_SMOOTH_PREDICTOR(16)
HIGHBD_SMOOTH_PREDICTOR(32)
HIGHBD_SMOOTH_PREDICTOR(64)
#undef HIGHBD_SMOOTH_PREDICTOR
#define HIGHBD_SMOOTH_NXM_WIDE(W, H) \
void aom_highbd_smooth_predictor_##W##x##H##_neon( \
uint16_t *dst, ptrdiff_t y_stride, const uint16_t *above, \
const uint16_t *left, int bd) { \
(void)bd; \
highbd_smooth_##W##xh_neon(dst, y_stride, above, left, H); \
}
HIGHBD_SMOOTH_NXM_WIDE(16, 4)
HIGHBD_SMOOTH_NXM_WIDE(16, 8)
HIGHBD_SMOOTH_NXM_WIDE(16, 16)
HIGHBD_SMOOTH_NXM_WIDE(16, 32)
HIGHBD_SMOOTH_NXM_WIDE(16, 64)
HIGHBD_SMOOTH_NXM_WIDE(32, 8)
HIGHBD_SMOOTH_NXM_WIDE(32, 16)
HIGHBD_SMOOTH_NXM_WIDE(32, 32)
HIGHBD_SMOOTH_NXM_WIDE(32, 64)
HIGHBD_SMOOTH_NXM_WIDE(64, 16)
HIGHBD_SMOOTH_NXM_WIDE(64, 32)
HIGHBD_SMOOTH_NXM_WIDE(64, 64)
#undef HIGHBD_SMOOTH_NXM_WIDE
static void highbd_smooth_v_4xh_neon(uint16_t *dst, ptrdiff_t stride,
const uint16_t *const top_row,
const uint16_t *const left_column,
const int height) {
const uint16_t bottom_left = left_column[height - 1];
const uint16_t *const weights_y = smooth_weights_u16 + height - 4;
const uint16x4_t top_v = vld1_u16(top_row);
const uint16x4_t bottom_left_v = vdup_n_u16(bottom_left);
for (int y = 0; y < height; ++y) {
const uint32x4_t weighted_bl =
vmull_n_u16(bottom_left_v, 256 - weights_y[y]);
const uint32x4_t weighted_top =
vmlal_n_u16(weighted_bl, top_v, weights_y[y]);
vst1_u16(dst, vrshrn_n_u32(weighted_top, SMOOTH_WEIGHT_LOG2_SCALE));
dst += stride;
}
}
static void highbd_smooth_v_8xh_neon(uint16_t *dst, const ptrdiff_t stride,
const uint16_t *const top_row,
const uint16_t *const left_column,
const int height) {
const uint16_t bottom_left = left_column[height - 1];
const uint16_t *const weights_y = smooth_weights_u16 + height - 4;
const uint16x4_t top_low = vld1_u16(top_row);
const uint16x4_t top_high = vld1_u16(top_row + 4);
const uint16x4_t bottom_left_v = vdup_n_u16(bottom_left);
for (int y = 0; y < height; ++y) {
const uint32x4_t weighted_bl =
vmull_n_u16(bottom_left_v, 256 - weights_y[y]);
const uint32x4_t weighted_top_low =
vmlal_n_u16(weighted_bl, top_low, weights_y[y]);
vst1_u16(dst, vrshrn_n_u32(weighted_top_low, SMOOTH_WEIGHT_LOG2_SCALE));
const uint32x4_t weighted_top_high =
vmlal_n_u16(weighted_bl, top_high, weights_y[y]);
vst1_u16(dst + 4,
vrshrn_n_u32(weighted_top_high, SMOOTH_WEIGHT_LOG2_SCALE));
dst += stride;
}
}
#define HIGHBD_SMOOTH_V_NXM(W, H) \
void aom_highbd_smooth_v_predictor_##W##x##H##_neon( \
uint16_t *dst, ptrdiff_t y_stride, const uint16_t *above, \
const uint16_t *left, int bd) { \
(void)bd; \
highbd_smooth_v_##W##xh_neon(dst, y_stride, above, left, H); \
}
HIGHBD_SMOOTH_V_NXM(4, 4)
HIGHBD_SMOOTH_V_NXM(4, 8)
HIGHBD_SMOOTH_V_NXM(4, 16)
HIGHBD_SMOOTH_V_NXM(8, 4)
HIGHBD_SMOOTH_V_NXM(8, 8)
HIGHBD_SMOOTH_V_NXM(8, 16)
HIGHBD_SMOOTH_V_NXM(8, 32)
#undef HIGHBD_SMOOTH_V_NXM
// For width 16 and above.
#define HIGHBD_SMOOTH_V_PREDICTOR(W) \
static void highbd_smooth_v_##W##xh_neon( \
uint16_t *dst, const ptrdiff_t stride, const uint16_t *const top_row, \
const uint16_t *const left_column, const int height) { \
const uint16_t bottom_left = left_column[height - 1]; \
const uint16_t *const weights_y = smooth_weights_u16 + height - 4; \
\
uint16x4x2_t top_vals[(W) >> 3]; \
for (int i = 0; i<(W)>> 3; ++i) { \
const int x = i << 3; \
top_vals[i].val[0] = vld1_u16(top_row + x); \
top_vals[i].val[1] = vld1_u16(top_row + x + 4); \
} \
\
const uint16x4_t bottom_left_v = vdup_n_u16(bottom_left); \
for (int y = 0; y < height; ++y) { \
const uint32x4_t weighted_bl = \
vmull_n_u16(bottom_left_v, 256 - weights_y[y]); \
\
uint16_t *dst_x = dst; \
for (int i = 0; i<(W)>> 3; ++i) { \
const uint32x4_t weighted_top_low = \
vmlal_n_u16(weighted_bl, top_vals[i].val[0], weights_y[y]); \
vst1_u16(dst_x, \
vrshrn_n_u32(weighted_top_low, SMOOTH_WEIGHT_LOG2_SCALE)); \
\
const uint32x4_t weighted_top_high = \
vmlal_n_u16(weighted_bl, top_vals[i].val[1], weights_y[y]); \
vst1_u16(dst_x + 4, \
vrshrn_n_u32(weighted_top_high, SMOOTH_WEIGHT_LOG2_SCALE)); \
dst_x += 8; \
} \
dst += stride; \
} \
}
HIGHBD_SMOOTH_V_PREDICTOR(16)
HIGHBD_SMOOTH_V_PREDICTOR(32)
HIGHBD_SMOOTH_V_PREDICTOR(64)
#undef HIGHBD_SMOOTH_V_PREDICTOR
#define HIGHBD_SMOOTH_V_NXM_WIDE(W, H) \
void aom_highbd_smooth_v_predictor_##W##x##H##_neon( \
uint16_t *dst, ptrdiff_t y_stride, const uint16_t *above, \
const uint16_t *left, int bd) { \
(void)bd; \
highbd_smooth_v_##W##xh_neon(dst, y_stride, above, left, H); \
}
HIGHBD_SMOOTH_V_NXM_WIDE(16, 4)
HIGHBD_SMOOTH_V_NXM_WIDE(16, 8)
HIGHBD_SMOOTH_V_NXM_WIDE(16, 16)
HIGHBD_SMOOTH_V_NXM_WIDE(16, 32)
HIGHBD_SMOOTH_V_NXM_WIDE(16, 64)
HIGHBD_SMOOTH_V_NXM_WIDE(32, 8)
HIGHBD_SMOOTH_V_NXM_WIDE(32, 16)
HIGHBD_SMOOTH_V_NXM_WIDE(32, 32)
HIGHBD_SMOOTH_V_NXM_WIDE(32, 64)
HIGHBD_SMOOTH_V_NXM_WIDE(64, 16)
HIGHBD_SMOOTH_V_NXM_WIDE(64, 32)
HIGHBD_SMOOTH_V_NXM_WIDE(64, 64)
#undef HIGHBD_SMOOTH_V_NXM_WIDE
static INLINE void highbd_smooth_h_4xh_neon(uint16_t *dst, ptrdiff_t stride,
const uint16_t *const top_row,
const uint16_t *const left_column,
const int height) {
const uint16_t top_right = top_row[3];
const uint16x4_t weights_x = vld1_u16(smooth_weights_u16);
const uint16x4_t scaled_weights_x = negate_s8(weights_x);
const uint32x4_t weighted_tr = vmull_n_u16(scaled_weights_x, top_right);
for (int y = 0; y < height; ++y) {
const uint32x4_t weighted_left =
vmlal_n_u16(weighted_tr, weights_x, left_column[y]);
vst1_u16(dst, vrshrn_n_u32(weighted_left, SMOOTH_WEIGHT_LOG2_SCALE));
dst += stride;
}
}
static INLINE void highbd_smooth_h_8xh_neon(uint16_t *dst, ptrdiff_t stride,
const uint16_t *const top_row,
const uint16_t *const left_column,
const int height) {
const uint16_t top_right = top_row[7];
const uint16x4x2_t weights_x = { { vld1_u16(smooth_weights_u16 + 4),
vld1_u16(smooth_weights_u16 + 8) } };
const uint32x4_t weighted_tr_low =
vmull_n_u16(negate_s8(weights_x.val[0]), top_right);
const uint32x4_t weighted_tr_high =
vmull_n_u16(negate_s8(weights_x.val[1]), top_right);
for (int y = 0; y < height; ++y) {
const uint16_t left_y = left_column[y];
const uint32x4_t weighted_left_low =
vmlal_n_u16(weighted_tr_low, weights_x.val[0], left_y);
vst1_u16(dst, vrshrn_n_u32(weighted_left_low, SMOOTH_WEIGHT_LOG2_SCALE));
const uint32x4_t weighted_left_high =
vmlal_n_u16(weighted_tr_high, weights_x.val[1], left_y);
vst1_u16(dst + 4,
vrshrn_n_u32(weighted_left_high, SMOOTH_WEIGHT_LOG2_SCALE));
dst += stride;
}
}
#define HIGHBD_SMOOTH_H_NXM(W, H) \
void aom_highbd_smooth_h_predictor_##W##x##H##_neon( \
uint16_t *dst, ptrdiff_t y_stride, const uint16_t *above, \
const uint16_t *left, int bd) { \
(void)bd; \
highbd_smooth_h_##W##xh_neon(dst, y_stride, above, left, H); \
}
HIGHBD_SMOOTH_H_NXM(4, 4)
HIGHBD_SMOOTH_H_NXM(4, 8)
HIGHBD_SMOOTH_H_NXM(4, 16)
HIGHBD_SMOOTH_H_NXM(8, 4)
HIGHBD_SMOOTH_H_NXM(8, 8)
HIGHBD_SMOOTH_H_NXM(8, 16)
HIGHBD_SMOOTH_H_NXM(8, 32)
#undef HIGHBD_SMOOTH_H_NXM
// For width 16 and above.
#define HIGHBD_SMOOTH_H_PREDICTOR(W) \
void highbd_smooth_h_##W##xh_neon( \
uint16_t *dst, ptrdiff_t stride, const uint16_t *const top_row, \
const uint16_t *const left_column, const int height) { \
const uint16_t top_right = top_row[(W)-1]; \
\
uint16x4_t weights_x_low[(W) >> 3]; \
uint16x4_t weights_x_high[(W) >> 3]; \
uint32x4_t weighted_tr_low[(W) >> 3]; \
uint32x4_t weighted_tr_high[(W) >> 3]; \
for (int i = 0; i<(W)>> 3; ++i) { \
const int x = i << 3; \
weights_x_low[i] = vld1_u16(smooth_weights_u16 + (W)-4 + x); \
weighted_tr_low[i] = \
vmull_n_u16(negate_s8(weights_x_low[i]), top_right); \
weights_x_high[i] = vld1_u16(smooth_weights_u16 + (W) + x); \
weighted_tr_high[i] = \
vmull_n_u16(negate_s8(weights_x_high[i]), top_right); \
} \
\
for (int y = 0; y < height; ++y) { \
uint16_t *dst_x = dst; \
const uint16_t left_y = left_column[y]; \
for (int i = 0; i<(W)>> 3; ++i) { \
const uint32x4_t weighted_left_low = \
vmlal_n_u16(weighted_tr_low[i], weights_x_low[i], left_y); \
vst1_u16(dst_x, \
vrshrn_n_u32(weighted_left_low, SMOOTH_WEIGHT_LOG2_SCALE)); \
\
const uint32x4_t weighted_left_high = \
vmlal_n_u16(weighted_tr_high[i], weights_x_high[i], left_y); \
vst1_u16(dst_x + 4, \
vrshrn_n_u32(weighted_left_high, SMOOTH_WEIGHT_LOG2_SCALE)); \
dst_x += 8; \
} \
dst += stride; \
} \
}
HIGHBD_SMOOTH_H_PREDICTOR(16)
HIGHBD_SMOOTH_H_PREDICTOR(32)
HIGHBD_SMOOTH_H_PREDICTOR(64)
#undef HIGHBD_SMOOTH_H_PREDICTOR
#define HIGHBD_SMOOTH_H_NXM_WIDE(W, H) \
void aom_highbd_smooth_h_predictor_##W##x##H##_neon( \
uint16_t *dst, ptrdiff_t y_stride, const uint16_t *above, \
const uint16_t *left, int bd) { \
(void)bd; \
highbd_smooth_h_##W##xh_neon(dst, y_stride, above, left, H); \
}
HIGHBD_SMOOTH_H_NXM_WIDE(16, 4)
HIGHBD_SMOOTH_H_NXM_WIDE(16, 8)
HIGHBD_SMOOTH_H_NXM_WIDE(16, 16)
HIGHBD_SMOOTH_H_NXM_WIDE(16, 32)
HIGHBD_SMOOTH_H_NXM_WIDE(16, 64)
HIGHBD_SMOOTH_H_NXM_WIDE(32, 8)
HIGHBD_SMOOTH_H_NXM_WIDE(32, 16)
HIGHBD_SMOOTH_H_NXM_WIDE(32, 32)
HIGHBD_SMOOTH_H_NXM_WIDE(32, 64)
HIGHBD_SMOOTH_H_NXM_WIDE(64, 16)
HIGHBD_SMOOTH_H_NXM_WIDE(64, 32)
HIGHBD_SMOOTH_H_NXM_WIDE(64, 64)
#undef HIGHBD_SMOOTH_H_NXM_WIDE

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,232 @@
/*
* Copyright (c) 2022, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include <arm_neon.h>
#include <assert.h>
#include "aom_dsp/arm/mem_neon.h"
#include "av1/common/quant_common.h"
#include "av1/encoder/av1_quantize.h"
static INLINE uint32_t sum_abs_coeff(const uint32x4_t a) {
#if defined(__aarch64__)
return vaddvq_u32(a);
#else
const uint64x2_t b = vpaddlq_u32(a);
const uint64x1_t c = vadd_u64(vget_low_u64(b), vget_high_u64(b));
return (uint32_t)vget_lane_u64(c, 0);
#endif
}
static INLINE uint16x4_t
quantize_4(const tran_low_t *coeff_ptr, tran_low_t *qcoeff_ptr,
tran_low_t *dqcoeff_ptr, int32x4_t v_quant_s32,
int32x4_t v_dequant_s32, int32x4_t v_round_s32, int32x4_t v_zbin_s32,
int32x4_t v_quant_shift_s32, int log_scale) {
const int32x4_t v_coeff = vld1q_s32(coeff_ptr);
const int32x4_t v_coeff_sign =
vreinterpretq_s32_u32(vcltq_s32(v_coeff, vdupq_n_s32(0)));
const int32x4_t v_abs_coeff = vabsq_s32(v_coeff);
// if (abs_coeff < zbins[rc != 0]),
const uint32x4_t v_zbin_mask = vcgeq_s32(v_abs_coeff, v_zbin_s32);
const int32x4_t v_log_scale = vdupq_n_s32(log_scale);
// const int64_t tmp = (int64_t)abs_coeff + log_scaled_round;
const int32x4_t v_tmp = vaddq_s32(v_abs_coeff, v_round_s32);
// const int32_t tmpw32 = tmp * wt;
const int32x4_t v_tmpw32 = vmulq_s32(v_tmp, vdupq_n_s32((1 << AOM_QM_BITS)));
// const int32_t tmp2 = (int32_t)((tmpw32 * quant64) >> 16);
const int32x4_t v_tmp2 = vqdmulhq_s32(v_tmpw32, v_quant_s32);
// const int32_t tmp3 =
// ((((tmp2 + tmpw32)<< log_scale) * (int64_t)(quant_shift << 15)) >> 32);
const int32x4_t v_tmp3 = vqdmulhq_s32(
vshlq_s32(vaddq_s32(v_tmp2, v_tmpw32), v_log_scale), v_quant_shift_s32);
// const int abs_qcoeff = vmask ? (int)tmp3 >> AOM_QM_BITS : 0;
const int32x4_t v_abs_qcoeff = vandq_s32(vreinterpretq_s32_u32(v_zbin_mask),
vshrq_n_s32(v_tmp3, AOM_QM_BITS));
// const tran_low_t abs_dqcoeff = (abs_qcoeff * dequant_iwt) >> log_scale;
// vshlq_s32 will shift right if shift value is negative.
const int32x4_t v_abs_dqcoeff =
vshlq_s32(vmulq_s32(v_abs_qcoeff, v_dequant_s32), vnegq_s32(v_log_scale));
// qcoeff_ptr[rc] = (tran_low_t)((abs_qcoeff ^ coeff_sign) - coeff_sign);
const int32x4_t v_qcoeff =
vsubq_s32(veorq_s32(v_abs_qcoeff, v_coeff_sign), v_coeff_sign);
// dqcoeff_ptr[rc] = (tran_low_t)((abs_dqcoeff ^ coeff_sign) - coeff_sign);
const int32x4_t v_dqcoeff =
vsubq_s32(veorq_s32(v_abs_dqcoeff, v_coeff_sign), v_coeff_sign);
vst1q_s32(qcoeff_ptr, v_qcoeff);
vst1q_s32(dqcoeff_ptr, v_dqcoeff);
// Used to find eob.
const uint32x4_t nz_qcoeff_mask = vcgtq_s32(v_abs_qcoeff, vdupq_n_s32(0));
return vmovn_u32(nz_qcoeff_mask);
}
static INLINE int16x8_t get_max_lane_eob(const int16_t *iscan,
int16x8_t v_eobmax,
uint16x8_t v_mask) {
const int16x8_t v_iscan = vld1q_s16(&iscan[0]);
const int16x8_t v_iscan_plus1 = vaddq_s16(v_iscan, vdupq_n_s16(1));
const int16x8_t v_nz_iscan = vbslq_s16(v_mask, v_iscan_plus1, vdupq_n_s16(0));
return vmaxq_s16(v_eobmax, v_nz_iscan);
}
static INLINE uint16_t get_max_eob(int16x8_t v_eobmax) {
#ifdef __aarch64__
return (uint16_t)vmaxvq_s16(v_eobmax);
#else
const int16x4_t v_eobmax_3210 =
vmax_s16(vget_low_s16(v_eobmax), vget_high_s16(v_eobmax));
const int64x1_t v_eobmax_xx32 =
vshr_n_s64(vreinterpret_s64_s16(v_eobmax_3210), 32);
const int16x4_t v_eobmax_tmp =
vmax_s16(v_eobmax_3210, vreinterpret_s16_s64(v_eobmax_xx32));
const int64x1_t v_eobmax_xxx3 =
vshr_n_s64(vreinterpret_s64_s16(v_eobmax_tmp), 16);
const int16x4_t v_eobmax_final =
vmax_s16(v_eobmax_tmp, vreinterpret_s16_s64(v_eobmax_xxx3));
return (uint16_t)vget_lane_s16(v_eobmax_final, 0);
#endif
}
static void highbd_quantize_b_neon(
const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr,
const int16_t *round_ptr, const int16_t *quant_ptr,
const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
const int16_t *scan, const int16_t *iscan, const int log_scale) {
(void)scan;
const int16x4_t v_quant = vld1_s16(quant_ptr);
const int16x4_t v_dequant = vld1_s16(dequant_ptr);
const int16x4_t v_zero = vdup_n_s16(0);
const uint16x4_t v_round_select = vcgt_s16(vdup_n_s16(log_scale), v_zero);
const int16x4_t v_round_no_scale = vld1_s16(round_ptr);
const int16x4_t v_round_log_scale =
vqrdmulh_n_s16(v_round_no_scale, (int16_t)(1 << (15 - log_scale)));
const int16x4_t v_round =
vbsl_s16(v_round_select, v_round_log_scale, v_round_no_scale);
const int16x4_t v_quant_shift = vld1_s16(quant_shift_ptr);
const int16x4_t v_zbin_no_scale = vld1_s16(zbin_ptr);
const int16x4_t v_zbin_log_scale =
vqrdmulh_n_s16(v_zbin_no_scale, (int16_t)(1 << (15 - log_scale)));
const int16x4_t v_zbin =
vbsl_s16(v_round_select, v_zbin_log_scale, v_zbin_no_scale);
int32x4_t v_round_s32 = vmovl_s16(v_round);
int32x4_t v_quant_s32 = vshlq_n_s32(vmovl_s16(v_quant), 15);
int32x4_t v_dequant_s32 = vmovl_s16(v_dequant);
int32x4_t v_quant_shift_s32 = vshlq_n_s32(vmovl_s16(v_quant_shift), 15);
int32x4_t v_zbin_s32 = vmovl_s16(v_zbin);
uint16x4_t v_mask_lo, v_mask_hi;
int16x8_t v_eobmax = vdupq_n_s16(-1);
intptr_t non_zero_count = n_coeffs;
assert(n_coeffs > 8);
// Pre-scan pass
const int32x4_t v_zbin_s32x = vdupq_lane_s32(vget_low_s32(v_zbin_s32), 1);
intptr_t i = n_coeffs;
do {
const int32x4_t v_coeff_a = vld1q_s32(coeff_ptr + i - 4);
const int32x4_t v_coeff_b = vld1q_s32(coeff_ptr + i - 8);
const int32x4_t v_abs_coeff_a = vabsq_s32(v_coeff_a);
const int32x4_t v_abs_coeff_b = vabsq_s32(v_coeff_b);
const uint32x4_t v_mask_a = vcgeq_s32(v_abs_coeff_a, v_zbin_s32x);
const uint32x4_t v_mask_b = vcgeq_s32(v_abs_coeff_b, v_zbin_s32x);
// If the coefficient is in the base ZBIN range, then discard.
if (sum_abs_coeff(v_mask_a) + sum_abs_coeff(v_mask_b) == 0) {
non_zero_count -= 8;
} else {
break;
}
i -= 8;
} while (i > 0);
const intptr_t remaining_zcoeffs = n_coeffs - non_zero_count;
memset(qcoeff_ptr + non_zero_count, 0,
remaining_zcoeffs * sizeof(*qcoeff_ptr));
memset(dqcoeff_ptr + non_zero_count, 0,
remaining_zcoeffs * sizeof(*dqcoeff_ptr));
// DC and first 3 AC
v_mask_lo =
quantize_4(coeff_ptr, qcoeff_ptr, dqcoeff_ptr, v_quant_s32, v_dequant_s32,
v_round_s32, v_zbin_s32, v_quant_shift_s32, log_scale);
// overwrite the DC constants with AC constants
v_round_s32 = vdupq_lane_s32(vget_low_s32(v_round_s32), 1);
v_quant_s32 = vdupq_lane_s32(vget_low_s32(v_quant_s32), 1);
v_dequant_s32 = vdupq_lane_s32(vget_low_s32(v_dequant_s32), 1);
v_quant_shift_s32 = vdupq_lane_s32(vget_low_s32(v_quant_shift_s32), 1);
v_zbin_s32 = vdupq_lane_s32(vget_low_s32(v_zbin_s32), 1);
// 4 more AC
v_mask_hi = quantize_4(coeff_ptr + 4, qcoeff_ptr + 4, dqcoeff_ptr + 4,
v_quant_s32, v_dequant_s32, v_round_s32, v_zbin_s32,
v_quant_shift_s32, log_scale);
v_eobmax =
get_max_lane_eob(iscan, v_eobmax, vcombine_u16(v_mask_lo, v_mask_hi));
intptr_t count = non_zero_count - 8;
for (; count > 0; count -= 8) {
coeff_ptr += 8;
qcoeff_ptr += 8;
dqcoeff_ptr += 8;
iscan += 8;
v_mask_lo = quantize_4(coeff_ptr, qcoeff_ptr, dqcoeff_ptr, v_quant_s32,
v_dequant_s32, v_round_s32, v_zbin_s32,
v_quant_shift_s32, log_scale);
v_mask_hi = quantize_4(coeff_ptr + 4, qcoeff_ptr + 4, dqcoeff_ptr + 4,
v_quant_s32, v_dequant_s32, v_round_s32, v_zbin_s32,
v_quant_shift_s32, log_scale);
// Find the max lane eob for 8 coeffs.
v_eobmax =
get_max_lane_eob(iscan, v_eobmax, vcombine_u16(v_mask_lo, v_mask_hi));
}
*eob_ptr = get_max_eob(v_eobmax);
}
void aom_highbd_quantize_b_neon(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
const int16_t *zbin_ptr,
const int16_t *round_ptr,
const int16_t *quant_ptr,
const int16_t *quant_shift_ptr,
tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr,
const int16_t *dequant_ptr, uint16_t *eob_ptr,
const int16_t *scan, const int16_t *iscan) {
highbd_quantize_b_neon(coeff_ptr, n_coeffs, zbin_ptr, round_ptr, quant_ptr,
quant_shift_ptr, qcoeff_ptr, dqcoeff_ptr, dequant_ptr,
eob_ptr, scan, iscan, 0);
}
void aom_highbd_quantize_b_32x32_neon(
const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr,
const int16_t *round_ptr, const int16_t *quant_ptr,
const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
const int16_t *scan, const int16_t *iscan) {
highbd_quantize_b_neon(coeff_ptr, n_coeffs, zbin_ptr, round_ptr, quant_ptr,
quant_shift_ptr, qcoeff_ptr, dqcoeff_ptr, dequant_ptr,
eob_ptr, scan, iscan, 1);
}
void aom_highbd_quantize_b_64x64_neon(
const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr,
const int16_t *round_ptr, const int16_t *quant_ptr,
const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
const int16_t *scan, const int16_t *iscan) {
highbd_quantize_b_neon(coeff_ptr, n_coeffs, zbin_ptr, round_ptr, quant_ptr,
quant_shift_ptr, qcoeff_ptr, dqcoeff_ptr, dequant_ptr,
eob_ptr, scan, iscan, 2);
}

View file

@ -0,0 +1,171 @@
/*
* Copyright (c) 2022, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include <arm_neon.h>
#include "config/aom_config.h"
#include "config/aom_dsp_rtcd.h"
#include "aom_dsp/variance.h"
#include "aom_dsp/aom_filter.h"
#include "aom_dsp/arm/sum_neon.h"
typedef void (*high_variance_fn_t)(const uint16_t *src, int src_stride,
const uint16_t *ref, int ref_stride,
uint32_t *sse, int *sum);
void aom_highbd_calc16x16var_neon(const uint16_t *src, int src_stride,
const uint16_t *ref, int ref_stride,
uint32_t *sse, int *sum) {
int i, j;
int16x8_t v_sum = vdupq_n_s16(0);
int32x4_t v_sse_lo = vdupq_n_s32(0);
int32x4_t v_sse_hi = vdupq_n_s32(0);
for (i = 0; i < 16; ++i) {
for (j = 0; j < 16; j += 8) {
const uint16x8_t v_a = vld1q_u16(&src[j]);
const uint16x8_t v_b = vld1q_u16(&ref[j]);
const int16x8_t sv_diff = vreinterpretq_s16_u16(vsubq_u16(v_a, v_b));
v_sum = vaddq_s16(v_sum, sv_diff);
v_sse_lo =
vmlal_s16(v_sse_lo, vget_low_s16(sv_diff), vget_low_s16(sv_diff));
v_sse_hi =
vmlal_s16(v_sse_hi, vget_high_s16(sv_diff), vget_high_s16(sv_diff));
}
src += src_stride;
ref += ref_stride;
}
*sum = horizontal_add_s16x8(v_sum);
*sse = (unsigned int)horizontal_add_s32x4(vaddq_s32(v_sse_lo, v_sse_hi));
}
void aom_highbd_calc8x8var_neon(const uint16_t *src, int src_stride,
const uint16_t *ref, int ref_stride,
uint32_t *sse, int *sum) {
int i;
int16x8_t v_sum = vdupq_n_s16(0);
int32x4_t v_sse_lo = vdupq_n_s32(0);
int32x4_t v_sse_hi = vdupq_n_s32(0);
for (i = 0; i < 8; ++i) {
const uint16x8_t v_a = vld1q_u16(&src[0]);
const uint16x8_t v_b = vld1q_u16(&ref[0]);
const int16x8_t sv_diff = vreinterpretq_s16_u16(vsubq_u16(v_a, v_b));
v_sum = vaddq_s16(v_sum, sv_diff);
v_sse_lo =
vmlal_s16(v_sse_lo, vget_low_s16(sv_diff), vget_low_s16(sv_diff));
v_sse_hi =
vmlal_s16(v_sse_hi, vget_high_s16(sv_diff), vget_high_s16(sv_diff));
src += src_stride;
ref += ref_stride;
}
*sum = horizontal_add_s16x8(v_sum);
*sse = (unsigned int)horizontal_add_s32x4(vaddq_s32(v_sse_lo, v_sse_hi));
}
void aom_highbd_calc4x4var_neon(const uint16_t *src, int src_stride,
const uint16_t *ref, int ref_stride,
uint32_t *sse, int *sum) {
int i;
int16x8_t v_sum = vdupq_n_s16(0);
int32x4_t v_sse_lo = vdupq_n_s32(0);
int32x4_t v_sse_hi = vdupq_n_s32(0);
for (i = 0; i < 4; i += 2) {
const uint16x4_t v_a_r0 = vld1_u16(&src[0]);
const uint16x4_t v_b_r0 = vld1_u16(&ref[0]);
const uint16x4_t v_a_r1 = vld1_u16(&src[src_stride]);
const uint16x4_t v_b_r1 = vld1_u16(&ref[ref_stride]);
const uint16x8_t v_a = vcombine_u16(v_a_r0, v_a_r1);
const uint16x8_t v_b = vcombine_u16(v_b_r0, v_b_r1);
const int16x8_t sv_diff = vreinterpretq_s16_u16(vsubq_u16(v_a, v_b));
v_sum = vaddq_s16(v_sum, sv_diff);
v_sse_lo =
vmlal_s16(v_sse_lo, vget_low_s16(sv_diff), vget_low_s16(sv_diff));
v_sse_hi =
vmlal_s16(v_sse_hi, vget_high_s16(sv_diff), vget_high_s16(sv_diff));
src += src_stride << 1;
ref += ref_stride << 1;
}
*sum = horizontal_add_s16x8(v_sum);
*sse = (unsigned int)horizontal_add_s32x4(vaddq_s32(v_sse_lo, v_sse_hi));
}
static void highbd_10_variance_neon(const uint16_t *src, int src_stride,
const uint16_t *ref, int ref_stride, int w,
int h, uint32_t *sse, int *sum,
high_variance_fn_t var_fn, int block_size) {
int i, j;
uint64_t sse_long = 0;
int32_t sum_long = 0;
for (i = 0; i < h; i += block_size) {
for (j = 0; j < w; j += block_size) {
unsigned int sse0;
int sum0;
var_fn(src + src_stride * i + j, src_stride, ref + ref_stride * i + j,
ref_stride, &sse0, &sum0);
sse_long += sse0;
sum_long += sum0;
}
}
*sum = ROUND_POWER_OF_TWO(sum_long, 2);
*sse = (uint32_t)ROUND_POWER_OF_TWO(sse_long, 4);
}
#define VAR_FN(w, h, block_size, shift) \
uint32_t aom_highbd_10_variance##w##x##h##_neon( \
const uint8_t *src8, int src_stride, const uint8_t *ref8, \
int ref_stride, uint32_t *sse) { \
int sum; \
int64_t var; \
uint16_t *src = CONVERT_TO_SHORTPTR(src8); \
uint16_t *ref = CONVERT_TO_SHORTPTR(ref8); \
highbd_10_variance_neon( \
src, src_stride, ref, ref_stride, w, h, sse, &sum, \
aom_highbd_calc##block_size##x##block_size##var_neon, block_size); \
var = (int64_t)(*sse) - (((int64_t)sum * sum) >> shift); \
return (var >= 0) ? (uint32_t)var : 0; \
}
VAR_FN(128, 128, 16, 14)
VAR_FN(128, 64, 16, 13)
VAR_FN(64, 128, 16, 13)
VAR_FN(64, 64, 16, 12)
VAR_FN(64, 32, 16, 11)
VAR_FN(32, 64, 16, 11)
VAR_FN(32, 32, 16, 10)
VAR_FN(32, 16, 16, 9)
VAR_FN(16, 32, 16, 9)
VAR_FN(16, 16, 16, 8)
VAR_FN(16, 8, 8, 7)
VAR_FN(8, 16, 8, 7)
VAR_FN(8, 8, 8, 6)
VAR_FN(16, 4, 4, 6)
VAR_FN(4, 16, 4, 6)
VAR_FN(8, 4, 4, 5)
VAR_FN(4, 8, 4, 5)
VAR_FN(4, 4, 4, 4)
#if !CONFIG_REALTIME_ONLY
VAR_FN(64, 16, 16, 10)
VAR_FN(16, 64, 16, 10)
VAR_FN(8, 32, 8, 8)
VAR_FN(32, 8, 8, 8)
#endif // !CONFIG_REALTIME_ONLY
#undef VAR_FN

File diff suppressed because it is too large Load diff

View file

@ -15,8 +15,8 @@
#include "config/aom_config.h"
#include "aom/aom_integer.h"
#include "av1/common/arm/mem_neon.h"
#include "av1/common/arm/transpose_neon.h"
#include "aom_dsp/arm/mem_neon.h"
#include "aom_dsp/arm/transpose_neon.h"
static INLINE uint8x8_t lpf_mask(uint8x8_t p3q3, uint8x8_t p2q2, uint8x8_t p1q1,
uint8x8_t p0q0, const uint8_t blimit,
@ -695,6 +695,23 @@ void aom_lpf_vertical_14_neon(uint8_t *src, int stride, const uint8_t *blimit,
store_u8_8x16(src - 8, stride, row0, row1, row2, row3);
}
void aom_lpf_vertical_14_dual_neon(
uint8_t *s, int pitch, const uint8_t *blimit0, const uint8_t *limit0,
const uint8_t *thresh0, const uint8_t *blimit1, const uint8_t *limit1,
const uint8_t *thresh1) {
aom_lpf_vertical_14_neon(s, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_14_neon(s + 4 * pitch, pitch, blimit1, limit1, thresh1);
}
void aom_lpf_vertical_14_quad_neon(uint8_t *s, int pitch, const uint8_t *blimit,
const uint8_t *limit,
const uint8_t *thresh) {
aom_lpf_vertical_14_dual_neon(s, pitch, blimit, limit, thresh, blimit, limit,
thresh);
aom_lpf_vertical_14_dual_neon(s + 2 * MI_SIZE * pitch, pitch, blimit, limit,
thresh, blimit, limit, thresh);
}
void aom_lpf_vertical_8_neon(uint8_t *src, int stride, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
uint32x2x2_t p2q2_p1q1, p3q3_p0q0;
@ -738,6 +755,22 @@ void aom_lpf_vertical_8_neon(uint8_t *src, int stride, const uint8_t *blimit,
store_u8_8x4(src - 4, stride, p3q0, p2q1, p1q2, p0q3);
}
void aom_lpf_vertical_8_dual_neon(uint8_t *s, int pitch, const uint8_t *blimit0,
const uint8_t *limit0, const uint8_t *thresh0,
const uint8_t *blimit1, const uint8_t *limit1,
const uint8_t *thresh1) {
aom_lpf_vertical_8_neon(s, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_8_neon(s + 4 * pitch, pitch, blimit1, limit1, thresh1);
}
void aom_lpf_vertical_8_quad_neon(uint8_t *s, int pitch, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
aom_lpf_vertical_8_dual_neon(s, pitch, blimit, limit, thresh, blimit, limit,
thresh);
aom_lpf_vertical_8_dual_neon(s + 2 * MI_SIZE * pitch, pitch, blimit, limit,
thresh, blimit, limit, thresh);
}
void aom_lpf_vertical_6_neon(uint8_t *src, int stride, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
uint32x2x2_t p2q2_p1q1, pxqy_p0q0;
@ -781,6 +814,22 @@ void aom_lpf_vertical_6_neon(uint8_t *src, int stride, const uint8_t *blimit,
store_u8_8x4(src - 4, stride, pxq0, p2q1, p1q2, p0qy);
}
void aom_lpf_vertical_6_dual_neon(uint8_t *s, int pitch, const uint8_t *blimit0,
const uint8_t *limit0, const uint8_t *thresh0,
const uint8_t *blimit1, const uint8_t *limit1,
const uint8_t *thresh1) {
aom_lpf_vertical_6_neon(s, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_6_neon(s + 4 * pitch, pitch, blimit1, limit1, thresh1);
}
void aom_lpf_vertical_6_quad_neon(uint8_t *s, int pitch, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
aom_lpf_vertical_6_dual_neon(s, pitch, blimit, limit, thresh, blimit, limit,
thresh);
aom_lpf_vertical_6_dual_neon(s + 2 * MI_SIZE * pitch, pitch, blimit, limit,
thresh, blimit, limit, thresh);
}
void aom_lpf_vertical_4_neon(uint8_t *src, int stride, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
uint32x2x2_t p1q0_p0q1, p1q1_p0q0, p1p0_q1q0;
@ -820,9 +869,28 @@ void aom_lpf_vertical_4_neon(uint8_t *src, int stride, const uint8_t *blimit,
store_unaligned_u8_4x1((src - 2) + 3 * stride, q0q1, 1);
}
void aom_lpf_vertical_4_dual_neon(uint8_t *s, int pitch, const uint8_t *blimit0,
const uint8_t *limit0, const uint8_t *thresh0,
const uint8_t *blimit1, const uint8_t *limit1,
const uint8_t *thresh1) {
aom_lpf_vertical_4_neon(s, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_4_neon(s + 4 * pitch, pitch, blimit1, limit1, thresh1);
}
void aom_lpf_vertical_4_quad_neon(uint8_t *s, int pitch, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
aom_lpf_vertical_4_dual_neon(s, pitch, blimit, limit, thresh, blimit, limit,
thresh);
aom_lpf_vertical_4_dual_neon(s + 2 * MI_SIZE * pitch, pitch, blimit, limit,
thresh, blimit, limit, thresh);
}
void aom_lpf_horizontal_14_neon(uint8_t *src, int stride, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
uint8x8_t p0q0, p1q1, p2q2, p3q3, p4q4, p5q5, UNINITIALIZED_IS_SAFE(p6q6);
uint8x8_t UNINITIALIZED_IS_SAFE(p0q0), UNINITIALIZED_IS_SAFE(p1q1),
UNINITIALIZED_IS_SAFE(p2q2), UNINITIALIZED_IS_SAFE(p3q3),
UNINITIALIZED_IS_SAFE(p4q4), UNINITIALIZED_IS_SAFE(p5q5),
UNINITIALIZED_IS_SAFE(p6q6);
load_u8_4x1(src - 7 * stride, &p6q6, 0);
load_u8_4x1(src - 6 * stride, &p5q5, 0);
@ -856,6 +924,26 @@ void aom_lpf_horizontal_14_neon(uint8_t *src, int stride, const uint8_t *blimit,
store_u8_4x1(src + 5 * stride, p5q5, 1);
}
void aom_lpf_horizontal_14_dual_neon(
uint8_t *s, int pitch, const uint8_t *blimit0, const uint8_t *limit0,
const uint8_t *thresh0, const uint8_t *blimit1, const uint8_t *limit1,
const uint8_t *thresh1) {
aom_lpf_horizontal_14_neon(s, pitch, blimit0, limit0, thresh0);
aom_lpf_horizontal_14_neon(s + 4, pitch, blimit1, limit1, thresh1);
}
// TODO(any): Rewrite in NEON (similar to quad SSE2 functions) for better speed
// up.
void aom_lpf_horizontal_14_quad_neon(uint8_t *s, int pitch,
const uint8_t *blimit,
const uint8_t *limit,
const uint8_t *thresh) {
aom_lpf_horizontal_14_dual_neon(s, pitch, blimit, limit, thresh, blimit,
limit, thresh);
aom_lpf_horizontal_14_dual_neon(s + 2 * MI_SIZE, pitch, blimit, limit, thresh,
blimit, limit, thresh);
}
void aom_lpf_horizontal_8_neon(uint8_t *src, int stride, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
uint8x8_t p0q0, p1q1, p2q2, p3q3;
@ -885,6 +973,25 @@ void aom_lpf_horizontal_8_neon(uint8_t *src, int stride, const uint8_t *blimit,
vst1_lane_u32((uint32_t *)(src + 3 * stride), vreinterpret_u32_u8(p3q3), 1);
}
void aom_lpf_horizontal_8_dual_neon(
uint8_t *s, int pitch, const uint8_t *blimit0, const uint8_t *limit0,
const uint8_t *thresh0, const uint8_t *blimit1, const uint8_t *limit1,
const uint8_t *thresh1) {
aom_lpf_horizontal_8_neon(s, pitch, blimit0, limit0, thresh0);
aom_lpf_horizontal_8_neon(s + 4, pitch, blimit1, limit1, thresh1);
}
// TODO(any): Rewrite in NEON (similar to quad SSE2 functions) for better speed
// up.
void aom_lpf_horizontal_8_quad_neon(uint8_t *s, int pitch,
const uint8_t *blimit, const uint8_t *limit,
const uint8_t *thresh) {
aom_lpf_horizontal_8_dual_neon(s, pitch, blimit, limit, thresh, blimit, limit,
thresh);
aom_lpf_horizontal_8_dual_neon(s + 2 * MI_SIZE, pitch, blimit, limit, thresh,
blimit, limit, thresh);
}
void aom_lpf_horizontal_6_neon(uint8_t *src, int stride, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
uint8x8_t p0q0, p1q1, p2q2;
@ -909,6 +1016,25 @@ void aom_lpf_horizontal_6_neon(uint8_t *src, int stride, const uint8_t *blimit,
vst1_lane_u32((uint32_t *)(src + 2 * stride), vreinterpret_u32_u8(p2q2), 1);
}
void aom_lpf_horizontal_6_dual_neon(
uint8_t *s, int pitch, const uint8_t *blimit0, const uint8_t *limit0,
const uint8_t *thresh0, const uint8_t *blimit1, const uint8_t *limit1,
const uint8_t *thresh1) {
aom_lpf_horizontal_6_neon(s, pitch, blimit0, limit0, thresh0);
aom_lpf_horizontal_6_neon(s + 4, pitch, blimit1, limit1, thresh1);
}
// TODO(any): Rewrite in NEON (similar to quad SSE2 functions) for better speed
// up.
void aom_lpf_horizontal_6_quad_neon(uint8_t *s, int pitch,
const uint8_t *blimit, const uint8_t *limit,
const uint8_t *thresh) {
aom_lpf_horizontal_6_dual_neon(s, pitch, blimit, limit, thresh, blimit, limit,
thresh);
aom_lpf_horizontal_6_dual_neon(s + 2 * MI_SIZE, pitch, blimit, limit, thresh,
blimit, limit, thresh);
}
void aom_lpf_horizontal_4_neon(uint8_t *src, int stride, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
uint8x8_t p0q0, UNINITIALIZED_IS_SAFE(p1q1);
@ -925,3 +1051,22 @@ void aom_lpf_horizontal_4_neon(uint8_t *src, int stride, const uint8_t *blimit,
store_u8_4x1(src + 0 * stride, p0q0, 1);
store_u8_4x1(src + 1 * stride, p1q1, 1);
}
void aom_lpf_horizontal_4_dual_neon(
uint8_t *s, int pitch, const uint8_t *blimit0, const uint8_t *limit0,
const uint8_t *thresh0, const uint8_t *blimit1, const uint8_t *limit1,
const uint8_t *thresh1) {
aom_lpf_horizontal_4_neon(s, pitch, blimit0, limit0, thresh0);
aom_lpf_horizontal_4_neon(s + 4, pitch, blimit1, limit1, thresh1);
}
// TODO(any): Rewrite in NEON (similar to quad SSE2 functions) for better speed
// up.
void aom_lpf_horizontal_4_quad_neon(uint8_t *s, int pitch,
const uint8_t *blimit, const uint8_t *limit,
const uint8_t *thresh) {
aom_lpf_horizontal_4_dual_neon(s, pitch, blimit, limit, thresh, blimit, limit,
thresh);
aom_lpf_horizontal_4_dual_neon(s + 2 * MI_SIZE, pitch, blimit, limit, thresh,
blimit, limit, thresh);
}

View file

@ -8,8 +8,8 @@
* be found in the AUTHORS file in the root of the source tree.
*/
#ifndef AOM_AV1_COMMON_ARM_MEM_NEON_H_
#define AOM_AV1_COMMON_ARM_MEM_NEON_H_
#ifndef AOM_AOM_DSP_ARM_MEM_NEON_H_
#define AOM_AOM_DSP_ARM_MEM_NEON_H_
#include <arm_neon.h>
#include <string.h>
@ -536,4 +536,4 @@ static INLINE void store_s16q_to_tran_low(tran_low_t *buf, const int16x8_t a) {
vst1q_s32(buf + 4, v1);
}
#endif // AOM_AV1_COMMON_ARM_MEM_NEON_H_
#endif // AOM_AOM_DSP_ARM_MEM_NEON_H_

View file

@ -82,7 +82,7 @@ static void sad_neon_32(const uint8x16_t vec_src_00,
void aom_sad64x64x4d_neon(const uint8_t *src, int src_stride,
const uint8_t *const ref[4], int ref_stride,
uint32_t *res) {
uint32_t res[4]) {
int i;
uint16x8_t vec_sum_ref0_lo = vdupq_n_u16(0);
uint16x8_t vec_sum_ref0_hi = vdupq_n_u16(0);
@ -128,7 +128,7 @@ void aom_sad64x64x4d_neon(const uint8_t *src, int src_stride,
void aom_sad32x32x4d_neon(const uint8_t *src, int src_stride,
const uint8_t *const ref[4], int ref_stride,
uint32_t *res) {
uint32_t res[4]) {
int i;
uint16x8_t vec_sum_ref0_lo = vdupq_n_u16(0);
uint16x8_t vec_sum_ref0_hi = vdupq_n_u16(0);
@ -172,7 +172,7 @@ void aom_sad32x32x4d_neon(const uint8_t *src, int src_stride,
void aom_sad16x16x4d_neon(const uint8_t *src, int src_stride,
const uint8_t *const ref[4], int ref_stride,
uint32_t *res) {
uint32_t res[4]) {
int i;
uint16x8_t vec_sum_ref0_lo = vdupq_n_u16(0);
uint16x8_t vec_sum_ref0_hi = vdupq_n_u16(0);
@ -224,3 +224,369 @@ void aom_sad16x16x4d_neon(const uint8_t *src, int src_stride,
res[2] = horizontal_long_add_16x8(vec_sum_ref2_lo, vec_sum_ref2_hi);
res[3] = horizontal_long_add_16x8(vec_sum_ref3_lo, vec_sum_ref3_hi);
}
static INLINE unsigned int horizontal_add_16x4(const uint16x4_t vec_16x4) {
const uint32x2_t a = vpaddl_u16(vec_16x4);
const uint64x1_t b = vpaddl_u32(a);
return vget_lane_u32(vreinterpret_u32_u64(b), 0);
}
static INLINE unsigned int horizontal_add_16x8(const uint16x8_t vec_16x8) {
const uint32x4_t a = vpaddlq_u16(vec_16x8);
const uint64x2_t b = vpaddlq_u32(a);
const uint32x2_t c = vadd_u32(vreinterpret_u32_u64(vget_low_u64(b)),
vreinterpret_u32_u64(vget_high_u64(b)));
return vget_lane_u32(c, 0);
}
static void sad_row4_neon(uint16x4_t *vec_src, const uint8x8_t q0,
const uint8x8_t ref) {
uint8x8_t q2 = vabd_u8(q0, ref);
*vec_src = vpadal_u8(*vec_src, q2);
}
static void sad_row8_neon(uint16x4_t *vec_src, const uint8x8_t *q0,
const uint8_t *ref_ptr) {
uint8x8_t q1 = vld1_u8(ref_ptr);
uint8x8_t q2 = vabd_u8(*q0, q1);
*vec_src = vpadal_u8(*vec_src, q2);
}
static void sad_row16_neon(uint16x8_t *vec_src, const uint8x16_t *q0,
const uint8_t *ref_ptr) {
uint8x16_t q1 = vld1q_u8(ref_ptr);
uint8x16_t q2 = vabdq_u8(*q0, q1);
*vec_src = vpadalq_u8(*vec_src, q2);
}
void aom_sadMxNx4d_neon(int width, int height, const uint8_t *src,
int src_stride, const uint8_t *const ref[4],
int ref_stride, uint32_t res[4]) {
const uint8_t *ref0, *ref1, *ref2, *ref3;
ref0 = ref[0];
ref1 = ref[1];
ref2 = ref[2];
ref3 = ref[3];
res[0] = 0;
res[1] = 0;
res[2] = 0;
res[3] = 0;
switch (width) {
case 4: {
uint32_t src4, ref40, ref41, ref42, ref43;
uint32x2_t q8 = vdup_n_u32(0);
uint32x2_t q4 = vdup_n_u32(0);
uint32x2_t q5 = vdup_n_u32(0);
uint32x2_t q6 = vdup_n_u32(0);
uint32x2_t q7 = vdup_n_u32(0);
for (int i = 0; i < height / 2; i++) {
uint16x4_t q0 = vdup_n_u16(0);
uint16x4_t q1 = vdup_n_u16(0);
uint16x4_t q2 = vdup_n_u16(0);
uint16x4_t q3 = vdup_n_u16(0);
memcpy(&src4, src, 4);
memcpy(&ref40, ref0, 4);
memcpy(&ref41, ref1, 4);
memcpy(&ref42, ref2, 4);
memcpy(&ref43, ref3, 4);
src += src_stride;
ref0 += ref_stride;
ref1 += ref_stride;
ref2 += ref_stride;
ref3 += ref_stride;
q8 = vset_lane_u32(src4, q8, 0);
q4 = vset_lane_u32(ref40, q4, 0);
q5 = vset_lane_u32(ref41, q5, 0);
q6 = vset_lane_u32(ref42, q6, 0);
q7 = vset_lane_u32(ref43, q7, 0);
memcpy(&src4, src, 4);
memcpy(&ref40, ref0, 4);
memcpy(&ref41, ref1, 4);
memcpy(&ref42, ref2, 4);
memcpy(&ref43, ref3, 4);
src += src_stride;
ref0 += ref_stride;
ref1 += ref_stride;
ref2 += ref_stride;
ref3 += ref_stride;
q8 = vset_lane_u32(src4, q8, 1);
q4 = vset_lane_u32(ref40, q4, 1);
q5 = vset_lane_u32(ref41, q5, 1);
q6 = vset_lane_u32(ref42, q6, 1);
q7 = vset_lane_u32(ref43, q7, 1);
sad_row4_neon(&q0, vreinterpret_u8_u32(q8), vreinterpret_u8_u32(q4));
sad_row4_neon(&q1, vreinterpret_u8_u32(q8), vreinterpret_u8_u32(q5));
sad_row4_neon(&q2, vreinterpret_u8_u32(q8), vreinterpret_u8_u32(q6));
sad_row4_neon(&q3, vreinterpret_u8_u32(q8), vreinterpret_u8_u32(q7));
res[0] += horizontal_add_16x4(q0);
res[1] += horizontal_add_16x4(q1);
res[2] += horizontal_add_16x4(q2);
res[3] += horizontal_add_16x4(q3);
}
break;
}
case 8: {
for (int i = 0; i < height; i++) {
uint16x4_t q0 = vdup_n_u16(0);
uint16x4_t q1 = vdup_n_u16(0);
uint16x4_t q2 = vdup_n_u16(0);
uint16x4_t q3 = vdup_n_u16(0);
uint8x8_t q5 = vld1_u8(src);
sad_row8_neon(&q0, &q5, ref0);
sad_row8_neon(&q1, &q5, ref1);
sad_row8_neon(&q2, &q5, ref2);
sad_row8_neon(&q3, &q5, ref3);
src += src_stride;
ref0 += ref_stride;
ref1 += ref_stride;
ref2 += ref_stride;
ref3 += ref_stride;
res[0] += horizontal_add_16x4(q0);
res[1] += horizontal_add_16x4(q1);
res[2] += horizontal_add_16x4(q2);
res[3] += horizontal_add_16x4(q3);
}
break;
}
case 16: {
for (int i = 0; i < height; i++) {
uint16x8_t q0 = vdupq_n_u16(0);
uint16x8_t q1 = vdupq_n_u16(0);
uint16x8_t q2 = vdupq_n_u16(0);
uint16x8_t q3 = vdupq_n_u16(0);
uint8x16_t q4 = vld1q_u8(src);
sad_row16_neon(&q0, &q4, ref0);
sad_row16_neon(&q1, &q4, ref1);
sad_row16_neon(&q2, &q4, ref2);
sad_row16_neon(&q3, &q4, ref3);
src += src_stride;
ref0 += ref_stride;
ref1 += ref_stride;
ref2 += ref_stride;
ref3 += ref_stride;
res[0] += horizontal_add_16x8(q0);
res[1] += horizontal_add_16x8(q1);
res[2] += horizontal_add_16x8(q2);
res[3] += horizontal_add_16x8(q3);
}
break;
}
case 32: {
for (int i = 0; i < height; i++) {
uint16x8_t q0 = vdupq_n_u16(0);
uint16x8_t q1 = vdupq_n_u16(0);
uint16x8_t q2 = vdupq_n_u16(0);
uint16x8_t q3 = vdupq_n_u16(0);
uint8x16_t q4 = vld1q_u8(src);
sad_row16_neon(&q0, &q4, ref0);
sad_row16_neon(&q1, &q4, ref1);
sad_row16_neon(&q2, &q4, ref2);
sad_row16_neon(&q3, &q4, ref3);
q4 = vld1q_u8(src + 16);
sad_row16_neon(&q0, &q4, ref0 + 16);
sad_row16_neon(&q1, &q4, ref1 + 16);
sad_row16_neon(&q2, &q4, ref2 + 16);
sad_row16_neon(&q3, &q4, ref3 + 16);
src += src_stride;
ref0 += ref_stride;
ref1 += ref_stride;
ref2 += ref_stride;
ref3 += ref_stride;
res[0] += horizontal_add_16x8(q0);
res[1] += horizontal_add_16x8(q1);
res[2] += horizontal_add_16x8(q2);
res[3] += horizontal_add_16x8(q3);
}
break;
}
case 64: {
for (int i = 0; i < height; i++) {
uint16x8_t q0 = vdupq_n_u16(0);
uint16x8_t q1 = vdupq_n_u16(0);
uint16x8_t q2 = vdupq_n_u16(0);
uint16x8_t q3 = vdupq_n_u16(0);
uint8x16_t q4 = vld1q_u8(src);
sad_row16_neon(&q0, &q4, ref0);
sad_row16_neon(&q1, &q4, ref1);
sad_row16_neon(&q2, &q4, ref2);
sad_row16_neon(&q3, &q4, ref3);
q4 = vld1q_u8(src + 16);
sad_row16_neon(&q0, &q4, ref0 + 16);
sad_row16_neon(&q1, &q4, ref1 + 16);
sad_row16_neon(&q2, &q4, ref2 + 16);
sad_row16_neon(&q3, &q4, ref3 + 16);
q4 = vld1q_u8(src + 32);
sad_row16_neon(&q0, &q4, ref0 + 32);
sad_row16_neon(&q1, &q4, ref1 + 32);
sad_row16_neon(&q2, &q4, ref2 + 32);
sad_row16_neon(&q3, &q4, ref3 + 32);
q4 = vld1q_u8(src + 48);
sad_row16_neon(&q0, &q4, ref0 + 48);
sad_row16_neon(&q1, &q4, ref1 + 48);
sad_row16_neon(&q2, &q4, ref2 + 48);
sad_row16_neon(&q3, &q4, ref3 + 48);
src += src_stride;
ref0 += ref_stride;
ref1 += ref_stride;
ref2 += ref_stride;
ref3 += ref_stride;
res[0] += horizontal_add_16x8(q0);
res[1] += horizontal_add_16x8(q1);
res[2] += horizontal_add_16x8(q2);
res[3] += horizontal_add_16x8(q3);
}
break;
}
case 128: {
for (int i = 0; i < height; i++) {
uint16x8_t q0 = vdupq_n_u16(0);
uint16x8_t q1 = vdupq_n_u16(0);
uint16x8_t q2 = vdupq_n_u16(0);
uint16x8_t q3 = vdupq_n_u16(0);
uint8x16_t q4 = vld1q_u8(src);
sad_row16_neon(&q0, &q4, ref0);
sad_row16_neon(&q1, &q4, ref1);
sad_row16_neon(&q2, &q4, ref2);
sad_row16_neon(&q3, &q4, ref3);
q4 = vld1q_u8(src + 16);
sad_row16_neon(&q0, &q4, ref0 + 16);
sad_row16_neon(&q1, &q4, ref1 + 16);
sad_row16_neon(&q2, &q4, ref2 + 16);
sad_row16_neon(&q3, &q4, ref3 + 16);
q4 = vld1q_u8(src + 32);
sad_row16_neon(&q0, &q4, ref0 + 32);
sad_row16_neon(&q1, &q4, ref1 + 32);
sad_row16_neon(&q2, &q4, ref2 + 32);
sad_row16_neon(&q3, &q4, ref3 + 32);
q4 = vld1q_u8(src + 48);
sad_row16_neon(&q0, &q4, ref0 + 48);
sad_row16_neon(&q1, &q4, ref1 + 48);
sad_row16_neon(&q2, &q4, ref2 + 48);
sad_row16_neon(&q3, &q4, ref3 + 48);
q4 = vld1q_u8(src + 64);
sad_row16_neon(&q0, &q4, ref0 + 64);
sad_row16_neon(&q1, &q4, ref1 + 64);
sad_row16_neon(&q2, &q4, ref2 + 64);
sad_row16_neon(&q3, &q4, ref3 + 64);
q4 = vld1q_u8(src + 80);
sad_row16_neon(&q0, &q4, ref0 + 80);
sad_row16_neon(&q1, &q4, ref1 + 80);
sad_row16_neon(&q2, &q4, ref2 + 80);
sad_row16_neon(&q3, &q4, ref3 + 80);
q4 = vld1q_u8(src + 96);
sad_row16_neon(&q0, &q4, ref0 + 96);
sad_row16_neon(&q1, &q4, ref1 + 96);
sad_row16_neon(&q2, &q4, ref2 + 96);
sad_row16_neon(&q3, &q4, ref3 + 96);
q4 = vld1q_u8(src + 112);
sad_row16_neon(&q0, &q4, ref0 + 112);
sad_row16_neon(&q1, &q4, ref1 + 112);
sad_row16_neon(&q2, &q4, ref2 + 112);
sad_row16_neon(&q3, &q4, ref3 + 112);
src += src_stride;
ref0 += ref_stride;
ref1 += ref_stride;
ref2 += ref_stride;
ref3 += ref_stride;
res[0] += horizontal_add_16x8(q0);
res[1] += horizontal_add_16x8(q1);
res[2] += horizontal_add_16x8(q2);
res[3] += horizontal_add_16x8(q3);
}
}
}
}
#define SAD_SKIP_MXN_NEON(m, n) \
void aom_sad_skip_##m##x##n##x4d_neon(const uint8_t *src, int src_stride, \
const uint8_t *const ref[4], \
int ref_stride, uint32_t res[4]) { \
aom_sadMxNx4d_neon(m, ((n) >> 1), src, 2 * src_stride, ref, \
2 * ref_stride, res); \
res[0] <<= 1; \
res[1] <<= 1; \
res[2] <<= 1; \
res[3] <<= 1; \
}
SAD_SKIP_MXN_NEON(4, 8)
SAD_SKIP_MXN_NEON(4, 16)
SAD_SKIP_MXN_NEON(4, 32)
SAD_SKIP_MXN_NEON(8, 8)
SAD_SKIP_MXN_NEON(8, 16)
SAD_SKIP_MXN_NEON(8, 32)
SAD_SKIP_MXN_NEON(16, 8)
SAD_SKIP_MXN_NEON(16, 16)
SAD_SKIP_MXN_NEON(16, 32)
SAD_SKIP_MXN_NEON(16, 64)
SAD_SKIP_MXN_NEON(32, 8)
SAD_SKIP_MXN_NEON(32, 16)
SAD_SKIP_MXN_NEON(32, 32)
SAD_SKIP_MXN_NEON(32, 64)
SAD_SKIP_MXN_NEON(64, 16)
SAD_SKIP_MXN_NEON(64, 32)
SAD_SKIP_MXN_NEON(64, 64)
SAD_SKIP_MXN_NEON(64, 128)
SAD_SKIP_MXN_NEON(128, 64)
SAD_SKIP_MXN_NEON(128, 128)
#undef SAD_SKIP_MXN_NEON

View file

@ -10,13 +10,12 @@
*/
#include <arm_neon.h>
#include "config/aom_config.h"
#include "config/aom_dsp_rtcd.h"
#include "aom/aom_integer.h"
unsigned int aom_sad8x16_neon(unsigned char *src_ptr, int src_stride,
unsigned char *ref_ptr, int ref_stride) {
unsigned int aom_sad8x16_neon(const uint8_t *src_ptr, int src_stride,
const uint8_t *ref_ptr, int ref_stride) {
uint8x8_t d0, d8;
uint16x8_t q12;
uint32x4_t q1;
@ -46,8 +45,8 @@ unsigned int aom_sad8x16_neon(unsigned char *src_ptr, int src_stride,
return vget_lane_u32(d5, 0);
}
unsigned int aom_sad4x4_neon(unsigned char *src_ptr, int src_stride,
unsigned char *ref_ptr, int ref_stride) {
unsigned int aom_sad4x4_neon(const uint8_t *src_ptr, int src_stride,
const uint8_t *ref_ptr, int ref_stride) {
uint8x8_t d0, d8;
uint16x8_t q12;
uint32x2_t d1;
@ -74,8 +73,8 @@ unsigned int aom_sad4x4_neon(unsigned char *src_ptr, int src_stride,
return vget_lane_u32(vreinterpret_u32_u64(d3), 0);
}
unsigned int aom_sad16x8_neon(unsigned char *src_ptr, int src_stride,
unsigned char *ref_ptr, int ref_stride) {
unsigned int aom_sad16x8_neon(const uint8_t *src_ptr, int src_stride,
const uint8_t *ref_ptr, int ref_stride) {
uint8x16_t q0, q4;
uint16x8_t q12, q13;
uint32x4_t q1;
@ -164,6 +163,77 @@ unsigned int aom_sad64x64_neon(const uint8_t *src, int src_stride,
return horizontal_long_add_16x8(vec_accum_lo, vec_accum_hi);
}
unsigned int aom_sad128x128_neon(const uint8_t *src, int src_stride,
const uint8_t *ref, int ref_stride) {
uint16x8_t vec_accum_lo, vec_accum_hi;
uint32x4_t vec_accum_32lo = vdupq_n_u32(0);
uint32x4_t vec_accum_32hi = vdupq_n_u32(0);
uint16x8_t tmp;
for (int i = 0; i < 128; ++i) {
const uint8x16_t vec_src_00 = vld1q_u8(src);
const uint8x16_t vec_src_16 = vld1q_u8(src + 16);
const uint8x16_t vec_src_32 = vld1q_u8(src + 32);
const uint8x16_t vec_src_48 = vld1q_u8(src + 48);
const uint8x16_t vec_src_64 = vld1q_u8(src + 64);
const uint8x16_t vec_src_80 = vld1q_u8(src + 80);
const uint8x16_t vec_src_96 = vld1q_u8(src + 96);
const uint8x16_t vec_src_112 = vld1q_u8(src + 112);
const uint8x16_t vec_ref_00 = vld1q_u8(ref);
const uint8x16_t vec_ref_16 = vld1q_u8(ref + 16);
const uint8x16_t vec_ref_32 = vld1q_u8(ref + 32);
const uint8x16_t vec_ref_48 = vld1q_u8(ref + 48);
const uint8x16_t vec_ref_64 = vld1q_u8(ref + 64);
const uint8x16_t vec_ref_80 = vld1q_u8(ref + 80);
const uint8x16_t vec_ref_96 = vld1q_u8(ref + 96);
const uint8x16_t vec_ref_112 = vld1q_u8(ref + 112);
src += src_stride;
ref += ref_stride;
vec_accum_lo = vdupq_n_u16(0);
vec_accum_hi = vdupq_n_u16(0);
vec_accum_lo = vabal_u8(vec_accum_lo, vget_low_u8(vec_src_00),
vget_low_u8(vec_ref_00));
vec_accum_hi = vabal_u8(vec_accum_hi, vget_high_u8(vec_src_00),
vget_high_u8(vec_ref_00));
vec_accum_lo = vabal_u8(vec_accum_lo, vget_low_u8(vec_src_16),
vget_low_u8(vec_ref_16));
vec_accum_hi = vabal_u8(vec_accum_hi, vget_high_u8(vec_src_16),
vget_high_u8(vec_ref_16));
vec_accum_lo = vabal_u8(vec_accum_lo, vget_low_u8(vec_src_32),
vget_low_u8(vec_ref_32));
vec_accum_hi = vabal_u8(vec_accum_hi, vget_high_u8(vec_src_32),
vget_high_u8(vec_ref_32));
vec_accum_lo = vabal_u8(vec_accum_lo, vget_low_u8(vec_src_48),
vget_low_u8(vec_ref_48));
vec_accum_hi = vabal_u8(vec_accum_hi, vget_high_u8(vec_src_48),
vget_high_u8(vec_ref_48));
vec_accum_lo = vabal_u8(vec_accum_lo, vget_low_u8(vec_src_64),
vget_low_u8(vec_ref_64));
vec_accum_hi = vabal_u8(vec_accum_hi, vget_high_u8(vec_src_64),
vget_high_u8(vec_ref_64));
vec_accum_lo = vabal_u8(vec_accum_lo, vget_low_u8(vec_src_80),
vget_low_u8(vec_ref_80));
vec_accum_hi = vabal_u8(vec_accum_hi, vget_high_u8(vec_src_80),
vget_high_u8(vec_ref_80));
vec_accum_lo = vabal_u8(vec_accum_lo, vget_low_u8(vec_src_96),
vget_low_u8(vec_ref_96));
vec_accum_hi = vabal_u8(vec_accum_hi, vget_high_u8(vec_src_96),
vget_high_u8(vec_ref_96));
vec_accum_lo = vabal_u8(vec_accum_lo, vget_low_u8(vec_src_112),
vget_low_u8(vec_ref_112));
vec_accum_hi = vabal_u8(vec_accum_hi, vget_high_u8(vec_src_112),
vget_high_u8(vec_ref_112));
tmp = vaddq_u16(vec_accum_lo, vec_accum_hi);
vec_accum_32lo = vaddw_u16(vec_accum_32lo, vget_low_u16(tmp));
vec_accum_32hi = vaddw_u16(vec_accum_32hi, vget_high_u16(tmp));
}
const uint32x4_t a = vaddq_u32(vec_accum_32lo, vec_accum_32hi);
const uint64x2_t b = vpaddlq_u32(a);
const uint32x2_t c = vadd_u32(vreinterpret_u32_u64(vget_low_u64(b)),
vreinterpret_u32_u64(vget_high_u64(b)));
return vget_lane_u32(c, 0);
}
unsigned int aom_sad32x32_neon(const uint8_t *src, int src_stride,
const uint8_t *ref, int ref_stride) {
int i;
@ -222,3 +292,273 @@ unsigned int aom_sad8x8_neon(const uint8_t *src, int src_stride,
}
return horizontal_add_16x8(vec_accum);
}
static INLINE unsigned int sad128xh_neon(const uint8_t *src_ptr, int src_stride,
const uint8_t *ref_ptr, int ref_stride,
int h) {
int sum = 0;
for (int i = 0; i < h; i++) {
uint16x8_t q3 = vdupq_n_u16(0);
uint8x16_t q0 = vld1q_u8(src_ptr);
uint8x16_t q1 = vld1q_u8(ref_ptr);
uint8x16_t q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 16);
q1 = vld1q_u8(ref_ptr + 16);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 32);
q1 = vld1q_u8(ref_ptr + 32);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 48);
q1 = vld1q_u8(ref_ptr + 48);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 64);
q1 = vld1q_u8(ref_ptr + 64);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 80);
q1 = vld1q_u8(ref_ptr + 80);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 96);
q1 = vld1q_u8(ref_ptr + 96);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 112);
q1 = vld1q_u8(ref_ptr + 112);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
src_ptr += src_stride;
ref_ptr += ref_stride;
sum += horizontal_add_16x8(q3);
}
return sum;
}
static INLINE unsigned int sad64xh_neon(const uint8_t *src_ptr, int src_stride,
const uint8_t *ref_ptr, int ref_stride,
int h) {
int sum = 0;
for (int i = 0; i < h; i++) {
uint16x8_t q3 = vdupq_n_u16(0);
uint8x16_t q0 = vld1q_u8(src_ptr);
uint8x16_t q1 = vld1q_u8(ref_ptr);
uint8x16_t q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 16);
q1 = vld1q_u8(ref_ptr + 16);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 32);
q1 = vld1q_u8(ref_ptr + 32);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 48);
q1 = vld1q_u8(ref_ptr + 48);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
src_ptr += src_stride;
ref_ptr += ref_stride;
sum += horizontal_add_16x8(q3);
}
return sum;
}
static INLINE unsigned int sad32xh_neon(const uint8_t *src_ptr, int src_stride,
const uint8_t *ref_ptr, int ref_stride,
int h) {
int sum = 0;
for (int i = 0; i < h; i++) {
uint16x8_t q3 = vdupq_n_u16(0);
uint8x16_t q0 = vld1q_u8(src_ptr);
uint8x16_t q1 = vld1q_u8(ref_ptr);
uint8x16_t q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
q0 = vld1q_u8(src_ptr + 16);
q1 = vld1q_u8(ref_ptr + 16);
q2 = vabdq_u8(q0, q1);
q3 = vpadalq_u8(q3, q2);
sum += horizontal_add_16x8(q3);
src_ptr += src_stride;
ref_ptr += ref_stride;
}
return sum;
}
static INLINE unsigned int sad16xh_neon(const uint8_t *src_ptr, int src_stride,
const uint8_t *ref_ptr, int ref_stride,
int h) {
int sum = 0;
for (int i = 0; i < h; i++) {
uint8x8_t q0 = vld1_u8(src_ptr);
uint8x8_t q1 = vld1_u8(ref_ptr);
sum += vget_lane_u16(vpaddl_u8(vabd_u8(q0, q1)), 0);
sum += vget_lane_u16(vpaddl_u8(vabd_u8(q0, q1)), 1);
sum += vget_lane_u16(vpaddl_u8(vabd_u8(q0, q1)), 2);
sum += vget_lane_u16(vpaddl_u8(vabd_u8(q0, q1)), 3);
q0 = vld1_u8(src_ptr + 8);
q1 = vld1_u8(ref_ptr + 8);
sum += vget_lane_u16(vpaddl_u8(vabd_u8(q0, q1)), 0);
sum += vget_lane_u16(vpaddl_u8(vabd_u8(q0, q1)), 1);
sum += vget_lane_u16(vpaddl_u8(vabd_u8(q0, q1)), 2);
sum += vget_lane_u16(vpaddl_u8(vabd_u8(q0, q1)), 3);
src_ptr += src_stride;
ref_ptr += ref_stride;
}
return sum;
}
static INLINE unsigned int sad8xh_neon(const uint8_t *src_ptr, int src_stride,
const uint8_t *ref_ptr, int ref_stride,
int h) {
uint16x8_t q3 = vdupq_n_u16(0);
for (int y = 0; y < h; y++) {
uint8x8_t q0 = vld1_u8(src_ptr);
uint8x8_t q1 = vld1_u8(ref_ptr);
src_ptr += src_stride;
ref_ptr += ref_stride;
q3 = vabal_u8(q3, q0, q1);
}
return horizontal_add_16x8(q3);
}
static INLINE unsigned int sad4xh_neon(const uint8_t *src_ptr, int src_stride,
const uint8_t *ref_ptr, int ref_stride,
int h) {
uint16x8_t q3 = vdupq_n_u16(0);
uint32x2_t q0 = vdup_n_u32(0);
uint32x2_t q1 = vdup_n_u32(0);
uint32_t src4, ref4;
for (int y = 0; y < h / 2; y++) {
memcpy(&src4, src_ptr, 4);
memcpy(&ref4, ref_ptr, 4);
src_ptr += src_stride;
ref_ptr += ref_stride;
q0 = vset_lane_u32(src4, q0, 0);
q1 = vset_lane_u32(ref4, q1, 0);
memcpy(&src4, src_ptr, 4);
memcpy(&ref4, ref_ptr, 4);
src_ptr += src_stride;
ref_ptr += ref_stride;
q0 = vset_lane_u32(src4, q0, 1);
q1 = vset_lane_u32(ref4, q1, 1);
q3 = vabal_u8(q3, vreinterpret_u8_u32(q0), vreinterpret_u8_u32(q1));
}
return horizontal_add_16x8(q3);
}
#define FSADS128_H(h) \
unsigned int aom_sad_skip_128x##h##_neon( \
const uint8_t *src_ptr, int src_stride, const uint8_t *ref_ptr, \
int ref_stride) { \
const uint32_t sum = sad128xh_neon(src_ptr, 2 * src_stride, ref_ptr, \
2 * ref_stride, h / 2); \
return 2 * sum; \
}
FSADS128_H(128)
FSADS128_H(64)
#undef FSADS128_H
#define FSADS64_H(h) \
unsigned int aom_sad_skip_64x##h##_neon( \
const uint8_t *src_ptr, int src_stride, const uint8_t *ref_ptr, \
int ref_stride) { \
return 2 * sad64xh_neon(src_ptr, src_stride * 2, ref_ptr, ref_stride * 2, \
h / 2); \
}
FSADS64_H(128)
FSADS64_H(64)
FSADS64_H(32)
FSADS64_H(16)
#undef FSADS64_H
#define FSADS32_H(h) \
unsigned int aom_sad_skip_32x##h##_neon( \
const uint8_t *src_ptr, int src_stride, const uint8_t *ref_ptr, \
int ref_stride) { \
return 2 * sad32xh_neon(src_ptr, src_stride * 2, ref_ptr, ref_stride * 2, \
h / 2); \
}
FSADS32_H(64)
FSADS32_H(32)
FSADS32_H(16)
FSADS32_H(8)
#undef FSADS32_H
#define FSADS16_H(h) \
unsigned int aom_sad_skip_16x##h##_neon( \
const uint8_t *src_ptr, int src_stride, const uint8_t *ref_ptr, \
int ref_stride) { \
return 2 * sad16xh_neon(src_ptr, src_stride * 2, ref_ptr, ref_stride * 2, \
h / 2); \
}
FSADS16_H(64)
FSADS16_H(32)
FSADS16_H(16)
FSADS16_H(8)
#undef FSADS16_H
#define FSADS8_H(h) \
unsigned int aom_sad_skip_8x##h##_neon( \
const uint8_t *src_ptr, int src_stride, const uint8_t *ref_ptr, \
int ref_stride) { \
return 2 * sad8xh_neon(src_ptr, src_stride * 2, ref_ptr, ref_stride * 2, \
h / 2); \
}
FSADS8_H(32)
FSADS8_H(16)
FSADS8_H(8)
#undef FSADS8_H
#define FSADS4_H(h) \
unsigned int aom_sad_skip_4x##h##_neon( \
const uint8_t *src_ptr, int src_stride, const uint8_t *ref_ptr, \
int ref_stride) { \
return 2 * sad4xh_neon(src_ptr, src_stride * 2, ref_ptr, ref_stride * 2, \
h / 2); \
}
FSADS4_H(16)
FSADS4_H(8)
#undef FSADS4_H

View file

@ -9,217 +9,176 @@
*/
#include <arm_neon.h>
#include "config/aom_config.h"
#include "config/aom_dsp_rtcd.h"
#include "aom/aom_integer.h"
#include "aom_dsp/arm/mem_neon.h"
#include "aom_dsp/arm/sum_neon.h"
#include "aom_dsp/arm/transpose_neon.h"
static INLINE uint32_t sse_W16x1_neon(uint8x16_t q2, uint8x16_t q3) {
const uint16_t sse1 = 0;
const uint16x8_t q1 = vld1q_dup_u16(&sse1);
uint32_t sse;
uint8x16_t q4 = vabdq_u8(q2, q3); // diff = abs(a[x] - b[x])
uint8x8_t d0 = vget_low_u8(q4);
uint8x8_t d1 = vget_high_u8(q4);
uint16x8_t q6 = vmlal_u8(q1, d0, d0);
uint16x8_t q7 = vmlal_u8(q1, d1, d1);
uint32x4_t q8 = vaddl_u16(vget_low_u16(q6), vget_high_u16(q6));
uint32x4_t q9 = vaddl_u16(vget_low_u16(q7), vget_high_u16(q7));
uint32x2_t d4 = vadd_u32(vget_low_u32(q8), vget_high_u32(q8));
uint32x2_t d5 = vadd_u32(vget_low_u32(q9), vget_high_u32(q9));
uint32x2_t d6 = vadd_u32(d4, d5);
sse = vget_lane_u32(d6, 0);
sse += vget_lane_u32(d6, 1);
return sse;
static INLINE void sse_w16_neon(uint32x4_t *sum, const uint8_t *a,
const uint8_t *b) {
const uint8x16_t v_a0 = vld1q_u8(a);
const uint8x16_t v_b0 = vld1q_u8(b);
const uint8x16_t diff = vabdq_u8(v_a0, v_b0);
const uint8x8_t diff_lo = vget_low_u8(diff);
const uint8x8_t diff_hi = vget_high_u8(diff);
*sum = vpadalq_u16(*sum, vmull_u8(diff_lo, diff_lo));
*sum = vpadalq_u16(*sum, vmull_u8(diff_hi, diff_hi));
}
static INLINE void aom_sse4x2_neon(const uint8_t *a, int a_stride,
const uint8_t *b, int b_stride,
uint32x4_t *sum) {
uint8x8_t v_a0, v_b0;
v_a0 = v_b0 = vcreate_u8(0);
// above line is only to shadow [-Werror=uninitialized]
v_a0 = vreinterpret_u8_u32(
vld1_lane_u32((uint32_t *)a, vreinterpret_u32_u8(v_a0), 0));
v_a0 = vreinterpret_u8_u32(
vld1_lane_u32((uint32_t *)(a + a_stride), vreinterpret_u32_u8(v_a0), 1));
v_b0 = vreinterpret_u8_u32(
vld1_lane_u32((uint32_t *)b, vreinterpret_u32_u8(v_b0), 0));
v_b0 = vreinterpret_u8_u32(
vld1_lane_u32((uint32_t *)(b + b_stride), vreinterpret_u32_u8(v_b0), 1));
const uint8x8_t v_a_w = vabd_u8(v_a0, v_b0);
*sum = vpadalq_u16(*sum, vmull_u8(v_a_w, v_a_w));
}
static INLINE void aom_sse8_neon(const uint8_t *a, const uint8_t *b,
uint32x4_t *sum) {
const uint8x8_t v_a_w = vld1_u8(a);
const uint8x8_t v_b_w = vld1_u8(b);
const uint8x8_t v_d_w = vabd_u8(v_a_w, v_b_w);
*sum = vpadalq_u16(*sum, vmull_u8(v_d_w, v_d_w));
}
int64_t aom_sse_neon(const uint8_t *a, int a_stride, const uint8_t *b,
int b_stride, int width, int height) {
const uint8x16_t q0 = {
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15
};
int addinc, x, y;
uint8x8_t d0, d1, d2, d3;
uint8_t dx;
uint8x16_t q2, q3, q4, q5;
uint32_t sse = 0;
uint8x8x2_t tmp, tmp2;
int y = 0;
int64_t sse = 0;
uint32x4_t sum = vdupq_n_u32(0);
switch (width) {
case 4:
for (y = 0; y < height; y += 4) {
d0 = vld1_u8(a); // load 4 data
a += a_stride;
d1 = vld1_u8(a);
a += a_stride;
d2 = vld1_u8(a);
a += a_stride;
d3 = vld1_u8(a);
a += a_stride;
tmp = vzip_u8(d0, d1);
tmp2 = vzip_u8(d2, d3);
q2 = vcombine_u8(tmp.val[0], tmp2.val[0]); // make a 16 data vector
d0 = vld1_u8(b);
b += b_stride;
d1 = vld1_u8(b);
b += b_stride;
d2 = vld1_u8(b);
b += b_stride;
d3 = vld1_u8(b);
b += b_stride;
tmp = vzip_u8(d0, d1);
tmp2 = vzip_u8(d2, d3);
q3 = vcombine_u8(tmp.val[0], tmp2.val[0]);
sse += sse_W16x1_neon(q2, q3);
}
do {
aom_sse4x2_neon(a, a_stride, b, b_stride, &sum);
a += a_stride << 1;
b += b_stride << 1;
y += 2;
} while (y < height);
#if defined(__aarch64__)
sse = vaddvq_u32(sum);
#else
sse = horizontal_add_s32x4(vreinterpretq_s32_u32(sum));
#endif // __aarch64__
break;
case 8:
for (y = 0; y < height; y += 2) {
d0 = vld1_u8(a); // load 8 data
d1 = vld1_u8(a + a_stride);
q2 = vcombine_u8(d0, d1); // make a 16 data vector
d0 = vld1_u8(b);
d1 = vld1_u8(b + b_stride);
q3 = vcombine_u8(d0, d1);
sse += sse_W16x1_neon(q2, q3);
a += 2 * a_stride;
b += 2 * b_stride;
}
do {
aom_sse8_neon(a, b, &sum);
a += a_stride;
b += b_stride;
y += 1;
} while (y < height);
#if defined(__aarch64__)
sse = vaddvq_u32(sum);
#else
sse = horizontal_add_s32x4(vreinterpretq_s32_u32(sum));
#endif // __aarch64__
break;
case 16:
for (y = 0; y < height; y++) {
q2 = vld1q_u8(a);
q3 = vld1q_u8(b);
sse += sse_W16x1_neon(q2, q3);
do {
sse_w16_neon(&sum, a, b);
a += a_stride;
b += b_stride;
}
y += 1;
} while (y < height);
#if defined(__aarch64__)
sse = vaddvq_u32(sum);
#else
sse = horizontal_add_s32x4(vreinterpretq_s32_u32(sum));
#endif // __aarch64__
break;
case 32:
for (y = 0; y < height; y++) {
q2 = vld1q_u8(a);
q3 = vld1q_u8(b);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 16);
q3 = vld1q_u8(b + 16);
sse += sse_W16x1_neon(q2, q3);
do {
sse_w16_neon(&sum, a, b);
sse_w16_neon(&sum, a + 16, b + 16);
a += a_stride;
b += b_stride;
}
y += 1;
} while (y < height);
#if defined(__aarch64__)
sse = vaddvq_u32(sum);
#else
sse = horizontal_add_s32x4(vreinterpretq_s32_u32(sum));
#endif // __aarch64__
break;
case 64:
for (y = 0; y < height; y++) {
q2 = vld1q_u8(a);
q3 = vld1q_u8(b);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 16);
q3 = vld1q_u8(b + 16);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 32);
q3 = vld1q_u8(b + 32);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 48);
q3 = vld1q_u8(b + 48);
sse += sse_W16x1_neon(q2, q3);
do {
sse_w16_neon(&sum, a, b);
sse_w16_neon(&sum, a + 16 * 1, b + 16 * 1);
sse_w16_neon(&sum, a + 16 * 2, b + 16 * 2);
sse_w16_neon(&sum, a + 16 * 3, b + 16 * 3);
a += a_stride;
b += b_stride;
}
y += 1;
} while (y < height);
#if defined(__aarch64__)
sse = vaddvq_u32(sum);
#else
sse = horizontal_add_s32x4(vreinterpretq_s32_u32(sum));
#endif // __aarch64__
break;
case 128:
for (y = 0; y < height; y++) {
q2 = vld1q_u8(a);
q3 = vld1q_u8(b);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 16);
q3 = vld1q_u8(b + 16);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 32);
q3 = vld1q_u8(b + 32);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 48);
q3 = vld1q_u8(b + 48);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 64);
q3 = vld1q_u8(b + 64);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 80);
q3 = vld1q_u8(b + 80);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 96);
q3 = vld1q_u8(b + 96);
sse += sse_W16x1_neon(q2, q3);
q2 = vld1q_u8(a + 112);
q3 = vld1q_u8(b + 112);
sse += sse_W16x1_neon(q2, q3);
do {
sse_w16_neon(&sum, a, b);
sse_w16_neon(&sum, a + 16 * 1, b + 16 * 1);
sse_w16_neon(&sum, a + 16 * 2, b + 16 * 2);
sse_w16_neon(&sum, a + 16 * 3, b + 16 * 3);
sse_w16_neon(&sum, a + 16 * 4, b + 16 * 4);
sse_w16_neon(&sum, a + 16 * 5, b + 16 * 5);
sse_w16_neon(&sum, a + 16 * 6, b + 16 * 6);
sse_w16_neon(&sum, a + 16 * 7, b + 16 * 7);
a += a_stride;
b += b_stride;
}
y += 1;
} while (y < height);
#if defined(__aarch64__)
sse = vaddvq_u32(sum);
#else
sse = horizontal_add_s32x4(vreinterpretq_s32_u32(sum));
#endif // __aarch64__
break;
default:
for (y = 0; y < height; y++) {
x = width;
while (x > 0) {
addinc = width - x;
q2 = vld1q_u8(a + addinc);
q3 = vld1q_u8(b + addinc);
if (x < 16) {
dx = x;
q4 = vld1q_dup_u8(&dx);
q5 = vcltq_u8(q0, q4);
q2 = vandq_u8(q2, q5);
q3 = vandq_u8(q3, q5);
}
sse += sse_W16x1_neon(q2, q3);
x -= 16;
}
a += a_stride;
b += b_stride;
if (width & 0x07) {
do {
int i = 0;
do {
aom_sse8_neon(a + i, b + i, &sum);
aom_sse8_neon(a + i + a_stride, b + i + b_stride, &sum);
i += 8;
} while (i + 4 < width);
aom_sse4x2_neon(a + i, a_stride, b + i, b_stride, &sum);
a += (a_stride << 1);
b += (b_stride << 1);
y += 2;
} while (y < height);
} else {
do {
int i = 0;
do {
aom_sse8_neon(a + i, b + i, &sum);
i += 8;
} while (i < width);
a += a_stride;
b += b_stride;
y += 1;
} while (y < height);
}
#if defined(__aarch64__)
sse = vaddvq_u32(sum);
#else
sse = horizontal_add_s32x4(vreinterpretq_s32_u32(sum));
#endif // __aarch64__
break;
}
return (int64_t)sse;
return sse;
}
#if CONFIG_AV1_HIGHBITDEPTH

View file

@ -20,6 +20,42 @@
#include "aom_dsp/aom_filter.h"
#include "aom_dsp/variance.h"
// Load 2 sets of 4 bytes when alignment is not guaranteed.
static INLINE uint8x8_t load_unaligned_u8(const uint8_t *buf, int stride) {
uint32_t a;
uint32x2_t a_u32 = vdup_n_u32(0);
if (stride == 4) return vld1_u8(buf);
memcpy(&a, buf, 4);
buf += stride;
a_u32 = vld1_lane_u32(&a, a_u32, 0);
memcpy(&a, buf, 4);
a_u32 = vld1_lane_u32(&a, a_u32, 1);
return vreinterpret_u8_u32(a_u32);
}
// Process a block exactly 4 wide and a multiple of 2 high.
static void var_filter_block2d_bil_w4(const uint8_t *src_ptr,
uint8_t *output_ptr,
unsigned int src_pixels_per_line,
int pixel_step,
unsigned int output_height,
const uint8_t *filter) {
const uint8x8_t f0 = vdup_n_u8(filter[0]);
const uint8x8_t f1 = vdup_n_u8(filter[1]);
unsigned int i;
for (i = 0; i < output_height; i += 2) {
const uint8x8_t src_0 = load_unaligned_u8(src_ptr, src_pixels_per_line);
const uint8x8_t src_1 =
load_unaligned_u8(src_ptr + pixel_step, src_pixels_per_line);
const uint16x8_t a = vmull_u8(src_0, f0);
const uint16x8_t b = vmlal_u8(a, src_1, f1);
const uint8x8_t out = vrshrn_n_u16(b, FILTER_BITS);
vst1_u8(output_ptr, out);
src_ptr += 2 * src_pixels_per_line;
output_ptr += 8;
}
}
static void var_filter_block2d_bil_w8(const uint8_t *src_ptr,
uint8_t *output_ptr,
unsigned int src_pixels_per_line,
@ -27,8 +63,8 @@ static void var_filter_block2d_bil_w8(const uint8_t *src_ptr,
unsigned int output_height,
unsigned int output_width,
const uint8_t *filter) {
const uint8x8_t f0 = vmov_n_u8(filter[0]);
const uint8x8_t f1 = vmov_n_u8(filter[1]);
const uint8x8_t f0 = vdup_n_u8(filter[0]);
const uint8x8_t f1 = vdup_n_u8(filter[1]);
unsigned int i;
for (i = 0; i < output_height; ++i) {
const uint8x8_t src_0 = vld1_u8(&src_ptr[0]);
@ -36,13 +72,14 @@ static void var_filter_block2d_bil_w8(const uint8_t *src_ptr,
const uint16x8_t a = vmull_u8(src_0, f0);
const uint16x8_t b = vmlal_u8(a, src_1, f1);
const uint8x8_t out = vrshrn_n_u16(b, FILTER_BITS);
vst1_u8(&output_ptr[0], out);
vst1_u8(output_ptr, out);
// Next row...
src_ptr += src_pixels_per_line;
output_ptr += output_width;
}
}
// Process a block which is a mutiple of 16 wide and any height.
static void var_filter_block2d_bil_w16(const uint8_t *src_ptr,
uint8_t *output_ptr,
unsigned int src_pixels_per_line,
@ -50,8 +87,8 @@ static void var_filter_block2d_bil_w16(const uint8_t *src_ptr,
unsigned int output_height,
unsigned int output_width,
const uint8_t *filter) {
const uint8x8_t f0 = vmov_n_u8(filter[0]);
const uint8x8_t f1 = vmov_n_u8(filter[1]);
const uint8x8_t f0 = vdup_n_u8(filter[0]);
const uint8x8_t f1 = vdup_n_u8(filter[1]);
unsigned int i, j;
for (i = 0; i < output_height; ++i) {
for (j = 0; j < output_width; j += 16) {
@ -63,9 +100,8 @@ static void var_filter_block2d_bil_w16(const uint8_t *src_ptr,
const uint16x8_t c = vmull_u8(vget_high_u8(src_0), f0);
const uint16x8_t d = vmlal_u8(c, vget_high_u8(src_1), f1);
const uint8x8_t out_hi = vrshrn_n_u16(d, FILTER_BITS);
vst1q_u8(&output_ptr[j], vcombine_u8(out_lo, out_hi));
vst1q_u8(output_ptr + j, vcombine_u8(out_lo, out_hi));
}
// Next row...
src_ptr += src_pixels_per_line;
output_ptr += output_width;
}
@ -129,3 +165,276 @@ unsigned int aom_sub_pixel_variance64x64_neon(const uint8_t *src,
bilinear_filters_2t[yoffset]);
return aom_variance64x64_neon(temp2, 64, dst, dst_stride, sse);
}
unsigned int aom_sub_pixel_variance4x4_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[4 * (4 + 2)];
uint8_t temp1[4 * 4];
var_filter_block2d_bil_w4(a, temp0, a_stride, 1, (4 + 2),
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w4(temp0, temp1, 4, 4, 4,
bilinear_filters_2t[yoffset]);
return aom_variance4x4(temp1, 4, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance4x8_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[4 * (8 + 2)];
uint8_t temp1[4 * 8];
var_filter_block2d_bil_w4(a, temp0, a_stride, 1, (8 + 2),
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w4(temp0, temp1, 4, 4, 8,
bilinear_filters_2t[yoffset]);
return aom_variance4x8(temp1, 4, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance8x4_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[8 * (4 + 1)];
uint8_t temp1[8 * 4];
var_filter_block2d_bil_w8(a, temp0, a_stride, 1, (4 + 1), 8,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w8(temp0, temp1, 8, 8, 4, 8,
bilinear_filters_2t[yoffset]);
return aom_variance8x4(temp1, 8, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance8x16_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[8 * (16 + 1)];
uint8_t temp1[8 * 16];
var_filter_block2d_bil_w8(a, temp0, a_stride, 1, (16 + 1), 8,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w8(temp0, temp1, 8, 8, 16, 8,
bilinear_filters_2t[yoffset]);
return aom_variance8x16(temp1, 8, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance16x8_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[16 * (8 + 1)];
uint8_t temp1[16 * 8];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (8 + 1), 16,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 16, 16, 8, 16,
bilinear_filters_2t[yoffset]);
return aom_variance16x8(temp1, 16, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance16x32_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[16 * (32 + 1)];
uint8_t temp1[16 * 32];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (32 + 1), 16,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 16, 16, 32, 16,
bilinear_filters_2t[yoffset]);
return aom_variance16x32(temp1, 16, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance32x16_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[32 * (16 + 1)];
uint8_t temp1[32 * 16];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (16 + 1), 32,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 32, 32, 16, 32,
bilinear_filters_2t[yoffset]);
return aom_variance32x16(temp1, 32, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance32x64_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[32 * (64 + 1)];
uint8_t temp1[32 * 64];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (64 + 1), 32,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 32, 32, 64, 32,
bilinear_filters_2t[yoffset]);
return aom_variance32x64(temp1, 32, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance64x32_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[64 * (32 + 1)];
uint8_t temp1[64 * 32];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (32 + 1), 64,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 64, 64, 32, 64,
bilinear_filters_2t[yoffset]);
return aom_variance64x32(temp1, 64, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance64x128_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[64 * (128 + 1)];
uint8_t temp1[64 * 128];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (128 + 1), 64,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 64, 64, 128, 64,
bilinear_filters_2t[yoffset]);
return aom_variance64x128(temp1, 64, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance128x64_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[128 * (64 + 1)];
uint8_t temp1[128 * 64];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (64 + 1), 128,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 128, 128, 64, 128,
bilinear_filters_2t[yoffset]);
return aom_variance128x64(temp1, 128, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance128x128_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[128 * (128 + 1)];
uint8_t temp1[128 * 128];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (128 + 1), 128,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 128, 128, 128, 128,
bilinear_filters_2t[yoffset]);
return aom_variance128x128(temp1, 128, b, b_stride, sse);
}
// Realtime mode doesn't use 4x rectangular blocks.
#if !CONFIG_REALTIME_ONLY
unsigned int aom_sub_pixel_variance4x16_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[4 * (16 + 2)];
uint8_t temp1[4 * 16];
var_filter_block2d_bil_w4(a, temp0, a_stride, 1, (16 + 2),
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w4(temp0, temp1, 4, 4, 16,
bilinear_filters_2t[yoffset]);
return aom_variance4x16(temp1, 4, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance8x32_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[8 * (32 + 1)];
uint8_t temp1[8 * 32];
var_filter_block2d_bil_w8(a, temp0, a_stride, 1, (32 + 1), 8,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w8(temp0, temp1, 8, 8, 32, 8,
bilinear_filters_2t[yoffset]);
return aom_variance8x32(temp1, 8, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance16x4_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[16 * (4 + 1)];
uint8_t temp1[16 * 4];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (4 + 1), 16,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 16, 16, 4, 16,
bilinear_filters_2t[yoffset]);
return aom_variance16x4(temp1, 16, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance64x16_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[64 * (16 + 1)];
uint8_t temp1[64 * 16];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (16 + 1), 64,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 64, 64, 16, 64,
bilinear_filters_2t[yoffset]);
return aom_variance64x16(temp1, 64, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance16x64_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[16 * (64 + 1)];
uint8_t temp1[16 * 64];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (64 + 1), 16,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 16, 16, 64, 16,
bilinear_filters_2t[yoffset]);
return aom_variance16x64(temp1, 16, b, b_stride, sse);
}
unsigned int aom_sub_pixel_variance32x8_neon(const uint8_t *a, int a_stride,
int xoffset, int yoffset,
const uint8_t *b, int b_stride,
uint32_t *sse) {
uint8_t temp0[32 * (8 + 1)];
uint8_t temp1[32 * 8];
var_filter_block2d_bil_w16(a, temp0, a_stride, 1, (8 + 1), 32,
bilinear_filters_2t[xoffset]);
var_filter_block2d_bil_w16(temp0, temp1, 32, 32, 8, 32,
bilinear_filters_2t[yoffset]);
return aom_variance32x8(temp1, 32, b, b_stride, sse);
}
#endif // !CONFIG_REALTIME_ONLY

View file

@ -14,16 +14,17 @@
#include "config/aom_config.h"
#include "aom/aom_integer.h"
#include "aom_ports/mem.h"
void aom_subtract_block_neon(int rows, int cols, int16_t *diff,
ptrdiff_t diff_stride, const uint8_t *src,
ptrdiff_t src_stride, const uint8_t *pred,
ptrdiff_t pred_stride) {
int r, c;
if (cols > 16) {
for (r = 0; r < rows; ++r) {
for (c = 0; c < cols; c += 32) {
int r = rows;
do {
int c = 0;
do {
const uint8x16_t v_src_00 = vld1q_u8(&src[c + 0]);
const uint8x16_t v_src_16 = vld1q_u8(&src[c + 16]);
const uint8x16_t v_pred_00 = vld1q_u8(&pred[c + 0]);
@ -40,13 +41,15 @@ void aom_subtract_block_neon(int rows, int cols, int16_t *diff,
vst1q_s16(&diff[c + 8], vreinterpretq_s16_u16(v_diff_hi_00));
vst1q_s16(&diff[c + 16], vreinterpretq_s16_u16(v_diff_lo_16));
vst1q_s16(&diff[c + 24], vreinterpretq_s16_u16(v_diff_hi_16));
}
c += 32;
} while (c < cols);
diff += diff_stride;
pred += pred_stride;
src += src_stride;
}
} while (--r != 0);
} else if (cols > 8) {
for (r = 0; r < rows; ++r) {
int r = rows;
do {
const uint8x16_t v_src = vld1q_u8(&src[0]);
const uint8x16_t v_pred = vld1q_u8(&pred[0]);
const uint16x8_t v_diff_lo =
@ -58,9 +61,10 @@ void aom_subtract_block_neon(int rows, int cols, int16_t *diff,
diff += diff_stride;
pred += pred_stride;
src += src_stride;
}
} while (--r != 0);
} else if (cols > 4) {
for (r = 0; r < rows; ++r) {
int r = rows;
do {
const uint8x8_t v_src = vld1_u8(&src[0]);
const uint8x8_t v_pred = vld1_u8(&pred[0]);
const uint16x8_t v_diff = vsubl_u8(v_src, v_pred);
@ -68,14 +72,95 @@ void aom_subtract_block_neon(int rows, int cols, int16_t *diff,
diff += diff_stride;
pred += pred_stride;
src += src_stride;
}
} while (--r != 0);
} else {
for (r = 0; r < rows; ++r) {
for (c = 0; c < cols; ++c) diff[c] = src[c] - pred[c];
int r = rows;
do {
int c = 0;
do {
diff[c] = src[c] - pred[c];
} while (++c < cols);
diff += diff_stride;
pred += pred_stride;
src += src_stride;
}
} while (--r != 0);
}
}
#if CONFIG_AV1_HIGHBITDEPTH
void aom_highbd_subtract_block_neon(int rows, int cols, int16_t *diff,
ptrdiff_t diff_stride, const uint8_t *src8,
ptrdiff_t src_stride, const uint8_t *pred8,
ptrdiff_t pred_stride) {
uint16_t *src = CONVERT_TO_SHORTPTR(src8);
uint16_t *pred = CONVERT_TO_SHORTPTR(pred8);
if (cols > 16) {
int r = rows;
do {
int c = 0;
do {
const uint16x8_t v_src_00 = vld1q_u16(&src[c + 0]);
const uint16x8_t v_pred_00 = vld1q_u16(&pred[c + 0]);
const uint16x8_t v_diff_00 = vsubq_u16(v_src_00, v_pred_00);
const uint16x8_t v_src_08 = vld1q_u16(&src[c + 8]);
const uint16x8_t v_pred_08 = vld1q_u16(&pred[c + 8]);
const uint16x8_t v_diff_08 = vsubq_u16(v_src_08, v_pred_08);
vst1q_s16(&diff[c + 0], vreinterpretq_s16_u16(v_diff_00));
vst1q_s16(&diff[c + 8], vreinterpretq_s16_u16(v_diff_08));
c += 16;
} while (c < cols);
diff += diff_stride;
pred += pred_stride;
src += src_stride;
} while (--r != 0);
} else if (cols > 8) {
int r = rows;
do {
const uint16x8_t v_src_00 = vld1q_u16(&src[0]);
const uint16x8_t v_pred_00 = vld1q_u16(&pred[0]);
const uint16x8_t v_diff_00 = vsubq_u16(v_src_00, v_pred_00);
const uint16x8_t v_src_08 = vld1q_u16(&src[8]);
const uint16x8_t v_pred_08 = vld1q_u16(&pred[8]);
const uint16x8_t v_diff_08 = vsubq_u16(v_src_08, v_pred_08);
vst1q_s16(&diff[0], vreinterpretq_s16_u16(v_diff_00));
vst1q_s16(&diff[8], vreinterpretq_s16_u16(v_diff_08));
diff += diff_stride;
pred += pred_stride;
src += src_stride;
} while (--r != 0);
} else if (cols > 4) {
int r = rows;
do {
const uint16x8_t v_src_r0 = vld1q_u16(&src[0]);
const uint16x8_t v_src_r1 = vld1q_u16(&src[src_stride]);
const uint16x8_t v_pred_r0 = vld1q_u16(&pred[0]);
const uint16x8_t v_pred_r1 = vld1q_u16(&pred[pred_stride]);
const uint16x8_t v_diff_r0 = vsubq_u16(v_src_r0, v_pred_r0);
const uint16x8_t v_diff_r1 = vsubq_u16(v_src_r1, v_pred_r1);
vst1q_s16(&diff[0], vreinterpretq_s16_u16(v_diff_r0));
vst1q_s16(&diff[diff_stride], vreinterpretq_s16_u16(v_diff_r1));
diff += diff_stride << 1;
pred += pred_stride << 1;
src += src_stride << 1;
r -= 2;
} while (r != 0);
} else {
int r = rows;
do {
const uint16x4_t v_src_r0 = vld1_u16(&src[0]);
const uint16x4_t v_src_r1 = vld1_u16(&src[src_stride]);
const uint16x4_t v_pred_r0 = vld1_u16(&pred[0]);
const uint16x4_t v_pred_r1 = vld1_u16(&pred[pred_stride]);
const uint16x4_t v_diff_r0 = vsub_u16(v_src_r0, v_pred_r0);
const uint16x4_t v_diff_r1 = vsub_u16(v_src_r1, v_pred_r1);
vst1_s16(&diff[0], vreinterpret_s16_u16(v_diff_r0));
vst1_s16(&diff[diff_stride], vreinterpret_s16_u16(v_diff_r1));
diff += diff_stride << 1;
pred += pred_stride << 1;
src += src_stride << 1;
r -= 2;
} while (r != 0);
}
}
#endif // CONFIG_AV1_HIGHBITDEPTH

View file

@ -0,0 +1,138 @@
/*
* Copyright (c) 2020, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include <arm_neon.h>
#include <assert.h>
#include "aom_dsp/arm/mem_neon.h"
#include "config/aom_dsp_rtcd.h"
static INLINE uint32x4_t sum_squares_i16_4x4_neon(const int16_t *src,
int stride) {
const int16x4_t v_val_01_lo = vld1_s16(src + 0 * stride);
const int16x4_t v_val_01_hi = vld1_s16(src + 1 * stride);
const int16x4_t v_val_23_lo = vld1_s16(src + 2 * stride);
const int16x4_t v_val_23_hi = vld1_s16(src + 3 * stride);
int32x4_t v_sq_01_d = vmull_s16(v_val_01_lo, v_val_01_lo);
v_sq_01_d = vmlal_s16(v_sq_01_d, v_val_01_hi, v_val_01_hi);
int32x4_t v_sq_23_d = vmull_s16(v_val_23_lo, v_val_23_lo);
v_sq_23_d = vmlal_s16(v_sq_23_d, v_val_23_hi, v_val_23_hi);
#if defined(__aarch64__)
return vreinterpretq_u32_s32(vpaddq_s32(v_sq_01_d, v_sq_23_d));
#else
return vreinterpretq_u32_s32(vcombine_s32(
vqmovn_s64(vpaddlq_s32(v_sq_01_d)), vqmovn_s64(vpaddlq_s32(v_sq_23_d))));
#endif
}
uint64_t aom_sum_squares_2d_i16_4x4_neon(const int16_t *src, int stride) {
const uint32x4_t v_sum_0123_d = sum_squares_i16_4x4_neon(src, stride);
#if defined(__aarch64__)
return (uint64_t)vaddvq_u32(v_sum_0123_d);
#else
uint64x2_t v_sum_d = vpaddlq_u32(v_sum_0123_d);
v_sum_d = vaddq_u64(v_sum_d, vextq_u64(v_sum_d, v_sum_d, 1));
return vgetq_lane_u64(v_sum_d, 0);
#endif
}
uint64_t aom_sum_squares_2d_i16_4xn_neon(const int16_t *src, int stride,
int height) {
int r = 0;
uint32x4_t v_acc_q = vdupq_n_u32(0);
do {
const uint32x4_t v_acc_d = sum_squares_i16_4x4_neon(src, stride);
v_acc_q = vaddq_u32(v_acc_q, v_acc_d);
src += stride << 2;
r += 4;
} while (r < height);
uint64x2_t v_acc_64 = vpaddlq_u32(v_acc_q);
#if defined(__aarch64__)
return vaddvq_u64(v_acc_64);
#else
v_acc_64 = vaddq_u64(v_acc_64, vextq_u64(v_acc_64, v_acc_64, 1));
return vgetq_lane_u64(v_acc_64, 0);
#endif
}
uint64_t aom_sum_squares_2d_i16_nxn_neon(const int16_t *src, int stride,
int width, int height) {
int r = 0;
const int32x4_t zero = vdupq_n_s32(0);
uint64x2_t v_acc_q = vreinterpretq_u64_s32(zero);
do {
int32x4_t v_sum = zero;
int c = 0;
do {
const int16_t *b = src + c;
const int16x8_t v_val_0 = vld1q_s16(b + 0 * stride);
const int16x8_t v_val_1 = vld1q_s16(b + 1 * stride);
const int16x8_t v_val_2 = vld1q_s16(b + 2 * stride);
const int16x8_t v_val_3 = vld1q_s16(b + 3 * stride);
const int16x4_t v_val_0_lo = vget_low_s16(v_val_0);
const int16x4_t v_val_1_lo = vget_low_s16(v_val_1);
const int16x4_t v_val_2_lo = vget_low_s16(v_val_2);
const int16x4_t v_val_3_lo = vget_low_s16(v_val_3);
int32x4_t v_sum_01 = vmull_s16(v_val_0_lo, v_val_0_lo);
v_sum_01 = vmlal_s16(v_sum_01, v_val_1_lo, v_val_1_lo);
int32x4_t v_sum_23 = vmull_s16(v_val_2_lo, v_val_2_lo);
v_sum_23 = vmlal_s16(v_sum_23, v_val_3_lo, v_val_3_lo);
#if defined(__aarch64__)
v_sum_01 = vmlal_high_s16(v_sum_01, v_val_0, v_val_0);
v_sum_01 = vmlal_high_s16(v_sum_01, v_val_1, v_val_1);
v_sum_23 = vmlal_high_s16(v_sum_23, v_val_2, v_val_2);
v_sum_23 = vmlal_high_s16(v_sum_23, v_val_3, v_val_3);
v_sum = vaddq_s32(v_sum, vpaddq_s32(v_sum_01, v_sum_23));
#else
const int16x4_t v_val_0_hi = vget_high_s16(v_val_0);
const int16x4_t v_val_1_hi = vget_high_s16(v_val_1);
const int16x4_t v_val_2_hi = vget_high_s16(v_val_2);
const int16x4_t v_val_3_hi = vget_high_s16(v_val_3);
v_sum_01 = vmlal_s16(v_sum_01, v_val_0_hi, v_val_0_hi);
v_sum_01 = vmlal_s16(v_sum_01, v_val_1_hi, v_val_1_hi);
v_sum_23 = vmlal_s16(v_sum_23, v_val_2_hi, v_val_2_hi);
v_sum_23 = vmlal_s16(v_sum_23, v_val_3_hi, v_val_3_hi);
v_sum = vaddq_s32(v_sum, vcombine_s32(vqmovn_s64(vpaddlq_s32(v_sum_01)),
vqmovn_s64(vpaddlq_s32(v_sum_23))));
#endif
c += 8;
} while (c < width);
v_acc_q = vpadalq_u32(v_acc_q, vreinterpretq_u32_s32(v_sum));
src += 4 * stride;
r += 4;
} while (r < height);
#if defined(__aarch64__)
return vaddvq_u64(v_acc_q);
#else
v_acc_q = vaddq_u64(v_acc_q, vextq_u64(v_acc_q, v_acc_q, 1));
return vgetq_lane_u64(v_acc_q, 0);
#endif
}
uint64_t aom_sum_squares_2d_i16_neon(const int16_t *src, int stride, int width,
int height) {
// 4 elements per row only requires half an SIMD register, so this
// must be a special case, but also note that over 75% of all calls
// are with size == 4, so it is also the common case.
if (LIKELY(width == 4 && height == 4)) {
return aom_sum_squares_2d_i16_4x4_neon(src, stride);
} else if (LIKELY(width == 4 && (height & 3) == 0)) {
return aom_sum_squares_2d_i16_4xn_neon(src, stride, height);
} else if (LIKELY((width & 7) == 0 && (height & 3) == 0)) {
// Generic case
return aom_sum_squares_2d_i16_nxn_neon(src, stride, width, height);
} else {
return aom_sum_squares_2d_i16_c(src, stride, width, height);
}
}

View file

@ -8,11 +8,16 @@
* be found in the AUTHORS file in the root of the source tree.
*/
#ifndef AOM_AV1_COMMON_ARM_TRANSPOSE_NEON_H_
#define AOM_AV1_COMMON_ARM_TRANSPOSE_NEON_H_
#ifndef AOM_AOM_DSP_ARM_TRANSPOSE_NEON_H_
#define AOM_AOM_DSP_ARM_TRANSPOSE_NEON_H_
#include <arm_neon.h>
// Swap high and low halves.
static INLINE uint16x8_t transpose64_u16q(const uint16x8_t a) {
return vextq_u16(a, a, 4);
}
static INLINE void transpose_u8_8x8(uint8x8_t *a0, uint8x8_t *a1, uint8x8_t *a2,
uint8x8_t *a3, uint8x8_t *a4, uint8x8_t *a5,
uint8x8_t *a6, uint8x8_t *a7) {
@ -185,6 +190,153 @@ static INLINE void transpose_u8_4x8(uint8x8_t *a0, uint8x8_t *a1, uint8x8_t *a2,
*a3 = d1.val[1];
}
// Input:
// 00 01 02 03
// 10 11 12 13
// 20 21 22 23
// 30 31 32 33
// Output:
// 00 10 20 30
// 01 11 21 31
// 02 12 22 32
// 03 13 23 33
static INLINE void transpose_u16_4x4(uint16x4_t a[4]) {
// b:
// 00 10 02 12
// 01 11 03 13
const uint16x4x2_t b = vtrn_u16(a[0], a[1]);
// c:
// 20 30 22 32
// 21 31 23 33
const uint16x4x2_t c = vtrn_u16(a[2], a[3]);
// d:
// 00 10 20 30
// 02 12 22 32
const uint32x2x2_t d =
vtrn_u32(vreinterpret_u32_u16(b.val[0]), vreinterpret_u32_u16(c.val[0]));
// e:
// 01 11 21 31
// 03 13 23 33
const uint32x2x2_t e =
vtrn_u32(vreinterpret_u32_u16(b.val[1]), vreinterpret_u32_u16(c.val[1]));
a[0] = vreinterpret_u16_u32(d.val[0]);
a[1] = vreinterpret_u16_u32(e.val[0]);
a[2] = vreinterpret_u16_u32(d.val[1]);
a[3] = vreinterpret_u16_u32(e.val[1]);
}
// 4x8 Input:
// a[0]: 00 01 02 03 04 05 06 07
// a[1]: 10 11 12 13 14 15 16 17
// a[2]: 20 21 22 23 24 25 26 27
// a[3]: 30 31 32 33 34 35 36 37
// 8x4 Output:
// a[0]: 00 10 20 30 04 14 24 34
// a[1]: 01 11 21 31 05 15 25 35
// a[2]: 02 12 22 32 06 16 26 36
// a[3]: 03 13 23 33 07 17 27 37
static INLINE void transpose_u16_4x8q(uint16x8_t a[4]) {
// b0.val[0]: 00 10 02 12 04 14 06 16
// b0.val[1]: 01 11 03 13 05 15 07 17
// b1.val[0]: 20 30 22 32 24 34 26 36
// b1.val[1]: 21 31 23 33 25 35 27 37
const uint16x8x2_t b0 = vtrnq_u16(a[0], a[1]);
const uint16x8x2_t b1 = vtrnq_u16(a[2], a[3]);
// c0.val[0]: 00 10 20 30 04 14 24 34
// c0.val[1]: 02 12 22 32 06 16 26 36
// c1.val[0]: 01 11 21 31 05 15 25 35
// c1.val[1]: 03 13 23 33 07 17 27 37
const uint32x4x2_t c0 = vtrnq_u32(vreinterpretq_u32_u16(b0.val[0]),
vreinterpretq_u32_u16(b1.val[0]));
const uint32x4x2_t c1 = vtrnq_u32(vreinterpretq_u32_u16(b0.val[1]),
vreinterpretq_u32_u16(b1.val[1]));
a[0] = vreinterpretq_u16_u32(c0.val[0]);
a[1] = vreinterpretq_u16_u32(c1.val[0]);
a[2] = vreinterpretq_u16_u32(c0.val[1]);
a[3] = vreinterpretq_u16_u32(c1.val[1]);
}
static INLINE uint16x8x2_t aom_vtrnq_u64_to_u16(const uint32x4_t a0,
const uint32x4_t a1) {
uint16x8x2_t b0;
b0.val[0] = vcombine_u16(vreinterpret_u16_u32(vget_low_u32(a0)),
vreinterpret_u16_u32(vget_low_u32(a1)));
b0.val[1] = vcombine_u16(vreinterpret_u16_u32(vget_high_u32(a0)),
vreinterpret_u16_u32(vget_high_u32(a1)));
return b0;
}
// Special transpose for loop filter.
// 4x8 Input:
// p_q: p3 p2 p1 p0 q0 q1 q2 q3
// a[0]: 00 01 02 03 04 05 06 07
// a[1]: 10 11 12 13 14 15 16 17
// a[2]: 20 21 22 23 24 25 26 27
// a[3]: 30 31 32 33 34 35 36 37
// 8x4 Output:
// a[0]: 03 13 23 33 04 14 24 34 p0q0
// a[1]: 02 12 22 32 05 15 25 35 p1q1
// a[2]: 01 11 21 31 06 16 26 36 p2q2
// a[3]: 00 10 20 30 07 17 27 37 p3q3
// Direct reapplication of the function will reset the high halves, but
// reverse the low halves:
// p_q: p0 p1 p2 p3 q0 q1 q2 q3
// a[0]: 33 32 31 30 04 05 06 07
// a[1]: 23 22 21 20 14 15 16 17
// a[2]: 13 12 11 10 24 25 26 27
// a[3]: 03 02 01 00 34 35 36 37
// Simply reordering the inputs (3, 2, 1, 0) will reset the low halves, but
// reverse the high halves.
// The standard transpose_u16_4x8q will produce the same reversals, but with the
// order of the low halves also restored relative to the high halves. This is
// preferable because it puts all values from the same source row back together,
// but some post-processing is inevitable.
static INLINE void loop_filter_transpose_u16_4x8q(uint16x8_t a[4]) {
// b0.val[0]: 00 10 02 12 04 14 06 16
// b0.val[1]: 01 11 03 13 05 15 07 17
// b1.val[0]: 20 30 22 32 24 34 26 36
// b1.val[1]: 21 31 23 33 25 35 27 37
const uint16x8x2_t b0 = vtrnq_u16(a[0], a[1]);
const uint16x8x2_t b1 = vtrnq_u16(a[2], a[3]);
// Reverse odd vectors to bring the appropriate items to the front of zips.
// b0.val[0]: 00 10 02 12 04 14 06 16
// r0 : 03 13 01 11 07 17 05 15
// b1.val[0]: 20 30 22 32 24 34 26 36
// r1 : 23 33 21 31 27 37 25 35
const uint32x4_t r0 = vrev64q_u32(vreinterpretq_u32_u16(b0.val[1]));
const uint32x4_t r1 = vrev64q_u32(vreinterpretq_u32_u16(b1.val[1]));
// Zip to complete the halves.
// c0.val[0]: 00 10 20 30 02 12 22 32 p3p1
// c0.val[1]: 04 14 24 34 06 16 26 36 q0q2
// c1.val[0]: 03 13 23 33 01 11 21 31 p0p2
// c1.val[1]: 07 17 27 37 05 15 25 35 q3q1
const uint32x4x2_t c0 = vzipq_u32(vreinterpretq_u32_u16(b0.val[0]),
vreinterpretq_u32_u16(b1.val[0]));
const uint32x4x2_t c1 = vzipq_u32(r0, r1);
// d0.val[0]: 00 10 20 30 07 17 27 37 p3q3
// d0.val[1]: 02 12 22 32 05 15 25 35 p1q1
// d1.val[0]: 03 13 23 33 04 14 24 34 p0q0
// d1.val[1]: 01 11 21 31 06 16 26 36 p2q2
const uint16x8x2_t d0 = aom_vtrnq_u64_to_u16(c0.val[0], c1.val[1]);
// The third row of c comes first here to swap p2 with q0.
const uint16x8x2_t d1 = aom_vtrnq_u64_to_u16(c1.val[0], c0.val[1]);
// 8x4 Output:
// a[0]: 03 13 23 33 04 14 24 34 p0q0
// a[1]: 02 12 22 32 05 15 25 35 p1q1
// a[2]: 01 11 21 31 06 16 26 36 p2q2
// a[3]: 00 10 20 30 07 17 27 37 p3q3
a[0] = d1.val[0]; // p0q0
a[1] = d0.val[1]; // p1q1
a[2] = d1.val[1]; // p2q2
a[3] = d0.val[0]; // p3q3
}
static INLINE void transpose_u16_4x8(uint16x4_t *a0, uint16x4_t *a1,
uint16x4_t *a2, uint16x4_t *a3,
uint16x4_t *a4, uint16x4_t *a5,
@ -599,4 +751,4 @@ static INLINE void transpose_s32_4x4(int32x4_t *a0, int32x4_t *a1,
*a3 = c1.val[1];
}
#endif // AOM_AV1_COMMON_ARM_TRANSPOSE_NEON_H_
#endif // AOM_AOM_DSP_ARM_TRANSPOSE_NEON_H_

View file

@ -56,6 +56,18 @@ void aom_get16x16var_neon(const uint8_t *a, int a_stride, const uint8_t *b,
variance_neon_w8(a, a_stride, b, b_stride, 16, 16, sse, sum);
}
// TODO(yunqingwang): Perform variance of two/four 8x8 blocks similar to that of
// AVX2.
void aom_get_sse_sum_8x8_quad_neon(const uint8_t *a, int a_stride,
const uint8_t *b, int b_stride,
unsigned int *sse, int *sum) {
// Loop over 4 8x8 blocks. Process one 8x32 block.
for (int k = 0; k < 4; k++) {
variance_neon_w8(a + (k * 8), a_stride, b + (k * 8), b_stride, 8, 8,
&sse[k], &sum[k]);
}
}
unsigned int aom_variance8x8_neon(const uint8_t *a, int a_stride,
const uint8_t *b, int b_stride,
unsigned int *sse) {
@ -399,3 +411,257 @@ unsigned int aom_get4x4sse_cs_neon(const unsigned char *src_ptr,
return vget_lane_u32(vreinterpret_u32_s64(d0s64), 0);
}
// Load 4 sets of 4 bytes when alignment is not guaranteed.
static INLINE uint8x16_t load_unaligned_u8q(const uint8_t *buf, int stride) {
uint32_t a;
uint32x4_t a_u32 = vdupq_n_u32(0);
if (stride == 4) return vld1q_u8(buf);
memcpy(&a, buf, 4);
buf += stride;
a_u32 = vld1q_lane_u32(&a, a_u32, 0);
memcpy(&a, buf, 4);
buf += stride;
a_u32 = vld1q_lane_u32(&a, a_u32, 1);
memcpy(&a, buf, 4);
buf += stride;
a_u32 = vld1q_lane_u32(&a, a_u32, 2);
memcpy(&a, buf, 4);
buf += stride;
a_u32 = vld1q_lane_u32(&a, a_u32, 3);
return vreinterpretq_u8_u32(a_u32);
}
// The variance helper functions use int16_t for sum. 8 values are accumulated
// and then added (at which point they expand up to int32_t). To avoid overflow,
// there can be no more than 32767 / 255 ~= 128 values accumulated in each
// column. For a 32x32 buffer, this results in 32 / 8 = 4 values per row * 32
// rows = 128. Asserts have been added to each function to warn against reaching
// this limit.
// Process a block of width 4 four rows at a time.
static void variance_neon_w4x4(const uint8_t *a, int a_stride, const uint8_t *b,
int b_stride, int h, uint32_t *sse, int *sum) {
const int32x4_t zero = vdupq_n_s32(0);
int16x8_t sum_s16 = vreinterpretq_s16_s32(zero);
int32x4_t sse_s32 = zero;
// Since width is only 4, sum_s16 only loads a half row per loop.
assert(h <= 256);
int i;
for (i = 0; i < h; i += 4) {
const uint8x16_t a_u8 = load_unaligned_u8q(a, a_stride);
const uint8x16_t b_u8 = load_unaligned_u8q(b, b_stride);
const int16x8_t diff_lo_s16 =
vreinterpretq_s16_u16(vsubl_u8(vget_low_u8(a_u8), vget_low_u8(b_u8)));
const int16x8_t diff_hi_s16 =
vreinterpretq_s16_u16(vsubl_u8(vget_high_u8(a_u8), vget_high_u8(b_u8)));
sum_s16 = vaddq_s16(sum_s16, diff_lo_s16);
sum_s16 = vaddq_s16(sum_s16, diff_hi_s16);
sse_s32 = vmlal_s16(sse_s32, vget_low_s16(diff_lo_s16),
vget_low_s16(diff_lo_s16));
sse_s32 = vmlal_s16(sse_s32, vget_high_s16(diff_lo_s16),
vget_high_s16(diff_lo_s16));
sse_s32 = vmlal_s16(sse_s32, vget_low_s16(diff_hi_s16),
vget_low_s16(diff_hi_s16));
sse_s32 = vmlal_s16(sse_s32, vget_high_s16(diff_hi_s16),
vget_high_s16(diff_hi_s16));
a += 4 * a_stride;
b += 4 * b_stride;
}
#if defined(__aarch64__)
*sum = vaddvq_s32(vpaddlq_s16(sum_s16));
*sse = (uint32_t)vaddvq_s32(sse_s32);
#else
*sum = horizontal_add_s16x8(sum_s16);
*sse = (uint32_t)horizontal_add_s32x4(sse_s32);
#endif
}
// Process a block of any size where the width is divisible by 16.
static void variance_neon_w16(const uint8_t *a, int a_stride, const uint8_t *b,
int b_stride, int w, int h, uint32_t *sse,
int *sum) {
const int32x4_t zero = vdupq_n_s32(0);
int16x8_t sum_s16 = vreinterpretq_s16_s32(zero);
int32x4_t sse_s32 = zero;
// The loop loads 16 values at a time but doubles them up when accumulating
// into sum_s16.
assert(w / 8 * h <= 128);
int i, j;
for (i = 0; i < h; ++i) {
for (j = 0; j < w; j += 16) {
const uint8x16_t a_u8 = vld1q_u8(a + j);
const uint8x16_t b_u8 = vld1q_u8(b + j);
const int16x8_t diff_lo_s16 =
vreinterpretq_s16_u16(vsubl_u8(vget_low_u8(a_u8), vget_low_u8(b_u8)));
const int16x8_t diff_hi_s16 = vreinterpretq_s16_u16(
vsubl_u8(vget_high_u8(a_u8), vget_high_u8(b_u8)));
sum_s16 = vaddq_s16(sum_s16, diff_lo_s16);
sum_s16 = vaddq_s16(sum_s16, diff_hi_s16);
sse_s32 = vmlal_s16(sse_s32, vget_low_s16(diff_lo_s16),
vget_low_s16(diff_lo_s16));
sse_s32 = vmlal_s16(sse_s32, vget_high_s16(diff_lo_s16),
vget_high_s16(diff_lo_s16));
sse_s32 = vmlal_s16(sse_s32, vget_low_s16(diff_hi_s16),
vget_low_s16(diff_hi_s16));
sse_s32 = vmlal_s16(sse_s32, vget_high_s16(diff_hi_s16),
vget_high_s16(diff_hi_s16));
}
a += a_stride;
b += b_stride;
}
#if defined(__aarch64__)
*sum = vaddvq_s32(vpaddlq_s16(sum_s16));
*sse = (uint32_t)vaddvq_s32(sse_s32);
#else
*sum = horizontal_add_s16x8(sum_s16);
*sse = (uint32_t)horizontal_add_s32x4(sse_s32);
#endif
}
// Process a block of width 8 two rows at a time.
static void variance_neon_w8x2(const uint8_t *a, int a_stride, const uint8_t *b,
int b_stride, int h, uint32_t *sse, int *sum) {
const int32x4_t zero = vdupq_n_s32(0);
int16x8_t sum_s16 = vreinterpretq_s16_s32(zero);
int32x4_t sse_s32 = zero;
// Each column has it's own accumulator entry in sum_s16.
assert(h <= 128);
int i = 0;
do {
const uint8x8_t a_0_u8 = vld1_u8(a);
const uint8x8_t a_1_u8 = vld1_u8(a + a_stride);
const uint8x8_t b_0_u8 = vld1_u8(b);
const uint8x8_t b_1_u8 = vld1_u8(b + b_stride);
const int16x8_t diff_0_s16 =
vreinterpretq_s16_u16(vsubl_u8(a_0_u8, b_0_u8));
const int16x8_t diff_1_s16 =
vreinterpretq_s16_u16(vsubl_u8(a_1_u8, b_1_u8));
sum_s16 = vaddq_s16(sum_s16, diff_0_s16);
sum_s16 = vaddq_s16(sum_s16, diff_1_s16);
sse_s32 =
vmlal_s16(sse_s32, vget_low_s16(diff_0_s16), vget_low_s16(diff_0_s16));
sse_s32 =
vmlal_s16(sse_s32, vget_low_s16(diff_1_s16), vget_low_s16(diff_1_s16));
sse_s32 = vmlal_s16(sse_s32, vget_high_s16(diff_0_s16),
vget_high_s16(diff_0_s16));
sse_s32 = vmlal_s16(sse_s32, vget_high_s16(diff_1_s16),
vget_high_s16(diff_1_s16));
a += a_stride + a_stride;
b += b_stride + b_stride;
i += 2;
} while (i < h);
#if defined(__aarch64__)
*sum = vaddvq_s32(vpaddlq_s16(sum_s16));
*sse = (uint32_t)vaddvq_s32(sse_s32);
#else
*sum = horizontal_add_s16x8(sum_s16);
*sse = (uint32_t)horizontal_add_s32x4(sse_s32);
#endif
}
#define VARIANCE_NXM(n, m, shift) \
unsigned int aom_variance##n##x##m##_neon(const uint8_t *a, int a_stride, \
const uint8_t *b, int b_stride, \
unsigned int *sse) { \
int sum; \
if (n == 4) \
variance_neon_w4x4(a, a_stride, b, b_stride, m, sse, &sum); \
else if (n == 8) \
variance_neon_w8x2(a, a_stride, b, b_stride, m, sse, &sum); \
else \
variance_neon_w16(a, a_stride, b, b_stride, n, m, sse, &sum); \
if (n * m < 16 * 16) \
return *sse - ((sum * sum) >> shift); \
else \
return *sse - (uint32_t)(((int64_t)sum * sum) >> shift); \
}
static void variance_neon_wide_block(const uint8_t *a, int a_stride,
const uint8_t *b, int b_stride, int w,
int h, uint32_t *sse, int *sum) {
const int32x4_t zero = vdupq_n_s32(0);
int32x4_t v_diff = zero;
int64x2_t v_sse = vreinterpretq_s64_s32(zero);
int s, i, j;
for (s = 0; s < 16; s++) {
int32x4_t sse_s32 = zero;
int16x8_t sum_s16 = vreinterpretq_s16_s32(zero);
for (i = (s * h) >> 4; i < (((s + 1) * h) >> 4); ++i) {
for (j = 0; j < w; j += 16) {
const uint8x16_t a_u8 = vld1q_u8(a + j);
const uint8x16_t b_u8 = vld1q_u8(b + j);
const int16x8_t diff_lo_s16 = vreinterpretq_s16_u16(
vsubl_u8(vget_low_u8(a_u8), vget_low_u8(b_u8)));
const int16x8_t diff_hi_s16 = vreinterpretq_s16_u16(
vsubl_u8(vget_high_u8(a_u8), vget_high_u8(b_u8)));
sum_s16 = vaddq_s16(sum_s16, diff_lo_s16);
sum_s16 = vaddq_s16(sum_s16, diff_hi_s16);
sse_s32 = vmlal_s16(sse_s32, vget_low_s16(diff_lo_s16),
vget_low_s16(diff_lo_s16));
sse_s32 = vmlal_s16(sse_s32, vget_high_s16(diff_lo_s16),
vget_high_s16(diff_lo_s16));
sse_s32 = vmlal_s16(sse_s32, vget_low_s16(diff_hi_s16),
vget_low_s16(diff_hi_s16));
sse_s32 = vmlal_s16(sse_s32, vget_high_s16(diff_hi_s16),
vget_high_s16(diff_hi_s16));
}
a += a_stride;
b += b_stride;
}
v_diff = vpadalq_s16(v_diff, sum_s16);
v_sse = vpadalq_s32(v_sse, sse_s32);
}
#if defined(__aarch64__)
int diff = vaddvq_s32(v_diff);
uint32_t sq = (uint32_t)vaddvq_u64(vreinterpretq_u64_s64(v_sse));
#else
int diff = horizontal_add_s32x4(v_diff);
uint32_t sq = vget_lane_u32(
vreinterpret_u32_s64(vadd_s64(vget_low_s64(v_sse), vget_high_s64(v_sse))),
0);
#endif
*sum = diff;
*sse = sq;
}
#define VARIANCE_NXM_WIDE(W, H) \
unsigned int aom_variance##W##x##H##_neon(const uint8_t *a, int a_stride, \
const uint8_t *b, int b_stride, \
uint32_t *sse) { \
int sum; \
variance_neon_wide_block(a, a_stride, b, b_stride, W, H, sse, &sum); \
return *sse - (uint32_t)(((int64_t)sum * sum) / (W * H)); \
}
VARIANCE_NXM(4, 4, 4)
VARIANCE_NXM(4, 8, 5)
VARIANCE_NXM(8, 4, 5)
VARIANCE_NXM(16, 32, 9)
VARIANCE_NXM(32, 16, 9)
VARIANCE_NXM_WIDE(128, 64)
VARIANCE_NXM_WIDE(64, 128)

View file

@ -9,6 +9,7 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include <assert.h>
#include <stdlib.h>
#include "config/aom_dsp_rtcd.h"
@ -48,6 +49,16 @@ unsigned int aom_avg_8x8_c(const uint8_t *s, int p) {
return (sum + 32) >> 6;
}
void aom_avg_8x8_quad_c(const uint8_t *s, int p, int x16_idx, int y16_idx,
int *avg) {
for (int k = 0; k < 4; k++) {
const int x8_idx = x16_idx + ((k & 1) << 3);
const int y8_idx = y16_idx + ((k >> 1) << 3);
const uint8_t *s_tmp = s + y8_idx * p + x8_idx;
avg[k] = aom_avg_8x8_c(s_tmp, p);
}
}
#if CONFIG_AV1_HIGHBITDEPTH
unsigned int aom_highbd_avg_8x8_c(const uint8_t *s8, int p) {
int i, j;
@ -88,6 +99,52 @@ void aom_highbd_minmax_8x8_c(const uint8_t *s8, int p, const uint8_t *d8,
}
#endif // CONFIG_AV1_HIGHBITDEPTH
void aom_pixel_scale_c(const int16_t *src_diff, ptrdiff_t src_stride,
int16_t *coeff, int log_scale, int h8, int w8) {
for (int idy = 0; idy < h8 * 8; ++idy)
for (int idx = 0; idx < w8 * 8; ++idx)
coeff[idy * (h8 * 8) + idx] = src_diff[idy * src_stride + idx]
<< log_scale;
}
static void hadamard_col4(const int16_t *src_diff, ptrdiff_t src_stride,
int16_t *coeff) {
int16_t b0 = (src_diff[0 * src_stride] + src_diff[1 * src_stride]) >> 1;
int16_t b1 = (src_diff[0 * src_stride] - src_diff[1 * src_stride]) >> 1;
int16_t b2 = (src_diff[2 * src_stride] + src_diff[3 * src_stride]) >> 1;
int16_t b3 = (src_diff[2 * src_stride] - src_diff[3 * src_stride]) >> 1;
coeff[0] = b0 + b2;
coeff[1] = b1 + b3;
coeff[2] = b0 - b2;
coeff[3] = b1 - b3;
}
void aom_hadamard_4x4_c(const int16_t *src_diff, ptrdiff_t src_stride,
tran_low_t *coeff) {
int idx;
int16_t buffer[16];
int16_t buffer2[16];
int16_t *tmp_buf = &buffer[0];
for (idx = 0; idx < 4; ++idx) {
hadamard_col4(src_diff, src_stride, tmp_buf); // src_diff: 9 bit
// dynamic range [-255, 255]
tmp_buf += 4;
++src_diff;
}
tmp_buf = &buffer[0];
for (idx = 0; idx < 4; ++idx) {
hadamard_col4(tmp_buf, 4, buffer2 + 4 * idx); // tmp_buf: 12 bit
// dynamic range [-2040, 2040]
// buffer2: 15 bit
// dynamic range [-16320, 16320]
++tmp_buf;
}
for (idx = 0; idx < 16; ++idx) coeff[idx] = (tran_low_t)buffer2[idx];
}
// src_diff: first pass, 9 bit, dynamic range [-255, 255]
// second pass, 12 bit, dynamic range [-2040, 2040]
static void hadamard_col8(const int16_t *src_diff, ptrdiff_t src_stride,
@ -171,6 +228,14 @@ void aom_hadamard_lp_8x8_c(const int16_t *src_diff, ptrdiff_t src_stride,
for (int idx = 0; idx < 64; ++idx) coeff[idx] = buffer2[idx];
}
void aom_hadamard_8x8_dual_c(const int16_t *src_diff, ptrdiff_t src_stride,
int16_t *coeff) {
for (int i = 0; i < 2; i++) {
aom_hadamard_lp_8x8_c(src_diff + (i * 8), src_stride,
(int16_t *)coeff + (i * 64));
}
}
// In place 16x16 2D Hadamard transform
void aom_hadamard_16x16_c(const int16_t *src_diff, ptrdiff_t src_stride,
tran_low_t *coeff) {
@ -446,6 +511,7 @@ void aom_int_pro_row_c(int16_t hbuf[16], const uint8_t *ref,
const int ref_stride, const int height) {
int idx;
const int norm_factor = height >> 1;
assert(height >= 2);
for (idx = 0; idx < 16; ++idx) {
int i;
hbuf[idx] = 0;

View file

@ -11,7 +11,6 @@
#include "aom_dsp/binary_codes_reader.h"
#include "aom_dsp/recenter.h"
#include "av1/common/common.h"
uint16_t aom_read_primitive_quniform_(aom_reader *r,
uint16_t n ACCT_STR_PARAM) {

View file

@ -13,7 +13,6 @@
#include "aom_dsp/binary_codes_writer.h"
#include "aom_dsp/recenter.h"
#include "aom_ports/bitops.h"
#include "av1/common/common.h"
// Codes a symbol v in [-2^mag_bits, 2^mag_bits].
// mag_bits is number of bits for magnitude. The alphabet is of size

View file

@ -20,8 +20,12 @@
#include "aom/aomdx.h"
#include "aom/aom_integer.h"
#include "aom_dsp/entdec.h"
#include "aom_dsp/odintrin.h"
#include "aom_dsp/prob.h"
#include "av1/common/odintrin.h"
#if CONFIG_BITSTREAM_DEBUG
#include "aom_util/debug_util.h"
#endif // CONFIG_BITSTREAM_DEBUG
#if CONFIG_ACCOUNTING
#include "av1/decoder/accounting.h"

View file

@ -29,3 +29,8 @@ int aom_stop_encode(aom_writer *w) {
od_ec_enc_clear(&w->ec);
return nb_bits;
}
int aom_tell_size(aom_writer *w) {
const int nb_bits = od_ec_enc_tell(&w->ec);
return nb_bits;
}

View file

@ -24,6 +24,10 @@
#include "av1/encoder/cost.h"
#endif
#if CONFIG_BITSTREAM_DEBUG
#include "aom_util/debug_util.h"
#endif // CONFIG_BITSTREAM_DEBUG
#ifdef __cplusplus
extern "C" {
#endif
@ -60,18 +64,12 @@ void aom_start_encode(aom_writer *w, uint8_t *buffer);
int aom_stop_encode(aom_writer *w);
int aom_tell_size(aom_writer *w);
static INLINE void aom_write(aom_writer *w, int bit, int probability) {
int p = (0x7FFFFF - (probability << 15) + probability) >> 8;
#if CONFIG_BITSTREAM_DEBUG
aom_cdf_prob cdf[2] = { (aom_cdf_prob)p, 32767 };
/*int queue_r = 0;
int frame_idx_r = 0;
int queue_w = bitstream_queue_get_write();
int frame_idx_w = aom_bitstream_queue_get_frame_writee();
if (frame_idx_w == frame_idx_r && queue_w == queue_r) {
fprintf(stderr, "\n *** bitstream queue at frame_idx_w %d queue_w %d\n",
frame_idx_w, queue_w);
}*/
bitstream_queue_push(bit, cdf, 2);
#endif
@ -91,14 +89,6 @@ static INLINE void aom_write_literal(aom_writer *w, int data, int bits) {
static INLINE void aom_write_cdf(aom_writer *w, int symb,
const aom_cdf_prob *cdf, int nsymbs) {
#if CONFIG_BITSTREAM_DEBUG
/*int queue_r = 0;
int frame_idx_r = 0;
int queue_w = bitstream_queue_get_write();
int frame_idx_w = aom_bitstream_queue_get_frame_writee();
if (frame_idx_w == frame_idx_r && queue_w == queue_r) {
fprintf(stderr, "\n *** bitstream queue at frame_idx_w %d queue_w %d\n",
frame_idx_w, queue_w);
}*/
bitstream_queue_push(symb, cdf, nsymbs);
#endif

View file

@ -22,7 +22,7 @@
// as described for AOM_BLEND_A64 in aom_dsp/blend.h. src0 or src1 can
// be the same as dst, or dst can be different from both sources.
// NOTE(david.barker): The input and output of aom_blend_a64_d16_mask_c() are
// NOTE(rachelbarker): The input and output of aom_blend_a64_d16_mask_c() are
// in a higher intermediate precision, and will later be rounded down to pixel
// precision.
// Thus, in order to avoid double-rounding, we want to use normal right shifts

View file

@ -0,0 +1,109 @@
/*
* Copyright (c) 2021, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include <assert.h>
#include <jxl/butteraugli.h>
#include "aom_dsp/butteraugli.h"
#include "aom_mem/aom_mem.h"
#include "third_party/libyuv/include/libyuv/convert_argb.h"
int aom_calc_butteraugli(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *distorted, int bit_depth,
aom_matrix_coefficients_t matrix_coefficients,
aom_color_range_t color_range, float *dist_map) {
(void)bit_depth;
assert(bit_depth == 8);
const int width = source->y_crop_width;
const int height = source->y_crop_height;
const int ss_x = source->subsampling_x;
const int ss_y = source->subsampling_y;
const struct YuvConstants *yuv_constants;
if (matrix_coefficients == AOM_CICP_MC_BT_709) {
if (color_range == AOM_CR_FULL_RANGE) return 0;
yuv_constants = &kYuvH709Constants;
} else {
yuv_constants = color_range == AOM_CR_FULL_RANGE ? &kYuvJPEGConstants
: &kYuvI601Constants;
}
const int stride_argb = width * 4;
const size_t buffer_size = (size_t)height * stride_argb;
uint8_t *src_argb = (uint8_t *)aom_malloc(buffer_size);
uint8_t *distorted_argb = (uint8_t *)aom_malloc(buffer_size);
if (!src_argb || !distorted_argb) {
aom_free(src_argb);
aom_free(distorted_argb);
return 0;
}
if (ss_x == 1 && ss_y == 1) {
I420ToARGBMatrix(source->y_buffer, source->y_stride, source->u_buffer,
source->uv_stride, source->v_buffer, source->uv_stride,
src_argb, stride_argb, yuv_constants, width, height);
I420ToARGBMatrix(distorted->y_buffer, distorted->y_stride,
distorted->u_buffer, distorted->uv_stride,
distorted->v_buffer, distorted->uv_stride, distorted_argb,
stride_argb, yuv_constants, width, height);
} else if (ss_x == 1 && ss_y == 0) {
I422ToARGBMatrix(source->y_buffer, source->y_stride, source->u_buffer,
source->uv_stride, source->v_buffer, source->uv_stride,
src_argb, stride_argb, yuv_constants, width, height);
I422ToARGBMatrix(distorted->y_buffer, distorted->y_stride,
distorted->u_buffer, distorted->uv_stride,
distorted->v_buffer, distorted->uv_stride, distorted_argb,
stride_argb, yuv_constants, width, height);
} else if (ss_x == 0 && ss_y == 0) {
I444ToARGBMatrix(source->y_buffer, source->y_stride, source->u_buffer,
source->uv_stride, source->v_buffer, source->uv_stride,
src_argb, stride_argb, yuv_constants, width, height);
I444ToARGBMatrix(distorted->y_buffer, distorted->y_stride,
distorted->u_buffer, distorted->uv_stride,
distorted->v_buffer, distorted->uv_stride, distorted_argb,
stride_argb, yuv_constants, width, height);
} else {
aom_free(src_argb);
aom_free(distorted_argb);
return 0;
}
JxlPixelFormat pixel_format = { 4, JXL_TYPE_UINT8, JXL_NATIVE_ENDIAN, 0 };
JxlButteraugliApi *api = JxlButteraugliApiCreate(NULL);
JxlButteraugliApiSetHFAsymmetry(api, 0.8f);
JxlButteraugliResult *result = JxlButteraugliCompute(
api, width, height, &pixel_format, src_argb, buffer_size, &pixel_format,
distorted_argb, buffer_size);
const float *distmap = NULL;
uint32_t row_stride;
JxlButteraugliResultGetDistmap(result, &distmap, &row_stride);
if (distmap == NULL) {
JxlButteraugliApiDestroy(api);
JxlButteraugliResultDestroy(result);
aom_free(src_argb);
aom_free(distorted_argb);
return 0;
}
for (int j = 0; j < height; ++j) {
for (int i = 0; i < width; ++i) {
dist_map[j * width + i] = distmap[j * row_stride + i];
}
}
JxlButteraugliApiDestroy(api);
JxlButteraugliResultDestroy(result);
aom_free(src_argb);
aom_free(distorted_argb);
return 1;
}

View file

@ -0,0 +1,23 @@
/*
* Copyright (c) 2021, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AOM_AOM_DSP_BUTTERAUGLI_H_
#define AOM_AOM_DSP_BUTTERAUGLI_H_
#include "aom_scale/yv12config.h"
// Returns a boolean that indicates success/failure.
int aom_calc_butteraugli(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *distorted, int bit_depth,
aom_matrix_coefficients_t matrix_coefficients,
aom_color_range_t color_range, float *dist_map);
#endif // AOM_AOM_DSP_BUTTERAUGLI_H_

View file

@ -14,7 +14,7 @@
#include <limits.h>
#include <stddef.h>
#include "av1/common/odintrin.h"
#include "aom_dsp/odintrin.h"
#include "aom_dsp/prob.h"
#define EC_PROB_SHIFT 6

View file

@ -20,7 +20,6 @@
#include "config/aom_dsp_rtcd.h"
#include "aom_dsp/ssim.h"
#include "aom_ports/system_state.h"
typedef struct fs_level fs_level;
typedef struct fs_ctx fs_ctx;
@ -31,6 +30,7 @@ typedef struct fs_ctx fs_ctx;
#define SSIM_C1_12 (4095 * 4095 * 0.01 * 0.01)
#define SSIM_C2_10 (1023 * 1023 * 0.03 * 0.03)
#define SSIM_C2_12 (4095 * 4095 * 0.03 * 0.03)
#define MAX_SSIM_DB 100.0
#define FS_MINI(_a, _b) ((_a) < (_b) ? (_a) : (_b))
#define FS_MAXI(_a, _b) ((_a) > (_b) ? (_a) : (_b))
@ -49,7 +49,7 @@ struct fs_ctx {
unsigned *col_buf;
};
static void fs_ctx_init(fs_ctx *_ctx, int _w, int _h, int _nlevels) {
static int fs_ctx_init(fs_ctx *_ctx, int _w, int _h, int _nlevels) {
unsigned char *data;
size_t data_size;
int lw;
@ -73,6 +73,7 @@ static void fs_ctx_init(fs_ctx *_ctx, int _w, int _h, int _nlevels) {
lh = (lh + 1) >> 1;
}
data = (unsigned char *)malloc(data_size);
if (!data) return -1;
_ctx->level = (fs_level *)data;
_ctx->nlevels = _nlevels;
data += _nlevels * sizeof(*_ctx->level);
@ -97,6 +98,7 @@ static void fs_ctx_init(fs_ctx *_ctx, int _w, int _h, int _nlevels) {
lh = (lh + 1) >> 1;
}
_ctx->col_buf = (unsigned *)data;
return 0;
}
static void fs_ctx_clear(fs_ctx *_ctx) { free(_ctx->level); }
@ -446,7 +448,7 @@ static double calc_ssim(const uint8_t *_src, int _systride, const uint8_t *_dst,
double ret;
int l;
ret = 1;
fs_ctx_init(&ctx, _w, _h, FS_NLEVELS);
if (fs_ctx_init(&ctx, _w, _h, FS_NLEVELS)) return 99.0;
fs_downsample_level0(&ctx, _src, _systride, _dst, _dystride, _w, _h, _shift,
buf_is_hbd);
for (l = 0; l < FS_NLEVELS - 1; l++) {
@ -467,7 +469,6 @@ double aom_calc_fastssim(const YV12_BUFFER_CONFIG *source,
uint32_t in_bd) {
double ssimv;
uint32_t bd_shift = 0;
aom_clear_system_state();
assert(bd >= in_bd);
assert(source->flags == dest->flags);
int buf_is_hbd = source->flags & YV12_FLAG_HIGHBITDEPTH;

View file

@ -76,15 +76,15 @@ static INLINE float add_float(float a, float b) { return a + b; }
static INLINE float sub_float(float a, float b) { return a - b; }
static INLINE float mul_float(float a, float b) { return a * b; }
GEN_FFT_2(void, float, float, float, *, store_float);
GEN_FFT_2(void, float, float, float, *, store_float)
GEN_FFT_4(void, float, float, float, *, store_float, (float), add_float,
sub_float);
sub_float)
GEN_FFT_8(void, float, float, float, *, store_float, (float), add_float,
sub_float, mul_float);
sub_float, mul_float)
GEN_FFT_16(void, float, float, float, *, store_float, (float), add_float,
sub_float, mul_float);
sub_float, mul_float)
GEN_FFT_32(void, float, float, float, *, store_float, (float), add_float,
sub_float, mul_float);
sub_float, mul_float)
void aom_fft2x2_float_c(const float *input, float *temp, float *output) {
aom_fft_2d_gen(input, temp, output, 2, aom_fft1d_2_float, simple_transpose,
@ -183,15 +183,15 @@ void aom_ifft_2d_gen(const float *input, float *temp, float *output, int n,
transpose(temp, output, n);
}
GEN_IFFT_2(void, float, float, float, *, store_float);
GEN_IFFT_2(void, float, float, float, *, store_float)
GEN_IFFT_4(void, float, float, float, *, store_float, (float), add_float,
sub_float);
sub_float)
GEN_IFFT_8(void, float, float, float, *, store_float, (float), add_float,
sub_float, mul_float);
sub_float, mul_float)
GEN_IFFT_16(void, float, float, float, *, store_float, (float), add_float,
sub_float, mul_float);
sub_float, mul_float)
GEN_IFFT_32(void, float, float, float, *, store_float, (float), add_float,
sub_float, mul_float);
sub_float, mul_float)
void aom_ifft2x2_float_c(const float *input, float *temp, float *output) {
aom_ifft_2d_gen(input, temp, output, 2, aom_fft1d_2_float, aom_fft1d_2_float,

View file

@ -10,20 +10,20 @@
*/
/*!\file
* \brief Describes film grain parameters and film grain synthesis
* \brief Describes film grain parameters
*
*/
#ifndef AOM_AOM_DSP_GRAIN_SYNTHESIS_H_
#define AOM_AOM_DSP_GRAIN_SYNTHESIS_H_
#ifndef AOM_AOM_DSP_GRAIN_PARAMS_H_
#define AOM_AOM_DSP_GRAIN_PARAMS_H_
#ifdef __cplusplus
extern "C" {
#endif
#include <stdint.h>
#include <string.h>
#include "aom_dsp/aom_dsp_common.h"
#include "aom/aom_image.h"
#include "config/aom_config.h"
/*!\brief Structure containing film grain synthesis parameters for a frame
*
@ -31,7 +31,7 @@ extern "C" {
*/
typedef struct {
// This structure is compared element-by-element in the function
// av1_check_grain_params_equiv: this function must be updated if any changes
// aom_check_grain_params_equiv: this function must be updated if any changes
// are made to this structure.
int apply_grain;
@ -85,7 +85,7 @@ typedef struct {
uint16_t random_seed;
// This structure is compared element-by-element in the function
// av1_check_grain_params_equiv: this function must be updated if any changes
// aom_check_grain_params_equiv: this function must be updated if any changes
// are made to this structure.
} aom_film_grain_t;
@ -98,7 +98,7 @@ typedef struct {
* \param[in] pb The second set of parameters to compare
* \return Returns 1 if the params are equivalent, 0 otherwise
*/
static INLINE int av1_check_grain_params_equiv(
static INLINE int aom_check_grain_params_equiv(
const aom_film_grain_t *const pa, const aom_film_grain_t *const pb) {
if (pa->apply_grain != pb->apply_grain) return 0;
// Don't compare update_parameters
@ -151,42 +151,8 @@ static INLINE int av1_check_grain_params_equiv(
return 1;
}
/*!\brief Add film grain
*
* Add film grain to an image
*
* Returns 0 for success, -1 for failure
*
* \param[in] grain_params Grain parameters
* \param[in] luma luma plane
* \param[in] cb cb plane
* \param[in] cr cr plane
* \param[in] height luma plane height
* \param[in] width luma plane width
* \param[in] luma_stride luma plane stride
* \param[in] chroma_stride chroma plane stride
*/
int av1_add_film_grain_run(const aom_film_grain_t *grain_params, uint8_t *luma,
uint8_t *cb, uint8_t *cr, int height, int width,
int luma_stride, int chroma_stride,
int use_high_bit_depth, int chroma_subsamp_y,
int chroma_subsamp_x, int mc_identity);
/*!\brief Add film grain
*
* Add film grain to an image
*
* Returns 0 for success, -1 for failure
*
* \param[in] grain_params Grain parameters
* \param[in] src Source image
* \param[out] dst Resulting image with grain
*/
int av1_add_film_grain(const aom_film_grain_t *grain_params,
const aom_image_t *src, aom_image_t *dst);
#ifdef __cplusplus
} // extern "C"
#endif
#endif // AOM_AOM_DSP_GRAIN_SYNTHESIS_H_
#endif // AOM_AOM_DSP_GRAIN_PARAMS_H_

View file

@ -105,7 +105,11 @@ static void grain_table_entry_read(FILE *file,
}
}
fscanf(file, "\n\tcY");
if (fscanf(file, "\n\tcY")) {
aom_internal_error(error_info, AOM_CODEC_ERROR,
"Unable to read Y coeffs header (cY)");
return;
}
const int n = 2 * pars->ar_coeff_lag * (pars->ar_coeff_lag + 1);
for (int i = 0; i < n; ++i) {
if (1 != fscanf(file, "%d", &pars->ar_coeffs_y[i])) {
@ -114,7 +118,11 @@ static void grain_table_entry_read(FILE *file,
return;
}
}
fscanf(file, "\n\tcCb");
if (fscanf(file, "\n\tcCb")) {
aom_internal_error(error_info, AOM_CODEC_ERROR,
"Unable to read Cb coeffs header (cCb)");
return;
}
for (int i = 0; i <= n; ++i) {
if (1 != fscanf(file, "%d", &pars->ar_coeffs_cb[i])) {
aom_internal_error(error_info, AOM_CODEC_ERROR,
@ -122,7 +130,11 @@ static void grain_table_entry_read(FILE *file,
return;
}
}
fscanf(file, "\n\tcCr");
if (fscanf(file, "\n\tcCr")) {
aom_internal_error(error_info, AOM_CODEC_ERROR,
"Unable read to Cr coeffs header (cCr)");
return;
}
for (int i = 0; i <= n; ++i) {
if (1 != fscanf(file, "%d", &pars->ar_coeffs_cr[i])) {
aom_internal_error(error_info, AOM_CODEC_ERROR,
@ -130,7 +142,7 @@ static void grain_table_entry_read(FILE *file,
return;
}
}
fscanf(file, "\n");
(void)fscanf(file, "\n");
}
}
@ -179,11 +191,14 @@ static void grain_table_entry_write(FILE *file,
}
}
// TODO(https://crbug.com/aomedia/3228): Update this function to return an
// integer status.
void aom_film_grain_table_append(aom_film_grain_table_t *t, int64_t time_stamp,
int64_t end_time,
const aom_film_grain_t *grain) {
if (!t->tail || memcmp(grain, &t->tail->params, sizeof(*grain))) {
aom_film_grain_table_entry_t *new_tail = aom_malloc(sizeof(*new_tail));
if (!new_tail) return;
memset(new_tail, 0, sizeof(*new_tail));
if (t->tail) t->tail->next = new_tail;
if (!t->head) t->head = new_tail;
@ -202,7 +217,7 @@ int aom_film_grain_table_lookup(aom_film_grain_table_t *t, int64_t time_stamp,
int64_t end_time, int erase,
aom_film_grain_t *grain) {
aom_film_grain_table_entry_t *entry = t->head;
aom_film_grain_table_entry_t *prev_entry = 0;
aom_film_grain_table_entry_t *prev_entry = NULL;
uint16_t random_seed = grain ? grain->random_seed : 0;
if (grain) memset(grain, 0, sizeof(*grain));
@ -233,6 +248,7 @@ int aom_film_grain_table_lookup(aom_film_grain_table_t *t, int64_t time_stamp,
} else {
aom_film_grain_table_entry_t *new_entry =
aom_malloc(sizeof(*new_entry));
if (!new_entry) return 0;
new_entry->next = entry->next;
new_entry->start_time = end_time;
new_entry->end_time = entry->end_time;
@ -241,10 +257,13 @@ int aom_film_grain_table_lookup(aom_film_grain_table_t *t, int64_t time_stamp,
entry->end_time = time_stamp;
if (t->tail == entry) t->tail = new_entry;
}
// If segments aren't aligned, delete from the beggining of subsequent
// If segments aren't aligned, delete from the beginning of subsequent
// segments
if (end_time > entry_end_time) {
aom_film_grain_table_lookup(t, entry->end_time, end_time, 1, 0);
// Ignoring the return value here is safe since we're erasing from the
// beginning of subsequent entries.
aom_film_grain_table_lookup(t, entry_end_time, end_time, /*erase=*/1,
NULL);
}
return 1;
}
@ -275,12 +294,17 @@ aom_codec_err_t aom_film_grain_table_read(
return error_info->error_code;
}
aom_film_grain_table_entry_t *prev_entry = 0;
aom_film_grain_table_entry_t *prev_entry = NULL;
while (!feof(file)) {
aom_film_grain_table_entry_t *entry = aom_malloc(sizeof(*entry));
if (!entry) {
aom_internal_error(error_info, AOM_CODEC_MEM_ERROR,
"Unable to allocate grain table entry");
break;
}
memset(entry, 0, sizeof(*entry));
grain_table_entry_read(file, error_info, entry);
entry->next = 0;
entry->next = NULL;
if (prev_entry) prev_entry->next = entry;
if (!t->head) t->head = entry;

View file

@ -34,7 +34,7 @@
extern "C" {
#endif
#include "aom_dsp/grain_synthesis.h"
#include "aom_dsp/grain_params.h"
#include "aom/internal/aom_codec_internal.h"
typedef struct aom_film_grain_table_entry_t {

View file

@ -86,11 +86,11 @@ static INLINE void smooth_predictor(uint8_t *dst, ptrdiff_t stride, int bw,
const uint8_t *left) {
const uint8_t below_pred = left[bh - 1]; // estimated by bottom-left pixel
const uint8_t right_pred = above[bw - 1]; // estimated by top-right pixel
const uint8_t *const sm_weights_w = sm_weight_arrays + bw;
const uint8_t *const sm_weights_h = sm_weight_arrays + bh;
// scale = 2 * 2^sm_weight_log2_scale
const int log2_scale = 1 + sm_weight_log2_scale;
const uint16_t scale = (1 << sm_weight_log2_scale);
const uint8_t *const sm_weights_w = smooth_weights + bw - 4;
const uint8_t *const sm_weights_h = smooth_weights + bh - 4;
// scale = 2 * 2^SMOOTH_WEIGHT_LOG2_SCALE
const int log2_scale = 1 + SMOOTH_WEIGHT_LOG2_SCALE;
const uint16_t scale = (1 << SMOOTH_WEIGHT_LOG2_SCALE);
sm_weights_sanity_checks(sm_weights_w, sm_weights_h, scale,
log2_scale + sizeof(*dst));
int r;
@ -116,10 +116,10 @@ static INLINE void smooth_v_predictor(uint8_t *dst, ptrdiff_t stride, int bw,
int bh, const uint8_t *above,
const uint8_t *left) {
const uint8_t below_pred = left[bh - 1]; // estimated by bottom-left pixel
const uint8_t *const sm_weights = sm_weight_arrays + bh;
// scale = 2^sm_weight_log2_scale
const int log2_scale = sm_weight_log2_scale;
const uint16_t scale = (1 << sm_weight_log2_scale);
const uint8_t *const sm_weights = smooth_weights + bh - 4;
// scale = 2^SMOOTH_WEIGHT_LOG2_SCALE
const int log2_scale = SMOOTH_WEIGHT_LOG2_SCALE;
const uint16_t scale = (1 << SMOOTH_WEIGHT_LOG2_SCALE);
sm_weights_sanity_checks(sm_weights, sm_weights, scale,
log2_scale + sizeof(*dst));
@ -145,10 +145,10 @@ static INLINE void smooth_h_predictor(uint8_t *dst, ptrdiff_t stride, int bw,
int bh, const uint8_t *above,
const uint8_t *left) {
const uint8_t right_pred = above[bw - 1]; // estimated by top-right pixel
const uint8_t *const sm_weights = sm_weight_arrays + bw;
// scale = 2^sm_weight_log2_scale
const int log2_scale = sm_weight_log2_scale;
const uint16_t scale = (1 << sm_weight_log2_scale);
const uint8_t *const sm_weights = smooth_weights + bw - 4;
// scale = 2^SMOOTH_WEIGHT_LOG2_SCALE
const int log2_scale = SMOOTH_WEIGHT_LOG2_SCALE;
const uint16_t scale = (1 << SMOOTH_WEIGHT_LOG2_SCALE);
sm_weights_sanity_checks(sm_weights, sm_weights, scale,
log2_scale + sizeof(*dst));
@ -405,11 +405,11 @@ static INLINE void highbd_smooth_predictor(uint16_t *dst, ptrdiff_t stride,
(void)bd;
const uint16_t below_pred = left[bh - 1]; // estimated by bottom-left pixel
const uint16_t right_pred = above[bw - 1]; // estimated by top-right pixel
const uint8_t *const sm_weights_w = sm_weight_arrays + bw;
const uint8_t *const sm_weights_h = sm_weight_arrays + bh;
// scale = 2 * 2^sm_weight_log2_scale
const int log2_scale = 1 + sm_weight_log2_scale;
const uint16_t scale = (1 << sm_weight_log2_scale);
const uint8_t *const sm_weights_w = smooth_weights + bw - 4;
const uint8_t *const sm_weights_h = smooth_weights + bh - 4;
// scale = 2 * 2^SMOOTH_WEIGHT_LOG2_SCALE
const int log2_scale = 1 + SMOOTH_WEIGHT_LOG2_SCALE;
const uint16_t scale = (1 << SMOOTH_WEIGHT_LOG2_SCALE);
sm_weights_sanity_checks(sm_weights_w, sm_weights_h, scale,
log2_scale + sizeof(*dst));
int r;
@ -437,10 +437,10 @@ static INLINE void highbd_smooth_v_predictor(uint16_t *dst, ptrdiff_t stride,
const uint16_t *left, int bd) {
(void)bd;
const uint16_t below_pred = left[bh - 1]; // estimated by bottom-left pixel
const uint8_t *const sm_weights = sm_weight_arrays + bh;
// scale = 2^sm_weight_log2_scale
const int log2_scale = sm_weight_log2_scale;
const uint16_t scale = (1 << sm_weight_log2_scale);
const uint8_t *const sm_weights = smooth_weights + bh - 4;
// scale = 2^SMOOTH_WEIGHT_LOG2_SCALE
const int log2_scale = SMOOTH_WEIGHT_LOG2_SCALE;
const uint16_t scale = (1 << SMOOTH_WEIGHT_LOG2_SCALE);
sm_weights_sanity_checks(sm_weights, sm_weights, scale,
log2_scale + sizeof(*dst));
@ -468,10 +468,10 @@ static INLINE void highbd_smooth_h_predictor(uint16_t *dst, ptrdiff_t stride,
const uint16_t *left, int bd) {
(void)bd;
const uint16_t right_pred = above[bw - 1]; // estimated by top-right pixel
const uint8_t *const sm_weights = sm_weight_arrays + bw;
// scale = 2^sm_weight_log2_scale
const int log2_scale = sm_weight_log2_scale;
const uint16_t scale = (1 << sm_weight_log2_scale);
const uint8_t *const sm_weights = smooth_weights + bw - 4;
// scale = 2^SMOOTH_WEIGHT_LOG2_SCALE
const int log2_scale = SMOOTH_WEIGHT_LOG2_SCALE;
const uint16_t scale = (1 << SMOOTH_WEIGHT_LOG2_SCALE);
sm_weights_sanity_checks(sm_weights, sm_weights, scale,
log2_scale + sizeof(*dst));
@ -752,6 +752,7 @@ void aom_highbd_dc_predictor_64x32_c(uint16_t *dst, ptrdiff_t stride,
intra_pred_highbd_sized(type, 32, 8) \
intra_pred_highbd_sized(type, 16, 64) \
intra_pred_highbd_sized(type, 64, 16)
#define intra_pred_above_4x4(type) \
intra_pred_sized(type, 8, 8) \
intra_pred_sized(type, 16, 16) \

View file

@ -15,18 +15,14 @@
#include "config/aom_config.h"
// Weights are quadratic from '1' to '1 / block_size', scaled by
// 2^sm_weight_log2_scale.
static const int sm_weight_log2_scale = 8;
// 2^SMOOTH_WEIGHT_LOG2_SCALE.
#define SMOOTH_WEIGHT_LOG2_SCALE 8
// max(block_size_wide[BLOCK_LARGEST], block_size_high[BLOCK_LARGEST])
#define MAX_BLOCK_DIM 64
/* clang-format off */
static const uint8_t sm_weight_arrays[2 * MAX_BLOCK_DIM] = {
// Unused, because we always offset by bs, which is at least 2.
0, 0,
// bs = 2
255, 128,
// Note these arrays are aligned to ensure NEON loads using a cast to uint32_t*
// have sufficient alignment. Using 8 preserves the potential for an alignment
// hint in load_weight_w8(). For that case, this could be increased to 16 to
// allow an aligned load in x86.
DECLARE_ALIGNED(8, static const uint8_t, smooth_weights[]) = {
// bs = 4
255, 149, 85, 64,
// bs = 8
@ -40,8 +36,24 @@ static const uint8_t sm_weight_arrays[2 * MAX_BLOCK_DIM] = {
255, 248, 240, 233, 225, 218, 210, 203, 196, 189, 182, 176, 169, 163, 156,
150, 144, 138, 133, 127, 121, 116, 111, 106, 101, 96, 91, 86, 82, 77, 73, 69,
65, 61, 57, 54, 50, 47, 44, 41, 38, 35, 32, 29, 27, 25, 22, 20, 18, 16, 15,
13, 12, 10, 9, 8, 7, 6, 6, 5, 5, 4, 4, 4,
13, 12, 10, 9, 8, 7, 6, 6, 5, 5, 4, 4, 4
};
DECLARE_ALIGNED(8, static const uint16_t, smooth_weights_u16[]) = {
// block dimension = 4
255, 149, 85, 64,
// block dimension = 8
255, 197, 146, 105, 73, 50, 37, 32,
// block dimension = 16
255, 225, 196, 170, 145, 123, 102, 84, 68, 54, 43, 33, 26, 20, 17, 16,
// block dimension = 32
255, 240, 225, 210, 196, 182, 169, 157, 145, 133, 122, 111, 101, 92, 83, 74,
66, 59, 52, 45, 39, 34, 29, 25, 21, 17, 14, 12, 10, 9, 8, 8,
// block dimension = 64
255, 248, 240, 233, 225, 218, 210, 203, 196, 189, 182, 176, 169, 163, 156,
150, 144, 138, 133, 127, 121, 116, 111, 106, 101, 96, 91, 86, 82, 77, 73, 69,
65, 61, 57, 54, 50, 47, 44, 41, 38, 35, 32, 29, 27, 25, 22, 20, 18, 16, 15,
13, 12, 10, 9, 8, 7, 6, 6, 5, 5, 4, 4, 4
};
/* clang-format on */
#endif // AOM_AOM_DSP_INTRAPRED_COMMON_H_

View file

@ -158,6 +158,15 @@ void aom_lpf_horizontal_4_dual_c(uint8_t *s, int p, const uint8_t *blimit0,
aom_lpf_horizontal_4_c(s + 4, p, blimit1, limit1, thresh1);
}
void aom_lpf_horizontal_4_quad_c(uint8_t *s, int p, const uint8_t *blimit0,
const uint8_t *limit0,
const uint8_t *thresh0) {
aom_lpf_horizontal_4_c(s, p, blimit0, limit0, thresh0);
aom_lpf_horizontal_4_c(s + 4, p, blimit0, limit0, thresh0);
aom_lpf_horizontal_4_c(s + 8, p, blimit0, limit0, thresh0);
aom_lpf_horizontal_4_c(s + 12, p, blimit0, limit0, thresh0);
}
void aom_lpf_vertical_4_c(uint8_t *s, int pitch, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
int i;
@ -182,6 +191,14 @@ void aom_lpf_vertical_4_dual_c(uint8_t *s, int pitch, const uint8_t *blimit0,
aom_lpf_vertical_4_c(s + 4 * pitch, pitch, blimit1, limit1, thresh1);
}
void aom_lpf_vertical_4_quad_c(uint8_t *s, int pitch, const uint8_t *blimit0,
const uint8_t *limit0, const uint8_t *thresh0) {
aom_lpf_vertical_4_c(s, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_4_c(s + 4 * pitch, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_4_c(s + 8 * pitch, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_4_c(s + 12 * pitch, pitch, blimit0, limit0, thresh0);
}
static INLINE void filter6(int8_t mask, uint8_t thresh, int8_t flat,
uint8_t *op2, uint8_t *op1, uint8_t *op0,
uint8_t *oq0, uint8_t *oq1, uint8_t *oq2) {
@ -247,6 +264,15 @@ void aom_lpf_horizontal_6_dual_c(uint8_t *s, int p, const uint8_t *blimit0,
aom_lpf_horizontal_6_c(s + 4, p, blimit1, limit1, thresh1);
}
void aom_lpf_horizontal_6_quad_c(uint8_t *s, int p, const uint8_t *blimit0,
const uint8_t *limit0,
const uint8_t *thresh0) {
aom_lpf_horizontal_6_c(s, p, blimit0, limit0, thresh0);
aom_lpf_horizontal_6_c(s + 4, p, blimit0, limit0, thresh0);
aom_lpf_horizontal_6_c(s + 8, p, blimit0, limit0, thresh0);
aom_lpf_horizontal_6_c(s + 12, p, blimit0, limit0, thresh0);
}
void aom_lpf_horizontal_8_c(uint8_t *s, int p, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
int i;
@ -275,6 +301,15 @@ void aom_lpf_horizontal_8_dual_c(uint8_t *s, int p, const uint8_t *blimit0,
aom_lpf_horizontal_8_c(s + 4, p, blimit1, limit1, thresh1);
}
void aom_lpf_horizontal_8_quad_c(uint8_t *s, int p, const uint8_t *blimit0,
const uint8_t *limit0,
const uint8_t *thresh0) {
aom_lpf_horizontal_8_c(s, p, blimit0, limit0, thresh0);
aom_lpf_horizontal_8_c(s + 4, p, blimit0, limit0, thresh0);
aom_lpf_horizontal_8_c(s + 8, p, blimit0, limit0, thresh0);
aom_lpf_horizontal_8_c(s + 12, p, blimit0, limit0, thresh0);
}
void aom_lpf_vertical_6_c(uint8_t *s, int pitch, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
int i;
@ -299,6 +334,14 @@ void aom_lpf_vertical_6_dual_c(uint8_t *s, int pitch, const uint8_t *blimit0,
aom_lpf_vertical_6_c(s + 4 * pitch, pitch, blimit1, limit1, thresh1);
}
void aom_lpf_vertical_6_quad_c(uint8_t *s, int pitch, const uint8_t *blimit0,
const uint8_t *limit0, const uint8_t *thresh0) {
aom_lpf_vertical_6_c(s, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_6_c(s + 4 * pitch, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_6_c(s + 8 * pitch, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_6_c(s + 12 * pitch, pitch, blimit0, limit0, thresh0);
}
void aom_lpf_vertical_8_c(uint8_t *s, int pitch, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh) {
int i;
@ -324,6 +367,14 @@ void aom_lpf_vertical_8_dual_c(uint8_t *s, int pitch, const uint8_t *blimit0,
aom_lpf_vertical_8_c(s + 4 * pitch, pitch, blimit1, limit1, thresh1);
}
void aom_lpf_vertical_8_quad_c(uint8_t *s, int pitch, const uint8_t *blimit0,
const uint8_t *limit0, const uint8_t *thresh0) {
aom_lpf_vertical_8_c(s, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_8_c(s + 4 * pitch, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_8_c(s + 8 * pitch, pitch, blimit0, limit0, thresh0);
aom_lpf_vertical_8_c(s + 12 * pitch, pitch, blimit0, limit0, thresh0);
}
static INLINE void filter14(int8_t mask, uint8_t thresh, int8_t flat,
int8_t flat2, uint8_t *op6, uint8_t *op5,
uint8_t *op4, uint8_t *op3, uint8_t *op2,
@ -410,6 +461,15 @@ void aom_lpf_horizontal_14_dual_c(uint8_t *s, int p, const uint8_t *blimit0,
mb_lpf_horizontal_edge_w(s + 4, p, blimit1, limit1, thresh1, 1);
}
void aom_lpf_horizontal_14_quad_c(uint8_t *s, int p, const uint8_t *blimit0,
const uint8_t *limit0,
const uint8_t *thresh0) {
mb_lpf_horizontal_edge_w(s, p, blimit0, limit0, thresh0, 1);
mb_lpf_horizontal_edge_w(s + 4, p, blimit0, limit0, thresh0, 1);
mb_lpf_horizontal_edge_w(s + 8, p, blimit0, limit0, thresh0, 1);
mb_lpf_horizontal_edge_w(s + 12, p, blimit0, limit0, thresh0, 1);
}
static void mb_lpf_vertical_edge_w(uint8_t *s, int p, const uint8_t *blimit,
const uint8_t *limit, const uint8_t *thresh,
int count) {
@ -444,6 +504,14 @@ void aom_lpf_vertical_14_dual_c(uint8_t *s, int pitch, const uint8_t *blimit0,
mb_lpf_vertical_edge_w(s + 4 * pitch, pitch, blimit1, limit1, thresh1, 4);
}
void aom_lpf_vertical_14_quad_c(uint8_t *s, int pitch, const uint8_t *blimit0,
const uint8_t *limit0, const uint8_t *thresh0) {
mb_lpf_vertical_edge_w(s, pitch, blimit0, limit0, thresh0, 4);
mb_lpf_vertical_edge_w(s + 4 * pitch, pitch, blimit0, limit0, thresh0, 4);
mb_lpf_vertical_edge_w(s + 8 * pitch, pitch, blimit0, limit0, thresh0, 4);
mb_lpf_vertical_edge_w(s + 12 * pitch, pitch, blimit0, limit0, thresh0, 4);
}
#if CONFIG_AV1_HIGHBITDEPTH
// Should we apply any filter at all: 11111111 yes, 00000000 no ?
static INLINE int8_t highbd_filter_mask2(uint8_t limit, uint8_t blimit,

View file

@ -9,14 +9,15 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AOM_AV1_ENCODER_MATHUTILS_H_
#define AOM_AV1_ENCODER_MATHUTILS_H_
#ifndef AOM_AOM_DSP_MATHUTILS_H_
#define AOM_AOM_DSP_MATHUTILS_H_
#include <memory.h>
#include <math.h>
#include <stdio.h>
#include <stdlib.h>
#include <assert.h>
#include <math.h>
#include <string.h>
#include "aom_dsp/aom_dsp_common.h"
#include "aom_mem/aom_mem.h"
static const double TINY_NEAR_ZERO = 1.0E-16;
@ -69,6 +70,7 @@ static INLINE int least_squares(int n, double *A, int rows, int stride,
double *AtA, *Atb;
if (!scratch) {
scratch_ = (double *)aom_malloc(sizeof(*scratch) * n * (n + 1));
if (!scratch_) return 0;
scratch = scratch_;
}
AtA = scratch;
@ -85,7 +87,7 @@ static INLINE int least_squares(int n, double *A, int rows, int stride,
for (k = 0; k < rows; ++k) Atb[i] += A[k * stride + i] * b[k];
}
int ret = linsolve(n, AtA, n, Atb, x);
if (scratch_) aom_free(scratch_);
aom_free(scratch_);
return ret;
}
@ -114,7 +116,7 @@ static INLINE void multiply_mat(const double *m1, const double *m2, double *res,
// svdcmp
// Adopted from Numerical Recipes in C
static INLINE double sign(double a, double b) {
static INLINE double apply_sign(double a, double b) {
return ((b) >= 0 ? fabs(a) : -fabs(a));
}
@ -137,6 +139,7 @@ static INLINE int svdcmp(double **u, int m, int n, double w[], double **v) {
int flag, i, its, j, jj, k, l, nm;
double anorm, c, f, g, h, s, scale, x, y, z;
double *rv1 = (double *)aom_malloc(sizeof(*rv1) * (n + 1));
if (!rv1) return 0;
g = scale = anorm = 0.0;
for (i = 0; i < n; i++) {
l = i + 1;
@ -150,7 +153,7 @@ static INLINE int svdcmp(double **u, int m, int n, double w[], double **v) {
s += u[k][i] * u[k][i];
}
f = u[i][i];
g = -sign(sqrt(s), f);
g = -apply_sign(sqrt(s), f);
h = f * g - s;
u[i][i] = f - g;
for (j = l; j < n; j++) {
@ -171,7 +174,7 @@ static INLINE int svdcmp(double **u, int m, int n, double w[], double **v) {
s += u[i][k] * u[i][k];
}
f = u[i][l];
g = -sign(sqrt(s), f);
g = -apply_sign(sqrt(s), f);
h = f * g - s;
u[i][l] = f - g;
for (k = l; k < n; k++) rv1[k] = u[i][k] / h;
@ -269,7 +272,7 @@ static INLINE int svdcmp(double **u, int m, int n, double w[], double **v) {
h = rv1[k];
f = ((y - z) * (y + z) + (g - h) * (g + h)) / (2.0 * h * y);
g = pythag(f, 1.0);
f = ((x - z) * (x + z) + h * ((y / (f + sign(g, f))) - h)) / x;
f = ((x - z) * (x + z) + h * ((y / (f + apply_sign(g, f))) - h)) / x;
c = s = 1.0;
for (j = l; j <= nm; j++) {
i = j + 1;
@ -332,8 +335,8 @@ static INLINE int SVD(double *U, double *W, double *V, double *matx, int M,
nrV[i] = &V[i * N];
}
} else {
if (nrU) aom_free(nrU);
if (nrV) aom_free(nrV);
aom_free(nrU);
aom_free(nrV);
return 1;
}
@ -356,4 +359,4 @@ static INLINE int SVD(double *U, double *W, double *V, double *matx, int M,
return 0;
}
#endif // AOM_AV1_ENCODER_MATHUTILS_H_
#endif // AOM_AOM_DSP_MATHUTILS_H_

View file

@ -21,17 +21,9 @@
#if HAVE_DSPR2
void aom_convolve_copy_dspr2(const uint8_t *src, ptrdiff_t src_stride,
uint8_t *dst, ptrdiff_t dst_stride,
const int16_t *filter_x, int filter_x_stride,
const int16_t *filter_y, int filter_y_stride,
int w, int h) {
uint8_t *dst, ptrdiff_t dst_stride, int w, int h) {
int x, y;
(void)filter_x;
(void)filter_x_stride;
(void)filter_y;
(void)filter_y_stride;
/* prefetch data to cache memory */
prefetch_load(src);
prefetch_load(src + 32);

View file

@ -198,15 +198,8 @@ static void copy_width64_msa(const uint8_t *src, int32_t src_stride,
}
void aom_convolve_copy_msa(const uint8_t *src, ptrdiff_t src_stride,
uint8_t *dst, ptrdiff_t dst_stride,
const int16_t *filter_x, int32_t filter_x_stride,
const int16_t *filter_y, int32_t filter_y_stride,
int32_t w, int32_t h) {
(void)filter_x;
(void)filter_y;
(void)filter_x_stride;
(void)filter_y_stride;
uint8_t *dst, ptrdiff_t dst_stride, int32_t w,
int32_t h) {
switch (w) {
case 4: {
uint32_t cnt, tmp;
@ -238,7 +231,7 @@ void aom_convolve_copy_msa(const uint8_t *src, ptrdiff_t src_stride,
default: {
uint32_t cnt;
for (cnt = h; cnt--;) {
memcpy(dst, src, w);
memmove(dst, src, w);
src += src_stride;
dst += dst_stride;
}

View file

@ -162,9 +162,9 @@ static uint32_t sad_64width_msa(const uint8_t *src, int32_t src_stride,
}
static void sad_4width_x4d_msa(const uint8_t *src_ptr, int32_t src_stride,
const uint8_t *const aref_ptr[],
const uint8_t *const aref_ptr[4],
int32_t ref_stride, int32_t height,
uint32_t *sad_array) {
uint32_t sad_array[4]) {
const uint8_t *ref0_ptr, *ref1_ptr, *ref2_ptr, *ref3_ptr;
int32_t ht_cnt;
uint32_t src0, src1, src2, src3;
@ -223,9 +223,9 @@ static void sad_4width_x4d_msa(const uint8_t *src_ptr, int32_t src_stride,
}
static void sad_8width_x4d_msa(const uint8_t *src_ptr, int32_t src_stride,
const uint8_t *const aref_ptr[],
const uint8_t *const aref_ptr[4],
int32_t ref_stride, int32_t height,
uint32_t *sad_array) {
uint32_t sad_array[4]) {
int32_t ht_cnt;
const uint8_t *ref0_ptr, *ref1_ptr, *ref2_ptr, *ref3_ptr;
v16u8 src0, src1, src2, src3;
@ -274,9 +274,9 @@ static void sad_8width_x4d_msa(const uint8_t *src_ptr, int32_t src_stride,
}
static void sad_16width_x4d_msa(const uint8_t *src_ptr, int32_t src_stride,
const uint8_t *const aref_ptr[],
const uint8_t *const aref_ptr[4],
int32_t ref_stride, int32_t height,
uint32_t *sad_array) {
uint32_t sad_array[4]) {
int32_t ht_cnt;
const uint8_t *ref0_ptr, *ref1_ptr, *ref2_ptr, *ref3_ptr;
v16u8 src, ref0, ref1, ref2, ref3, diff;
@ -339,9 +339,9 @@ static void sad_16width_x4d_msa(const uint8_t *src_ptr, int32_t src_stride,
}
static void sad_32width_x4d_msa(const uint8_t *src, int32_t src_stride,
const uint8_t *const aref_ptr[],
const uint8_t *const aref_ptr[4],
int32_t ref_stride, int32_t height,
uint32_t *sad_array) {
uint32_t sad_array[4]) {
const uint8_t *ref0_ptr, *ref1_ptr, *ref2_ptr, *ref3_ptr;
int32_t ht_cnt;
v16u8 src0, src1, ref0, ref1;
@ -383,9 +383,9 @@ static void sad_32width_x4d_msa(const uint8_t *src, int32_t src_stride,
}
static void sad_64width_x4d_msa(const uint8_t *src, int32_t src_stride,
const uint8_t *const aref_ptr[],
const uint8_t *const aref_ptr[4],
int32_t ref_stride, int32_t height,
uint32_t *sad_array) {
uint32_t sad_array[4]) {
const uint8_t *ref0_ptr, *ref1_ptr, *ref2_ptr, *ref3_ptr;
int32_t ht_cnt;
v16u8 src0, src1, src2, src3;
@ -659,36 +659,36 @@ static uint32_t avgsad_64width_msa(const uint8_t *src, int32_t src_stride,
#define AOM_SAD_4xHEIGHTx4D_MSA(height) \
void aom_sad4x##height##x4d_msa(const uint8_t *src, int32_t src_stride, \
const uint8_t *const refs[], \
int32_t ref_stride, uint32_t *sads) { \
const uint8_t *const refs[4], \
int32_t ref_stride, uint32_t sads[4]) { \
sad_4width_x4d_msa(src, src_stride, refs, ref_stride, height, sads); \
}
#define AOM_SAD_8xHEIGHTx4D_MSA(height) \
void aom_sad8x##height##x4d_msa(const uint8_t *src, int32_t src_stride, \
const uint8_t *const refs[], \
int32_t ref_stride, uint32_t *sads) { \
const uint8_t *const refs[4], \
int32_t ref_stride, uint32_t sads[4]) { \
sad_8width_x4d_msa(src, src_stride, refs, ref_stride, height, sads); \
}
#define AOM_SAD_16xHEIGHTx4D_MSA(height) \
void aom_sad16x##height##x4d_msa(const uint8_t *src, int32_t src_stride, \
const uint8_t *const refs[], \
int32_t ref_stride, uint32_t *sads) { \
const uint8_t *const refs[4], \
int32_t ref_stride, uint32_t sads[4]) { \
sad_16width_x4d_msa(src, src_stride, refs, ref_stride, height, sads); \
}
#define AOM_SAD_32xHEIGHTx4D_MSA(height) \
void aom_sad32x##height##x4d_msa(const uint8_t *src, int32_t src_stride, \
const uint8_t *const refs[], \
int32_t ref_stride, uint32_t *sads) { \
const uint8_t *const refs[4], \
int32_t ref_stride, uint32_t sads[4]) { \
sad_32width_x4d_msa(src, src_stride, refs, ref_stride, height, sads); \
}
#define AOM_SAD_64xHEIGHTx4D_MSA(height) \
void aom_sad64x##height##x4d_msa(const uint8_t *src, int32_t src_stride, \
const uint8_t *const refs[], \
int32_t ref_stride, uint32_t *sads) { \
const uint8_t *const refs[4], \
int32_t ref_stride, uint32_t sads[4]) { \
sad_64width_x4d_msa(src, src_stride, refs, ref_stride, height, sads); \
}

View file

@ -15,11 +15,10 @@
#include <string.h>
#include "aom_dsp/aom_dsp_common.h"
#include "aom_dsp/mathutils.h"
#include "aom_dsp/noise_model.h"
#include "aom_dsp/noise_util.h"
#include "aom_mem/aom_mem.h"
#include "av1/common/common.h"
#include "av1/encoder/mathutils.h"
#define kLowPolyNumParams 3
@ -42,8 +41,8 @@ static const int kMaxLag = 4;
return block_mean / (max_w * max_h); \
}
GET_BLOCK_MEAN(uint8_t, lowbd);
GET_BLOCK_MEAN(uint16_t, highbd);
GET_BLOCK_MEAN(uint8_t, lowbd)
GET_BLOCK_MEAN(uint16_t, highbd)
static INLINE double get_block_mean(const uint8_t *data, int w, int h,
int stride, int x_o, int y_o,
@ -76,8 +75,8 @@ static INLINE double get_block_mean(const uint8_t *data, int w, int h,
return noise_var / (max_w * max_h) - noise_mean * noise_mean; \
}
GET_NOISE_VAR(uint8_t, lowbd);
GET_NOISE_VAR(uint16_t, highbd);
GET_NOISE_VAR(uint8_t, lowbd)
GET_NOISE_VAR(uint16_t, highbd)
static INLINE double get_noise_var(const uint8_t *data, const uint8_t *denoised,
int w, int h, int stride, int x_o, int y_o,
@ -214,6 +213,7 @@ static void set_chroma_coefficient_fallback_soln(aom_equation_system_t *eqns) {
int aom_noise_strength_lut_init(aom_noise_strength_lut_t *lut, int num_points) {
if (!lut) return 0;
if (num_points <= 0) return 0;
lut->num_points = 0;
lut->points = (double(*)[2])aom_malloc(num_points * sizeof(*lut->points));
if (!lut->points) return 0;
@ -388,6 +388,10 @@ int aom_noise_strength_solver_fit_piecewise(
}
double *residual = aom_malloc(solver->num_bins * sizeof(*residual));
if (!residual) {
aom_noise_strength_lut_free(lut);
return 0;
}
memset(residual, 0, sizeof(*residual) * solver->num_bins);
update_piecewise_linear_residual(solver, lut, residual, 0, solver->num_bins);
@ -694,6 +698,10 @@ int aom_noise_model_init(aom_noise_model_t *model,
kMaxLag);
return 0;
}
if (!(params.bit_depth == 8 || params.bit_depth == 10 ||
params.bit_depth == 12)) {
return 0;
}
memcpy(&model->params, &params, sizeof(params));
for (c = 0; c < 3; ++c) {
@ -710,6 +718,10 @@ int aom_noise_model_init(aom_noise_model_t *model,
}
model->n = n;
model->coords = (int(*)[2])aom_malloc(sizeof(*model->coords) * n);
if (!model->coords) {
aom_noise_model_free(model);
return 0;
}
for (y = -lag; y <= 0; ++y) {
const int max_x = y == 0 ? -1 : lag;
@ -787,8 +799,8 @@ void aom_noise_model_free(aom_noise_model_t *model) {
return val; \
}
EXTRACT_AR_ROW(uint8_t, lowbd);
EXTRACT_AR_ROW(uint16_t, highbd);
EXTRACT_AR_ROW(uint8_t, lowbd)
EXTRACT_AR_ROW(uint16_t, highbd)
static int add_block_observations(
aom_noise_model_t *noise_model, int c, const uint8_t *const data,
@ -1152,12 +1164,24 @@ int aom_noise_model_get_grain_parameters(aom_noise_model_t *const noise_model,
// Convert the scaling functions to 8 bit values
aom_noise_strength_lut_t scaling_points[3];
aom_noise_strength_solver_fit_piecewise(
&noise_model->combined_state[0].strength_solver, 14, scaling_points + 0);
aom_noise_strength_solver_fit_piecewise(
&noise_model->combined_state[1].strength_solver, 10, scaling_points + 1);
aom_noise_strength_solver_fit_piecewise(
&noise_model->combined_state[2].strength_solver, 10, scaling_points + 2);
if (!aom_noise_strength_solver_fit_piecewise(
&noise_model->combined_state[0].strength_solver, 14,
scaling_points + 0)) {
return 0;
}
if (!aom_noise_strength_solver_fit_piecewise(
&noise_model->combined_state[1].strength_solver, 10,
scaling_points + 1)) {
aom_noise_strength_lut_free(scaling_points + 0);
return 0;
}
if (!aom_noise_strength_solver_fit_piecewise(
&noise_model->combined_state[2].strength_solver, 10,
scaling_points + 2)) {
aom_noise_strength_lut_free(scaling_points + 0);
aom_noise_strength_lut_free(scaling_points + 1);
return 0;
}
// Both the domain and the range of the scaling functions in the film_grain
// are normalized to 8-bit (e.g., they are implicitly scaled during grain
@ -1287,6 +1311,7 @@ static void pointwise_multiply(const float *a, float *b, int n) {
static float *get_half_cos_window(int block_size) {
float *window_function =
(float *)aom_malloc(block_size * block_size * sizeof(*window_function));
if (!window_function) return NULL;
for (int y = 0; y < block_size; ++y) {
const double cos_yd = cos((.5 + y) * PI / block_size - PI / 2);
for (int x = 0; x < block_size; ++x) {
@ -1329,8 +1354,8 @@ static float *get_half_cos_window(int block_size) {
} \
}
DITHER_AND_QUANTIZE(uint8_t, lowbd);
DITHER_AND_QUANTIZE(uint16_t, highbd);
DITHER_AND_QUANTIZE(uint8_t, lowbd)
DITHER_AND_QUANTIZE(uint16_t, highbd)
int aom_wiener_denoise_2d(const uint8_t *const data[3], uint8_t *denoised[3],
int w, int h, int stride[3], int chroma_sub[2],
@ -1353,7 +1378,7 @@ int aom_wiener_denoise_2d(const uint8_t *const data[3], uint8_t *denoised[3],
if (chroma_sub[0] != chroma_sub[1]) {
fprintf(stderr,
"aom_wiener_denoise_2d doesn't handle different chroma "
"subsampling");
"subsampling\n");
return 0;
}
init_success &= aom_flat_block_finder_init(&block_finder_full, block_size,
@ -1560,6 +1585,10 @@ static int denoise_and_model_realloc_if_necessary(
ctx->num_blocks_w = (sd->y_width + ctx->block_size - 1) / ctx->block_size;
ctx->num_blocks_h = (sd->y_height + ctx->block_size - 1) / ctx->block_size;
ctx->flat_blocks = aom_malloc(ctx->num_blocks_w * ctx->num_blocks_h);
if (!ctx->flat_blocks) {
fprintf(stderr, "Unable to allocate flat_blocks buffer\n");
return 0;
}
aom_flat_block_finder_free(&ctx->flat_block_finder);
if (!aom_flat_block_finder_init(&ctx->flat_block_finder, ctx->block_size,
@ -1591,7 +1620,7 @@ static int denoise_and_model_realloc_if_necessary(
int aom_denoise_and_model_run(struct aom_denoise_and_model_t *ctx,
YV12_BUFFER_CONFIG *sd,
aom_film_grain_t *film_grain) {
aom_film_grain_t *film_grain, int apply_denoise) {
const int block_size = ctx->block_size;
const int use_highbd = (sd->flags & YV12_FLAG_HIGHBITDEPTH) != 0;
uint8_t *raw_data[3] = {
@ -1643,12 +1672,14 @@ int aom_denoise_and_model_run(struct aom_denoise_and_model_t *ctx,
if (!film_grain->random_seed) {
film_grain->random_seed = 7391;
}
memcpy(raw_data[0], ctx->denoised[0],
(strides[0] * sd->y_height) << use_highbd);
memcpy(raw_data[1], ctx->denoised[1],
(strides[1] * sd->uv_height) << use_highbd);
memcpy(raw_data[2], ctx->denoised[2],
(strides[2] * sd->uv_height) << use_highbd);
if (apply_denoise) {
memcpy(raw_data[0], ctx->denoised[0],
(strides[0] * sd->y_height) << use_highbd);
memcpy(raw_data[1], ctx->denoised[1],
(strides[1] * sd->uv_height) << use_highbd);
memcpy(raw_data[2], ctx->denoised[2],
(strides[2] * sd->uv_height) << use_highbd);
}
}
return 1;
}

View file

@ -17,7 +17,8 @@ extern "C" {
#endif // __cplusplus
#include <stdint.h>
#include "aom_dsp/grain_synthesis.h"
#include "aom_dsp/grain_params.h"
#include "aom_ports/mem.h"
#include "aom_scale/yv12config.h"
/*!\brief Wrapper of data required to represent linear system of eqns and soln.
@ -292,14 +293,18 @@ struct aom_denoise_and_model_t;
* parameter will be true when the input buffer was successfully denoised and
* grain was modelled. Returns false on error.
*
* \param[in] ctx Struct allocated with aom_denoise_and_model_alloc
* that holds some buffers for denoising and the current
* noise estimate.
* \param[in/out] buf The raw input buffer to be denoised.
* \param[out] grain Output film grain parameters
* \param[in] ctx Struct allocated with
* aom_denoise_and_model_alloc that holds some
* buffers for denoising and the current noise
* estimate.
* \param[in/out] buf The raw input buffer to be denoised.
* \param[out] grain Output film grain parameters
* \param[out] apply_denoise Whether or not to apply the denoising to the
* frame that will be encoded
*/
int aom_denoise_and_model_run(struct aom_denoise_and_model_t *ctx,
YV12_BUFFER_CONFIG *buf, aom_film_grain_t *grain);
YV12_BUFFER_CONFIG *buf, aom_film_grain_t *grain,
int apply_denoise);
/*!\brief Allocates a context that can be used for denoising and noise modeling.
*

View file

@ -160,15 +160,17 @@ int aom_noise_data_validate(const double *data, int w, int h) {
// Check that noise variance is not increasing in x or y
// and that the data is zero mean.
mean_x = (double *)aom_malloc(sizeof(*mean_x) * w);
var_x = (double *)aom_malloc(sizeof(*var_x) * w);
mean_y = (double *)aom_malloc(sizeof(*mean_x) * h);
var_y = (double *)aom_malloc(sizeof(*var_y) * h);
memset(mean_x, 0, sizeof(*mean_x) * w);
memset(var_x, 0, sizeof(*var_x) * w);
memset(mean_y, 0, sizeof(*mean_y) * h);
memset(var_y, 0, sizeof(*var_y) * h);
mean_x = (double *)aom_calloc(w, sizeof(*mean_x));
var_x = (double *)aom_calloc(w, sizeof(*var_x));
mean_y = (double *)aom_calloc(h, sizeof(*mean_x));
var_y = (double *)aom_calloc(h, sizeof(*var_y));
if (!(mean_x && var_x && mean_y && var_y)) {
aom_free(mean_x);
aom_free(mean_y);
aom_free(var_x);
aom_free(var_y);
return 0;
}
for (y = 0; y < h; ++y) {
for (x = 0; x < w; ++x) {

View file

@ -11,7 +11,7 @@
/* clang-format off */
#include "av1/common/odintrin.h"
#include "aom_dsp/odintrin.h"
/*Constants for use with OD_DIVU_SMALL().
See \cite{Rob05} for details on computing these constants.

View file

@ -11,8 +11,8 @@
/* clang-format off */
#ifndef AOM_AV1_COMMON_ODINTRIN_H_
#define AOM_AV1_COMMON_ODINTRIN_H_
#ifndef AOM_AOM_DSP_ODINTRIN_H_
#define AOM_AOM_DSP_ODINTRIN_H_
#include <stdlib.h>
#include <string.h>
@ -20,7 +20,6 @@
#include "aom/aom_integer.h"
#include "aom_dsp/aom_dsp_common.h"
#include "aom_ports/bitops.h"
#include "av1/common/enums.h"
#ifdef __cplusplus
extern "C" {
@ -93,4 +92,4 @@ extern uint32_t OD_DIVU_SMALL_CONSTS[OD_DIVU_DMAX][2];
} // extern "C"
#endif
#endif // AOM_AV1_COMMON_ODINTRIN_H_
#endif // AOM_AOM_DSP_ODINTRIN_H_

View file

@ -363,6 +363,10 @@ int64_t aom_get_sse_plane(const YV12_BUFFER_CONFIG *a,
void aom_calc_highbd_psnr(const YV12_BUFFER_CONFIG *a,
const YV12_BUFFER_CONFIG *b, PSNR_STATS *psnr,
uint32_t bit_depth, uint32_t in_bit_depth) {
assert(a->y_crop_width == b->y_crop_width);
assert(a->y_crop_height == b->y_crop_height);
assert(a->uv_crop_width == b->uv_crop_width);
assert(a->uv_crop_height == b->uv_crop_height);
const int widths[3] = { a->y_crop_width, a->uv_crop_width, a->uv_crop_width };
const int heights[3] = { a->y_crop_height, a->uv_crop_height,
a->uv_crop_height };
@ -371,7 +375,7 @@ void aom_calc_highbd_psnr(const YV12_BUFFER_CONFIG *a,
int i;
uint64_t total_sse = 0;
uint32_t total_samples = 0;
const double peak = (double)((1 << in_bit_depth) - 1);
double peak = (double)((1 << in_bit_depth) - 1);
const unsigned int input_shift = bit_depth - in_bit_depth;
for (i = 0; i < 3; ++i) {
@ -403,11 +407,40 @@ void aom_calc_highbd_psnr(const YV12_BUFFER_CONFIG *a,
psnr->samples[0] = total_samples;
psnr->psnr[0] =
aom_sse_to_psnr((double)total_samples, peak, (double)total_sse);
// Compute PSNR based on stream bit depth
if ((a->flags & YV12_FLAG_HIGHBITDEPTH) && (in_bit_depth < bit_depth)) {
peak = (double)((1 << bit_depth) - 1);
total_sse = 0;
total_samples = 0;
for (i = 0; i < 3; ++i) {
const int w = widths[i];
const int h = heights[i];
const uint32_t samples = w * h;
uint64_t sse;
sse = highbd_get_sse(a->buffers[i], a_strides[i], b->buffers[i],
b_strides[i], w, h);
psnr->sse_hbd[1 + i] = sse;
psnr->samples_hbd[1 + i] = samples;
psnr->psnr_hbd[1 + i] = aom_sse_to_psnr(samples, peak, (double)sse);
total_sse += sse;
total_samples += samples;
}
psnr->sse_hbd[0] = total_sse;
psnr->samples_hbd[0] = total_samples;
psnr->psnr_hbd[0] =
aom_sse_to_psnr((double)total_samples, peak, (double)total_sse);
}
}
#endif
void aom_calc_psnr(const YV12_BUFFER_CONFIG *a, const YV12_BUFFER_CONFIG *b,
PSNR_STATS *psnr) {
assert(a->y_crop_width == b->y_crop_width);
assert(a->y_crop_height == b->y_crop_height);
assert(a->uv_crop_width == b->uv_crop_width);
assert(a->uv_crop_height == b->uv_crop_height);
static const double peak = 255.0;
const int widths[3] = { a->y_crop_width, a->uv_crop_width, a->uv_crop_width };
const int heights[3] = { a->y_crop_height, a->uv_crop_height,

View file

@ -21,9 +21,12 @@ extern "C" {
#endif
typedef struct {
double psnr[4]; // total/y/u/v
uint64_t sse[4]; // total/y/u/v
uint32_t samples[4]; // total/y/u/v
double psnr[4]; // total/y/u/v
uint64_t sse[4]; // total/y/u/v
uint32_t samples[4]; // total/y/u/v
double psnr_hbd[4]; // total/y/u/v when input-bit-depth < bit-depth
uint64_t sse_hbd[4]; // total/y/u/v when input-bit-depth < bit-depth
uint32_t samples_hbd[4]; // total/y/u/v when input-bit-depth < bit-depth
} PSNR_STATS;
/*!\brief Converts SSE to PSNR

View file

@ -22,7 +22,6 @@
#include "aom_dsp/psnr.h"
#include "aom_dsp/ssim.h"
#include "aom_ports/system_state.h"
static void od_bin_fdct8x8(tran_low_t *y, int ystride, const int16_t *x,
int xstride) {
@ -34,6 +33,7 @@ static void od_bin_fdct8x8(tran_low_t *y, int ystride, const int16_t *x,
*(y + ystride * i + j) = (*(y + ystride * i + j) + 4) >> 3;
}
#if CONFIG_AV1_HIGHBITDEPTH
static void hbd_od_bin_fdct8x8(tran_low_t *y, int ystride, const int16_t *x,
int xstride) {
int i, j;
@ -43,6 +43,7 @@ static void hbd_od_bin_fdct8x8(tran_low_t *y, int ystride, const int16_t *x,
for (j = 0; j < 8; j++)
*(y + ystride * i + j) = (*(y + ystride * i + j) + 4) >> 3;
}
#endif // CONFIG_AV1_HIGHBITDEPTH
/* Normalized inverse quantization matrix for 8x8 DCT at the point of
* transparency. This is not the JPEG based matrix from the paper,
@ -210,6 +211,7 @@ static double calc_psnrhvs(const unsigned char *src, int _systride,
}
}
s_gvar = 1.f / (36 - n + 1) * s_gmean / 36.f;
#if CONFIG_AV1_HIGHBITDEPTH
if (!buf_is_hbd) {
od_bin_fdct8x8(dct_s_coef, 8, dct_s, 8);
od_bin_fdct8x8(dct_d_coef, 8, dct_d, 8);
@ -217,6 +219,10 @@ static double calc_psnrhvs(const unsigned char *src, int _systride,
hbd_od_bin_fdct8x8(dct_s_coef, 8, dct_s, 8);
hbd_od_bin_fdct8x8(dct_d_coef, 8, dct_d, 8);
}
#else
od_bin_fdct8x8(dct_s_coef, 8, dct_s, 8);
od_bin_fdct8x8(dct_d_coef, 8, dct_d, 8);
#endif // CONFIG_AV1_HIGHBITDEPTH
for (i = 0; i < 8; i++)
for (j = (i == 0); j < 8; j++)
s_mask += dct_s_coef[i * 8 + j] * dct_s_coef[i * 8 + j] * mask[i][j];
@ -246,7 +252,6 @@ double aom_psnrhvs(const YV12_BUFFER_CONFIG *src, const YV12_BUFFER_CONFIG *dst,
const double par = 1.0;
const int step = 7;
uint32_t bd_shift = 0;
aom_clear_system_state();
assert(bd == 8 || bd == 10 || bd == 12);
assert(bd >= in_bd);
assert(src->flags == dst->flags);

View file

@ -11,7 +11,6 @@
#include "aom_dsp/quantize.h"
#include "aom_mem/aom_mem.h"
#include "av1/encoder/av1_quantize.h"
void aom_quantize_b_adaptive_helper_c(
const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr,

View file

@ -20,6 +20,9 @@
extern "C" {
#endif
#define EOB_FACTOR 325
#define SKIP_EOB_FACTOR_ADJUST 200
void aom_quantize_b_adaptive_helper_c(
const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr,
const int16_t *round_ptr, const int16_t *quant_ptr,

View file

@ -35,14 +35,14 @@ static INLINE unsigned int sad(const uint8_t *a, int a_stride, const uint8_t *b,
return sad;
}
#define sadMxh(m) \
#define SAD_MXH(m) \
unsigned int aom_sad##m##xh_c(const uint8_t *a, int a_stride, \
const uint8_t *b, int b_stride, int width, \
int height) { \
return sad(a, a_stride, b, b_stride, width, height); \
}
#define sadMxN(m, n) \
#define SADMXN(m, n) \
unsigned int aom_sad##m##x##n##_c(const uint8_t *src, int src_stride, \
const uint8_t *ref, int ref_stride) { \
return sad(src, src_stride, ref, ref_stride, m, n); \
@ -61,112 +61,149 @@ static INLINE unsigned int sad(const uint8_t *a, int a_stride, const uint8_t *b,
aom_dist_wtd_comp_avg_pred_c(comp_pred, second_pred, m, n, ref, \
ref_stride, jcp_param); \
return sad(src, src_stride, comp_pred, m, m, n); \
} \
unsigned int aom_sad_skip_##m##x##n##_c(const uint8_t *src, int src_stride, \
const uint8_t *ref, \
int ref_stride) { \
return 2 * sad(src, 2 * src_stride, ref, 2 * ref_stride, (m), (n / 2)); \
}
#if CONFIG_REALTIME_ONLY
// Calculate sad against 4 reference locations and store each in sad_array
#define sadMxNx4D(m, n) \
void aom_sad##m##x##n##x4d_c(const uint8_t *src, int src_stride, \
const uint8_t *const ref_array[], \
int ref_stride, uint32_t *sad_array) { \
int i; \
for (i = 0; i < 4; ++i) { \
sad_array[i] = \
aom_sad##m##x##n##_c(src, src_stride, ref_array[i], ref_stride); \
} \
} \
void aom_sad##m##x##n##x4d_avg_c( \
const uint8_t *src, int src_stride, const uint8_t *const ref_array[], \
int ref_stride, const uint8_t *second_pred, uint32_t *sad_array) { \
int i; \
for (i = 0; i < 4; ++i) { \
sad_array[i] = aom_sad##m##x##n##_avg_c(src, src_stride, ref_array[i], \
ref_stride, second_pred); \
} \
#define SAD_MXNX4D(m, n) \
void aom_sad##m##x##n##x4d_c(const uint8_t *src, int src_stride, \
const uint8_t *const ref_array[4], \
int ref_stride, uint32_t sad_array[4]) { \
int i; \
for (i = 0; i < 4; ++i) { \
sad_array[i] = \
aom_sad##m##x##n##_c(src, src_stride, ref_array[i], ref_stride); \
} \
} \
void aom_sad_skip_##m##x##n##x4d_c(const uint8_t *src, int src_stride, \
const uint8_t *const ref_array[4], \
int ref_stride, uint32_t sad_array[4]) { \
int i; \
for (i = 0; i < 4; ++i) { \
sad_array[i] = 2 * sad(src, 2 * src_stride, ref_array[i], \
2 * ref_stride, (m), (n / 2)); \
} \
}
#else // !CONFIG_REALTIME_ONLY
// Calculate sad against 4 reference locations and store each in sad_array
#define SAD_MXNX4D(m, n) \
void aom_sad##m##x##n##x4d_c(const uint8_t *src, int src_stride, \
const uint8_t *const ref_array[4], \
int ref_stride, uint32_t sad_array[4]) { \
int i; \
for (i = 0; i < 4; ++i) { \
sad_array[i] = \
aom_sad##m##x##n##_c(src, src_stride, ref_array[i], ref_stride); \
} \
} \
void aom_sad##m##x##n##x4d_avg_c( \
const uint8_t *src, int src_stride, const uint8_t *const ref_array[4], \
int ref_stride, const uint8_t *second_pred, uint32_t sad_array[4]) { \
int i; \
for (i = 0; i < 4; ++i) { \
sad_array[i] = aom_sad##m##x##n##_avg_c(src, src_stride, ref_array[i], \
ref_stride, second_pred); \
} \
} \
void aom_sad_skip_##m##x##n##x4d_c(const uint8_t *src, int src_stride, \
const uint8_t *const ref_array[4], \
int ref_stride, uint32_t sad_array[4]) { \
int i; \
for (i = 0; i < 4; ++i) { \
sad_array[i] = 2 * sad(src, 2 * src_stride, ref_array[i], \
2 * ref_stride, (m), (n / 2)); \
} \
}
#endif // CONFIG_REALTIME_ONLY
// 128x128
sadMxN(128, 128);
sadMxNx4D(128, 128);
SADMXN(128, 128)
SAD_MXNX4D(128, 128)
// 128x64
sadMxN(128, 64);
sadMxNx4D(128, 64);
SADMXN(128, 64)
SAD_MXNX4D(128, 64)
// 64x128
sadMxN(64, 128);
sadMxNx4D(64, 128);
SADMXN(64, 128)
SAD_MXNX4D(64, 128)
// 64x64
sadMxN(64, 64);
sadMxNx4D(64, 64);
SADMXN(64, 64)
SAD_MXNX4D(64, 64)
// 64x32
sadMxN(64, 32);
sadMxNx4D(64, 32);
SADMXN(64, 32)
SAD_MXNX4D(64, 32)
// 32x64
sadMxN(32, 64);
sadMxNx4D(32, 64);
SADMXN(32, 64)
SAD_MXNX4D(32, 64)
// 32x32
sadMxN(32, 32);
sadMxNx4D(32, 32);
SADMXN(32, 32)
SAD_MXNX4D(32, 32)
// 32x16
sadMxN(32, 16);
sadMxNx4D(32, 16);
SADMXN(32, 16)
SAD_MXNX4D(32, 16)
// 16x32
sadMxN(16, 32);
sadMxNx4D(16, 32);
SADMXN(16, 32)
SAD_MXNX4D(16, 32)
// 16x16
sadMxN(16, 16);
sadMxNx4D(16, 16);
SADMXN(16, 16)
SAD_MXNX4D(16, 16)
// 16x8
sadMxN(16, 8);
sadMxNx4D(16, 8);
SADMXN(16, 8)
SAD_MXNX4D(16, 8)
// 8x16
sadMxN(8, 16);
sadMxNx4D(8, 16);
SADMXN(8, 16)
SAD_MXNX4D(8, 16)
// 8x8
sadMxN(8, 8);
sadMxNx4D(8, 8);
SADMXN(8, 8)
SAD_MXNX4D(8, 8)
// 8x4
sadMxN(8, 4);
sadMxNx4D(8, 4);
SADMXN(8, 4)
SAD_MXNX4D(8, 4)
// 4x8
sadMxN(4, 8);
sadMxNx4D(4, 8);
SADMXN(4, 8)
SAD_MXNX4D(4, 8)
// 4x4
sadMxN(4, 4);
sadMxNx4D(4, 4);
SADMXN(4, 4)
SAD_MXNX4D(4, 4)
sadMxh(128);
sadMxh(64);
sadMxh(32);
sadMxh(16);
sadMxh(8);
sadMxh(4);
SAD_MXH(128)
SAD_MXH(64)
SAD_MXH(32)
SAD_MXH(16)
SAD_MXH(8)
SAD_MXH(4)
sadMxN(4, 16);
sadMxNx4D(4, 16);
sadMxN(16, 4);
sadMxNx4D(16, 4);
sadMxN(8, 32);
sadMxNx4D(8, 32);
sadMxN(32, 8);
sadMxNx4D(32, 8);
sadMxN(16, 64);
sadMxNx4D(16, 64);
sadMxN(64, 16);
sadMxNx4D(64, 16);
SADMXN(4, 16)
SAD_MXNX4D(4, 16)
SADMXN(16, 4)
SAD_MXNX4D(16, 4)
SADMXN(8, 32)
SAD_MXNX4D(8, 32)
SADMXN(32, 8)
SAD_MXNX4D(32, 8)
SADMXN(16, 64)
SAD_MXNX4D(16, 64)
SADMXN(64, 16)
SAD_MXNX4D(64, 16)
#if CONFIG_AV1_HIGHBITDEPTH
static INLINE unsigned int highbd_sad(const uint8_t *a8, int a_stride,
@ -205,7 +242,7 @@ static INLINE unsigned int highbd_sadb(const uint8_t *a8, int a_stride,
return sad;
}
#define highbd_sadMxN(m, n) \
#define HIGHBD_SADMXN(m, n) \
unsigned int aom_highbd_sad##m##x##n##_c(const uint8_t *src, int src_stride, \
const uint8_t *ref, \
int ref_stride) { \
@ -227,9 +264,15 @@ static INLINE unsigned int highbd_sadb(const uint8_t *a8, int a_stride,
aom_highbd_dist_wtd_comp_avg_pred(comp_pred8, second_pred, m, n, ref, \
ref_stride, jcp_param); \
return highbd_sadb(src, src_stride, comp_pred8, m, m, n); \
} \
unsigned int aom_highbd_sad_skip_##m##x##n##_c( \
const uint8_t *src, int src_stride, const uint8_t *ref, \
int ref_stride) { \
return 2 * \
highbd_sad(src, 2 * src_stride, ref, 2 * ref_stride, (m), (n / 2)); \
}
#define highbd_sadMxNx4D(m, n) \
#define HIGHBD_SAD_MXNX4D(m, n) \
void aom_highbd_sad##m##x##n##x4d_c(const uint8_t *src, int src_stride, \
const uint8_t *const ref_array[], \
int ref_stride, uint32_t *sad_array) { \
@ -238,82 +281,91 @@ static INLINE unsigned int highbd_sadb(const uint8_t *a8, int a_stride,
sad_array[i] = aom_highbd_sad##m##x##n##_c(src, src_stride, \
ref_array[i], ref_stride); \
} \
} \
void aom_highbd_sad_skip_##m##x##n##x4d_c( \
const uint8_t *src, int src_stride, const uint8_t *const ref_array[], \
int ref_stride, uint32_t *sad_array) { \
int i; \
for (i = 0; i < 4; ++i) { \
sad_array[i] = 2 * highbd_sad(src, 2 * src_stride, ref_array[i], \
2 * ref_stride, (m), (n / 2)); \
} \
}
// 128x128
highbd_sadMxN(128, 128);
highbd_sadMxNx4D(128, 128);
HIGHBD_SADMXN(128, 128)
HIGHBD_SAD_MXNX4D(128, 128)
// 128x64
highbd_sadMxN(128, 64);
highbd_sadMxNx4D(128, 64);
HIGHBD_SADMXN(128, 64)
HIGHBD_SAD_MXNX4D(128, 64)
// 64x128
highbd_sadMxN(64, 128);
highbd_sadMxNx4D(64, 128);
HIGHBD_SADMXN(64, 128)
HIGHBD_SAD_MXNX4D(64, 128)
// 64x64
highbd_sadMxN(64, 64);
highbd_sadMxNx4D(64, 64);
HIGHBD_SADMXN(64, 64)
HIGHBD_SAD_MXNX4D(64, 64)
// 64x32
highbd_sadMxN(64, 32);
highbd_sadMxNx4D(64, 32);
HIGHBD_SADMXN(64, 32)
HIGHBD_SAD_MXNX4D(64, 32)
// 32x64
highbd_sadMxN(32, 64);
highbd_sadMxNx4D(32, 64);
HIGHBD_SADMXN(32, 64)
HIGHBD_SAD_MXNX4D(32, 64)
// 32x32
highbd_sadMxN(32, 32);
highbd_sadMxNx4D(32, 32);
HIGHBD_SADMXN(32, 32)
HIGHBD_SAD_MXNX4D(32, 32)
// 32x16
highbd_sadMxN(32, 16);
highbd_sadMxNx4D(32, 16);
HIGHBD_SADMXN(32, 16)
HIGHBD_SAD_MXNX4D(32, 16)
// 16x32
highbd_sadMxN(16, 32);
highbd_sadMxNx4D(16, 32);
HIGHBD_SADMXN(16, 32)
HIGHBD_SAD_MXNX4D(16, 32)
// 16x16
highbd_sadMxN(16, 16);
highbd_sadMxNx4D(16, 16);
HIGHBD_SADMXN(16, 16)
HIGHBD_SAD_MXNX4D(16, 16)
// 16x8
highbd_sadMxN(16, 8);
highbd_sadMxNx4D(16, 8);
HIGHBD_SADMXN(16, 8)
HIGHBD_SAD_MXNX4D(16, 8)
// 8x16
highbd_sadMxN(8, 16);
highbd_sadMxNx4D(8, 16);
HIGHBD_SADMXN(8, 16)
HIGHBD_SAD_MXNX4D(8, 16)
// 8x8
highbd_sadMxN(8, 8);
highbd_sadMxNx4D(8, 8);
HIGHBD_SADMXN(8, 8)
HIGHBD_SAD_MXNX4D(8, 8)
// 8x4
highbd_sadMxN(8, 4);
highbd_sadMxNx4D(8, 4);
HIGHBD_SADMXN(8, 4)
HIGHBD_SAD_MXNX4D(8, 4)
// 4x8
highbd_sadMxN(4, 8);
highbd_sadMxNx4D(4, 8);
HIGHBD_SADMXN(4, 8)
HIGHBD_SAD_MXNX4D(4, 8)
// 4x4
highbd_sadMxN(4, 4);
highbd_sadMxNx4D(4, 4);
HIGHBD_SADMXN(4, 4)
HIGHBD_SAD_MXNX4D(4, 4)
highbd_sadMxN(4, 16);
highbd_sadMxNx4D(4, 16);
highbd_sadMxN(16, 4);
highbd_sadMxNx4D(16, 4);
highbd_sadMxN(8, 32);
highbd_sadMxNx4D(8, 32);
highbd_sadMxN(32, 8);
highbd_sadMxNx4D(32, 8);
highbd_sadMxN(16, 64);
highbd_sadMxNx4D(16, 64);
highbd_sadMxN(64, 16);
highbd_sadMxNx4D(64, 16);
HIGHBD_SADMXN(4, 16)
HIGHBD_SAD_MXNX4D(4, 16)
HIGHBD_SADMXN(16, 4)
HIGHBD_SAD_MXNX4D(16, 4)
HIGHBD_SADMXN(8, 32)
HIGHBD_SAD_MXNX4D(8, 32)
HIGHBD_SADMXN(32, 8)
HIGHBD_SAD_MXNX4D(32, 8)
HIGHBD_SADMXN(16, 64)
HIGHBD_SAD_MXNX4D(16, 64)
HIGHBD_SADMXN(64, 16)
HIGHBD_SAD_MXNX4D(64, 16)
#endif // CONFIG_AV1_HIGHBITDEPTH

View file

@ -51,9 +51,9 @@ static INLINE unsigned int masked_sad(const uint8_t *src, int src_stride,
msk_stride, m, n); \
} \
void aom_masked_sad##m##x##n##x4d_c( \
const uint8_t *src, int src_stride, const uint8_t *ref[], \
const uint8_t *src, int src_stride, const uint8_t *ref[4], \
int ref_stride, const uint8_t *second_pred, const uint8_t *msk, \
int msk_stride, int invert_mask, unsigned sads[]) { \
int msk_stride, int invert_mask, unsigned sads[4]) { \
if (!invert_mask) \
for (int i = 0; i < 4; i++) { \
sads[i] = masked_sad(src, src_stride, ref[i], ref_stride, second_pred, \
@ -156,6 +156,7 @@ HIGHBD_MASKSADMXN(16, 64)
HIGHBD_MASKSADMXN(64, 16)
#endif // CONFIG_AV1_HIGHBITDEPTH
#if !CONFIG_REALTIME_ONLY
// pre: predictor being evaluated
// wsrc: target weighted prediction (has been *4096 to keep precision)
// mask: 2d weights (scaled by 4096)
@ -262,3 +263,4 @@ HIGHBD_OBMCSADMXN(16, 64)
HIGHBD_OBMCSADMXN(64, 16)
/* clang-format on */
#endif // CONFIG_AV1_HIGHBITDEPTH
#endif // !CONFIG_REALTIME_ONLY

View file

@ -64,9 +64,9 @@ SIMD_INLINE c_v128 c_v128_from_32(uint32_t a, uint32_t b, uint32_t c,
SIMD_INLINE c_v128 c_v128_load_unaligned(const void *p) {
c_v128 t;
uint8_t *pp = (uint8_t *)p;
uint8_t *q = (uint8_t *)&t;
int c;
for (c = 0; c < 16; c++) q[c] = pp[c];
// Note memcpy is avoided due to some versions of gcc issuing -Warray-bounds.
for (c = 0; c < 16; c++) t.u8[c] = pp[c];
return t;
}
@ -80,9 +80,8 @@ SIMD_INLINE c_v128 c_v128_load_aligned(const void *p) {
SIMD_INLINE void c_v128_store_unaligned(void *p, c_v128 a) {
uint8_t *pp = (uint8_t *)p;
uint8_t *q = (uint8_t *)&a;
int c;
for (c = 0; c < 16; c++) pp[c] = q[c];
for (c = 0; c < 16; c++) pp[c] = a.u8[c];
}
SIMD_INLINE void c_v128_store_aligned(void *p, c_v128 a) {

View file

@ -71,9 +71,9 @@ SIMD_INLINE c_v256 c_v256_from_v64(c_v64 a, c_v64 b, c_v64 c, c_v64 d) {
SIMD_INLINE c_v256 c_v256_load_unaligned(const void *p) {
c_v256 t;
uint8_t *pp = (uint8_t *)p;
uint8_t *q = (uint8_t *)&t;
int c;
for (c = 0; c < 32; c++) q[c] = pp[c];
// Note memcpy is avoided due to some versions of gcc issuing -Warray-bounds.
for (c = 0; c < 32; c++) t.u8[c] = pp[c];
return t;
}
@ -87,9 +87,8 @@ SIMD_INLINE c_v256 c_v256_load_aligned(const void *p) {
SIMD_INLINE void c_v256_store_unaligned(void *p, c_v256 a) {
uint8_t *pp = (uint8_t *)p;
uint8_t *q = (uint8_t *)&a;
int c;
for (c = 0; c < 32; c++) pp[c] = q[c];
for (c = 0; c < 32; c++) pp[c] = a.u8[c];
}
SIMD_INLINE void c_v256_store_aligned(void *p, c_v256 a) {

View file

@ -664,15 +664,14 @@ SIMD_INLINE v256 v256_shr_s64(v256 a, unsigned int c) {
v128_shl_n_byte(v256_low_v128(a), (n)-16), 1))
// _mm256_srli_si256 works on 128 bit lanes and can't be used
#define v256_shr_n_byte(a, n) \
((n) < 16 \
? _mm256_alignr_epi8( \
_mm256_permute2x128_si256(a, a, _MM_SHUFFLE(2, 0, 0, 1)), a, n) \
: ((n) == 16 \
? _mm256_permute2x128_si256(_mm256_setzero_si256(), a, 3) \
: _mm256_inserti128_si256( \
_mm256_setzero_si256(), \
v128_align(v256_high_v128(a), v256_high_v128(a), n), 0)))
#define v256_shr_n_byte(a, n) \
((n) < 16 \
? _mm256_alignr_epi8( \
_mm256_permute2x128_si256(a, a, _MM_SHUFFLE(2, 0, 0, 1)), a, n) \
: ((n) == 16 ? _mm256_permute2x128_si256(_mm256_setzero_si256(), a, 3) \
: _mm256_inserti128_si256( \
_mm256_setzero_si256(), \
v128_shr_n_byte(v256_high_v128(a), (n)-16), 0)))
// _mm256_alignr_epi8 works on two 128 bit lanes and can't be used
#define v256_align(a, b, c) \

View file

@ -16,8 +16,8 @@
#include "aom_dsp/ssim.h"
#include "aom_ports/mem.h"
#include "aom_ports/system_state.h"
#if CONFIG_INTERNAL_STATS
void aom_ssim_parms_16x16_c(const uint8_t *s, int sp, const uint8_t *r, int rp,
uint32_t *sum_s, uint32_t *sum_r,
uint32_t *sum_sq_s, uint32_t *sum_sq_r,
@ -33,6 +33,7 @@ void aom_ssim_parms_16x16_c(const uint8_t *s, int sp, const uint8_t *r, int rp,
}
}
}
#endif // CONFIG_INTERNAL_STATS
void aom_ssim_parms_8x8_c(const uint8_t *s, int sp, const uint8_t *r, int rp,
uint32_t *sum_s, uint32_t *sum_r, uint32_t *sum_sq_s,
@ -49,24 +50,6 @@ void aom_ssim_parms_8x8_c(const uint8_t *s, int sp, const uint8_t *r, int rp,
}
}
#if CONFIG_AV1_HIGHBITDEPTH
void aom_highbd_ssim_parms_8x8_c(const uint16_t *s, int sp, const uint16_t *r,
int rp, uint32_t *sum_s, uint32_t *sum_r,
uint32_t *sum_sq_s, uint32_t *sum_sq_r,
uint32_t *sum_sxr) {
int i, j;
for (i = 0; i < 8; i++, s += sp, r += rp) {
for (j = 0; j < 8; j++) {
*sum_s += s[j];
*sum_r += r[j];
*sum_sq_s += s[j] * s[j];
*sum_sq_r += r[j] * r[j];
*sum_sxr += s[j] * r[j];
}
}
}
#endif
static const int64_t cc1 = 26634; // (64^2*(.01*255)^2
static const int64_t cc2 = 239708; // (64^2*(.03*255)^2
static const int64_t cc1_10 = 428658; // (64^2*(.01*1023)^2
@ -77,8 +60,8 @@ static const int64_t cc2_12 = 61817334; // (64^2*(.03*4095)^2
static double similarity(uint32_t sum_s, uint32_t sum_r, uint32_t sum_sq_s,
uint32_t sum_sq_r, uint32_t sum_sxr, int count,
uint32_t bd) {
int64_t ssim_n, ssim_d;
int64_t c1, c2;
double ssim_n, ssim_d;
int64_t c1 = 0, c2 = 0;
if (bd == 8) {
// scale the constants by number of pixels
c1 = (cc1 * count * count) >> 12;
@ -90,18 +73,19 @@ static double similarity(uint32_t sum_s, uint32_t sum_r, uint32_t sum_sq_s,
c1 = (cc1_12 * count * count) >> 12;
c2 = (cc2_12 * count * count) >> 12;
} else {
c1 = c2 = 0;
assert(0);
// Return similarity as zero for unsupported bit-depth values.
return 0;
}
ssim_n = (2 * sum_s * sum_r + c1) *
((int64_t)2 * count * sum_sxr - (int64_t)2 * sum_s * sum_r + c2);
ssim_n = (2.0 * sum_s * sum_r + c1) *
(2.0 * count * sum_sxr - 2.0 * sum_s * sum_r + c2);
ssim_d = (sum_s * sum_s + sum_r * sum_r + c1) *
((int64_t)count * sum_sq_s - (int64_t)sum_s * sum_s +
(int64_t)count * sum_sq_r - (int64_t)sum_r * sum_r + c2);
ssim_d = ((double)sum_s * sum_s + (double)sum_r * sum_r + c1) *
((double)count * sum_sq_s - (double)sum_s * sum_s +
(double)count * sum_sq_r - (double)sum_r * sum_r + c2);
return ssim_n * 1.0 / ssim_d;
return ssim_n / ssim_d;
}
static double ssim_8x8(const uint8_t *s, int sp, const uint8_t *r, int rp) {
@ -111,21 +95,11 @@ static double ssim_8x8(const uint8_t *s, int sp, const uint8_t *r, int rp) {
return similarity(sum_s, sum_r, sum_sq_s, sum_sq_r, sum_sxr, 64, 8);
}
static double highbd_ssim_8x8(const uint16_t *s, int sp, const uint16_t *r,
int rp, uint32_t bd, uint32_t shift) {
uint32_t sum_s = 0, sum_r = 0, sum_sq_s = 0, sum_sq_r = 0, sum_sxr = 0;
aom_highbd_ssim_parms_8x8(s, sp, r, rp, &sum_s, &sum_r, &sum_sq_s, &sum_sq_r,
&sum_sxr);
return similarity(sum_s >> shift, sum_r >> shift, sum_sq_s >> (2 * shift),
sum_sq_r >> (2 * shift), sum_sxr >> (2 * shift), 64, bd);
}
// We are using a 8x8 moving window with starting location of each 8x8 window
// on the 4x4 pixel grid. Such arrangement allows the windows to overlap
// block boundaries to penalize blocking artifacts.
static double aom_ssim2(const uint8_t *img1, const uint8_t *img2,
int stride_img1, int stride_img2, int width,
int height) {
double aom_ssim2(const uint8_t *img1, const uint8_t *img2, int stride_img1,
int stride_img2, int width, int height) {
int i, j;
int samples = 0;
double ssim_total = 0;
@ -143,30 +117,10 @@ static double aom_ssim2(const uint8_t *img1, const uint8_t *img2,
return ssim_total;
}
static double aom_highbd_ssim2(const uint8_t *img1, const uint8_t *img2,
int stride_img1, int stride_img2, int width,
int height, uint32_t bd, uint32_t shift) {
int i, j;
int samples = 0;
double ssim_total = 0;
// sample point start with each 4x4 location
for (i = 0; i <= height - 8;
i += 4, img1 += stride_img1 * 4, img2 += stride_img2 * 4) {
for (j = 0; j <= width - 8; j += 4) {
double v = highbd_ssim_8x8(CONVERT_TO_SHORTPTR(img1 + j), stride_img1,
CONVERT_TO_SHORTPTR(img2 + j), stride_img2, bd,
shift);
ssim_total += v;
samples++;
}
}
ssim_total /= samples;
return ssim_total;
}
double aom_calc_ssim(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *dest, double *weight) {
#if CONFIG_INTERNAL_STATS
void aom_lowbd_calc_ssim(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *dest, double *weight,
double *fast_ssim) {
double abc[3];
for (int i = 0; i < 3; ++i) {
const int is_uv = i > 0;
@ -176,7 +130,7 @@ double aom_calc_ssim(const YV12_BUFFER_CONFIG *source,
}
*weight = 1;
return abc[0] * .8 + .1 * (abc[1] + abc[2]);
*fast_ssim = abc[0] * .8 + .1 * (abc[1] + abc[2]);
}
// traditional ssim as per: http://en.wikipedia.org/wiki/Structural_similarity
@ -272,7 +226,6 @@ double aom_get_ssim_metrics(uint8_t *img1, int img1_pitch, uint8_t *img2,
int c = 0;
double norm;
double old_ssim_total = 0;
aom_clear_system_state();
// We can sample points as frequently as we like start with 1 per 4x4.
for (i = 0; i < height;
i += 4, img1 += img1_pitch * 4, img2 += img2_pitch * 4) {
@ -420,12 +373,62 @@ double aom_get_ssim_metrics(uint8_t *img1, int img1_pitch, uint8_t *img2,
m->dssim = dssim_total;
return inconsistency_total;
}
#endif // CONFIG_INTERNAL_STATS
double aom_highbd_calc_ssim(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *dest, double *weight,
uint32_t bd, uint32_t in_bd) {
#if CONFIG_AV1_HIGHBITDEPTH
void aom_highbd_ssim_parms_8x8_c(const uint16_t *s, int sp, const uint16_t *r,
int rp, uint32_t *sum_s, uint32_t *sum_r,
uint32_t *sum_sq_s, uint32_t *sum_sq_r,
uint32_t *sum_sxr) {
int i, j;
for (i = 0; i < 8; i++, s += sp, r += rp) {
for (j = 0; j < 8; j++) {
*sum_s += s[j];
*sum_r += r[j];
*sum_sq_s += s[j] * s[j];
*sum_sq_r += r[j] * r[j];
*sum_sxr += s[j] * r[j];
}
}
}
static double highbd_ssim_8x8(const uint16_t *s, int sp, const uint16_t *r,
int rp, uint32_t bd, uint32_t shift) {
uint32_t sum_s = 0, sum_r = 0, sum_sq_s = 0, sum_sq_r = 0, sum_sxr = 0;
aom_highbd_ssim_parms_8x8(s, sp, r, rp, &sum_s, &sum_r, &sum_sq_s, &sum_sq_r,
&sum_sxr);
return similarity(sum_s >> shift, sum_r >> shift, sum_sq_s >> (2 * shift),
sum_sq_r >> (2 * shift), sum_sxr >> (2 * shift), 64, bd);
}
double aom_highbd_ssim2(const uint8_t *img1, const uint8_t *img2,
int stride_img1, int stride_img2, int width, int height,
uint32_t bd, uint32_t shift) {
int i, j;
int samples = 0;
double ssim_total = 0;
// sample point start with each 4x4 location
for (i = 0; i <= height - 8;
i += 4, img1 += stride_img1 * 4, img2 += stride_img2 * 4) {
for (j = 0; j <= width - 8; j += 4) {
double v = highbd_ssim_8x8(CONVERT_TO_SHORTPTR(img1 + j), stride_img1,
CONVERT_TO_SHORTPTR(img2 + j), stride_img2, bd,
shift);
ssim_total += v;
samples++;
}
}
ssim_total /= samples;
return ssim_total;
}
#if CONFIG_INTERNAL_STATS
void aom_highbd_calc_ssim(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *dest, double *weight,
uint32_t bd, uint32_t in_bd, double *fast_ssim) {
assert(bd >= in_bd);
const uint32_t shift = bd - in_bd;
uint32_t shift = bd - in_bd;
double abc[3];
for (int i = 0; i < 3; ++i) {
@ -436,6 +439,43 @@ double aom_highbd_calc_ssim(const YV12_BUFFER_CONFIG *source,
source->crop_heights[is_uv], in_bd, shift);
}
*weight = 1;
return abc[0] * .8 + .1 * (abc[1] + abc[2]);
weight[0] = 1;
fast_ssim[0] = abc[0] * .8 + .1 * (abc[1] + abc[2]);
if (bd > in_bd) {
// Compute SSIM based on stream bit depth
shift = 0;
for (int i = 0; i < 3; ++i) {
const int is_uv = i > 0;
abc[i] = aom_highbd_ssim2(source->buffers[i], dest->buffers[i],
source->strides[is_uv], dest->strides[is_uv],
source->crop_widths[is_uv],
source->crop_heights[is_uv], bd, shift);
}
weight[1] = 1;
fast_ssim[1] = abc[0] * .8 + .1 * (abc[1] + abc[2]);
}
}
#endif // CONFIG_INTERNAL_STATS
#endif // CONFIG_AV1_HIGHBITDEPTH
#if CONFIG_INTERNAL_STATS
void aom_calc_ssim(const YV12_BUFFER_CONFIG *orig,
const YV12_BUFFER_CONFIG *recon, const uint32_t bit_depth,
const uint32_t in_bit_depth, int is_hbd, double *weight,
double *frame_ssim2) {
#if CONFIG_AV1_HIGHBITDEPTH
if (is_hbd) {
aom_highbd_calc_ssim(orig, recon, weight, bit_depth, in_bit_depth,
frame_ssim2);
return;
}
#else
(void)bit_depth;
(void)in_bit_depth;
(void)is_hbd;
#endif // CONFIG_AV1_HIGHBITDEPTH
aom_lowbd_calc_ssim(orig, recon, weight, frame_ssim2);
}
#endif // CONFIG_INTERNAL_STATS

View file

@ -12,14 +12,13 @@
#ifndef AOM_AOM_DSP_SSIM_H_
#define AOM_AOM_DSP_SSIM_H_
#define MAX_SSIM_DB 100.0;
#ifdef __cplusplus
extern "C" {
#endif
#include "config/aom_config.h"
#if CONFIG_INTERNAL_STATS
#include "aom_scale/yv12config.h"
// metrics used for calculating ssim, ssim2, dssim, and ssimc
@ -68,17 +67,35 @@ double aom_get_ssim_metrics(uint8_t *img1, int img1_pitch, uint8_t *img2,
int img2_pitch, int width, int height, Ssimv *sv2,
Metrics *m, int do_inconsistency);
double aom_calc_ssim(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *dest, double *weight);
void aom_lowbd_calc_ssim(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *dest, double *weight,
double *fast_ssim);
double aom_calc_fastssim(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *dest, double *ssim_y,
double *ssim_u, double *ssim_v, uint32_t bd,
uint32_t in_bd);
double aom_highbd_calc_ssim(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *dest, double *weight,
uint32_t bd, uint32_t in_bd);
#if CONFIG_AV1_HIGHBITDEPTH
void aom_highbd_calc_ssim(const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *dest, double *weight,
uint32_t bd, uint32_t in_bd, double *fast_ssim);
#endif // CONFIG_AV1_HIGHBITDEPTH
void aom_calc_ssim(const YV12_BUFFER_CONFIG *orig,
const YV12_BUFFER_CONFIG *recon, const uint32_t bit_depth,
const uint32_t in_bit_depth, int is_hbd, double *weight,
double *frame_ssim2);
#endif // CONFIG_INTERNAL_STATS
double aom_ssim2(const uint8_t *img1, const uint8_t *img2, int stride_img1,
int stride_img2, int width, int height);
#if CONFIG_AV1_HIGHBITDEPTH
double aom_highbd_ssim2(const uint8_t *img1, const uint8_t *img2,
int stride_img1, int stride_img2, int width, int height,
uint32_t bd, uint32_t shift);
#endif // CONFIG_AV1_HIGHBITDEPTH
#ifdef __cplusplus
} // extern "C"

View file

@ -36,11 +36,10 @@ void aom_subtract_block_c(int rows, int cols, int16_t *diff,
void aom_highbd_subtract_block_c(int rows, int cols, int16_t *diff,
ptrdiff_t diff_stride, const uint8_t *src8,
ptrdiff_t src_stride, const uint8_t *pred8,
ptrdiff_t pred_stride, int bd) {
ptrdiff_t pred_stride) {
int r, c;
uint16_t *src = CONVERT_TO_SHORTPTR(src8);
uint16_t *pred = CONVERT_TO_SHORTPTR(pred8);
(void)bd;
for (r = 0; r < rows; r++) {
for (c = 0; c < cols; c++) {

View file

@ -71,3 +71,20 @@ uint64_t aom_var_2d_u16_c(uint8_t *src, int src_stride, int width, int height) {
return (ss - s * s / (width * height));
}
uint64_t aom_sum_sse_2d_i16_c(const int16_t *src, int src_stride, int width,
int height, int *sum) {
int r, c;
int16_t *srcp = (int16_t *)src;
int64_t ss = 0;
for (r = 0; r < height; r++) {
for (c = 0; c < width; c++) {
const int16_t v = srcp[c];
ss += v * v;
*sum += v;
}
srcp += src_stride;
}
return ss;
}

View file

@ -13,7 +13,6 @@
#define AOM_AOM_DSP_TXFM_COMMON_H_
#include "aom_dsp/aom_dsp_common.h"
#include "av1/common/enums.h"
// Constants and Macros used by all idct/dct functions
#define DCT_CONST_BITS 14
@ -22,6 +21,71 @@
#define UNIT_QUANT_SHIFT 2
#define UNIT_QUANT_FACTOR (1 << UNIT_QUANT_SHIFT)
// block transform size
enum {
TX_4X4, // 4x4 transform
TX_8X8, // 8x8 transform
TX_16X16, // 16x16 transform
TX_32X32, // 32x32 transform
TX_64X64, // 64x64 transform
TX_4X8, // 4x8 transform
TX_8X4, // 8x4 transform
TX_8X16, // 8x16 transform
TX_16X8, // 16x8 transform
TX_16X32, // 16x32 transform
TX_32X16, // 32x16 transform
TX_32X64, // 32x64 transform
TX_64X32, // 64x32 transform
TX_4X16, // 4x16 transform
TX_16X4, // 16x4 transform
TX_8X32, // 8x32 transform
TX_32X8, // 32x8 transform
TX_16X64, // 16x64 transform
TX_64X16, // 64x16 transform
TX_SIZES_ALL, // Includes rectangular transforms
TX_SIZES = TX_4X8, // Does NOT include rectangular transforms
TX_SIZES_LARGEST = TX_64X64,
TX_INVALID = 255 // Invalid transform size
} UENUM1BYTE(TX_SIZE);
enum {
DCT_DCT, // DCT in both horizontal and vertical
ADST_DCT, // ADST in vertical, DCT in horizontal
DCT_ADST, // DCT in vertical, ADST in horizontal
ADST_ADST, // ADST in both directions
FLIPADST_DCT, // FLIPADST in vertical, DCT in horizontal
DCT_FLIPADST, // DCT in vertical, FLIPADST in horizontal
FLIPADST_FLIPADST, // FLIPADST in both directions
ADST_FLIPADST, // ADST in vertical, FLIPADST in horizontal
FLIPADST_ADST, // FLIPADST in vertical, ADST in horizontal
IDTX, // Identity in both directions
V_DCT, // DCT in vertical, identity in horizontal
H_DCT, // Identity in vertical, DCT in horizontal
V_ADST, // ADST in vertical, identity in horizontal
H_ADST, // Identity in vertical, ADST in horizontal
V_FLIPADST, // FLIPADST in vertical, identity in horizontal
H_FLIPADST, // Identity in vertical, FLIPADST in horizontal
TX_TYPES,
DCT_ADST_TX_MASK = 0x000F, // Either DCT or ADST in each direction
TX_TYPE_INVALID = 255, // Invalid transform type
} UENUM1BYTE(TX_TYPE);
enum {
// DCT only
EXT_TX_SET_DCTONLY,
// DCT + Identity only
EXT_TX_SET_DCT_IDTX,
// Discrete Trig transforms w/o flip (4) + Identity (1)
EXT_TX_SET_DTT4_IDTX,
// Discrete Trig transforms w/o flip (4) + Identity (1) + 1D Hor/vert DCT (2)
EXT_TX_SET_DTT4_IDTX_1DDCT,
// Discrete Trig transforms w/ flip (9) + Identity (1) + 1D Hor/Ver DCT (2)
EXT_TX_SET_DTT9_IDTX_1DDCT,
// Discrete Trig transforms w/ flip (9) + Identity (1) + 1D Hor/Ver (6)
EXT_TX_SET_ALL16,
EXT_TX_SET_TYPES
} UENUM1BYTE(TxSetType);
typedef struct txfm_param {
// for both forward and inverse transforms
TX_TYPE tx_type;

View file

@ -14,7 +14,6 @@
#include "config/aom_config.h"
#include "config/aom_dsp_rtcd.h"
#include "config/av1_rtcd.h"
#include "aom/aom_integer.h"
#include "aom_ports/mem.h"
@ -23,10 +22,8 @@
#include "aom_dsp/blend.h"
#include "aom_dsp/variance.h"
#include "av1/common/av1_common_int.h"
#include "av1/common/filter.h"
#include "av1/common/reconinter.h"
#include "av1/encoder/reconinter_enc.h"
uint32_t aom_get4x4sse_cs_c(const uint8_t *a, int a_stride, const uint8_t *b,
int b_stride) {
@ -212,6 +209,16 @@ void aom_var_filter_block2d_bil_second_pass_c(const uint16_t *a, uint8_t *b,
variance(a, a_stride, b, b_stride, W, H, sse, sum); \
}
void aom_get_sse_sum_8x8_quad_c(const uint8_t *a, int a_stride,
const uint8_t *b, int b_stride, uint32_t *sse,
int *sum) {
// Loop over 4 8x8 blocks. Process one 8x32 block.
for (int k = 0; k < 4; k++) {
variance(a + (k * 8), a_stride, b + (k * 8), b_stride, 8, 8, &sse[k],
&sum[k]);
}
}
/* Identical to the variance call except it does not calculate the
* sse - sum^2 / w*h and returns sse in addtion to modifying the passed in
* variable.
@ -250,12 +257,16 @@ VARIANCES(4, 4)
VARIANCES(4, 2)
VARIANCES(2, 4)
VARIANCES(2, 2)
// Realtime mode doesn't use rectangular blocks.
#if !CONFIG_REALTIME_ONLY
VARIANCES(4, 16)
VARIANCES(16, 4)
VARIANCES(8, 32)
VARIANCES(32, 8)
VARIANCES(16, 64)
VARIANCES(64, 16)
#endif
GET_VAR(16, 16)
GET_VAR(8, 8)
@ -280,100 +291,6 @@ void aom_comp_avg_pred_c(uint8_t *comp_pred, const uint8_t *pred, int width,
}
}
// Get pred block from up-sampled reference.
void aom_upsampled_pred_c(MACROBLOCKD *xd, const AV1_COMMON *const cm,
int mi_row, int mi_col, const MV *const mv,
uint8_t *comp_pred, int width, int height,
int subpel_x_q3, int subpel_y_q3, const uint8_t *ref,
int ref_stride, int subpel_search) {
// expect xd == NULL only in tests
if (xd != NULL) {
const MB_MODE_INFO *mi = xd->mi[0];
const int ref_num = 0;
const int is_intrabc = is_intrabc_block(mi);
const struct scale_factors *const sf =
is_intrabc ? &cm->sf_identity : xd->block_ref_scale_factors[ref_num];
const int is_scaled = av1_is_scaled(sf);
if (is_scaled) {
int plane = 0;
const int mi_x = mi_col * MI_SIZE;
const int mi_y = mi_row * MI_SIZE;
const struct macroblockd_plane *const pd = &xd->plane[plane];
const struct buf_2d *const dst_buf = &pd->dst;
const struct buf_2d *const pre_buf =
is_intrabc ? dst_buf : &pd->pre[ref_num];
InterPredParams inter_pred_params;
inter_pred_params.conv_params = get_conv_params(0, plane, xd->bd);
const int_interpfilters filters =
av1_broadcast_interp_filter(EIGHTTAP_REGULAR);
av1_init_inter_params(
&inter_pred_params, width, height, mi_y >> pd->subsampling_y,
mi_x >> pd->subsampling_x, pd->subsampling_x, pd->subsampling_y,
xd->bd, is_cur_buf_hbd(xd), is_intrabc, sf, pre_buf, filters);
av1_enc_build_one_inter_predictor(comp_pred, width, mv,
&inter_pred_params);
return;
}
}
const InterpFilterParams *filter = av1_get_filter(subpel_search);
if (!subpel_x_q3 && !subpel_y_q3) {
for (int i = 0; i < height; i++) {
memcpy(comp_pred, ref, width * sizeof(*comp_pred));
comp_pred += width;
ref += ref_stride;
}
} else if (!subpel_y_q3) {
const int16_t *const kernel =
av1_get_interp_filter_subpel_kernel(filter, subpel_x_q3 << 1);
aom_convolve8_horiz_c(ref, ref_stride, comp_pred, width, kernel, 16, NULL,
-1, width, height);
} else if (!subpel_x_q3) {
const int16_t *const kernel =
av1_get_interp_filter_subpel_kernel(filter, subpel_y_q3 << 1);
aom_convolve8_vert_c(ref, ref_stride, comp_pred, width, NULL, -1, kernel,
16, width, height);
} else {
DECLARE_ALIGNED(16, uint8_t,
temp[((MAX_SB_SIZE * 2 + 16) + 16) * MAX_SB_SIZE]);
const int16_t *const kernel_x =
av1_get_interp_filter_subpel_kernel(filter, subpel_x_q3 << 1);
const int16_t *const kernel_y =
av1_get_interp_filter_subpel_kernel(filter, subpel_y_q3 << 1);
const int intermediate_height =
(((height - 1) * 8 + subpel_y_q3) >> 3) + filter->taps;
assert(intermediate_height <= (MAX_SB_SIZE * 2 + 16) + 16);
aom_convolve8_horiz_c(ref - ref_stride * ((filter->taps >> 1) - 1),
ref_stride, temp, MAX_SB_SIZE, kernel_x, 16, NULL, -1,
width, intermediate_height);
aom_convolve8_vert_c(temp + MAX_SB_SIZE * ((filter->taps >> 1) - 1),
MAX_SB_SIZE, comp_pred, width, NULL, -1, kernel_y, 16,
width, height);
}
}
void aom_comp_avg_upsampled_pred_c(MACROBLOCKD *xd, const AV1_COMMON *const cm,
int mi_row, int mi_col, const MV *const mv,
uint8_t *comp_pred, const uint8_t *pred,
int width, int height, int subpel_x_q3,
int subpel_y_q3, const uint8_t *ref,
int ref_stride, int subpel_search) {
int i, j;
aom_upsampled_pred(xd, cm, mi_row, mi_col, mv, comp_pred, width, height,
subpel_x_q3, subpel_y_q3, ref, ref_stride, subpel_search);
for (i = 0; i < height; i++) {
for (j = 0; j < width; j++) {
comp_pred[j] = ROUND_POWER_OF_TWO(comp_pred[j] + pred[j], 1);
}
comp_pred += width;
pred += width;
}
}
void aom_dist_wtd_comp_avg_pred_c(uint8_t *comp_pred, const uint8_t *pred,
int width, int height, const uint8_t *ref,
int ref_stride,
@ -394,30 +311,6 @@ void aom_dist_wtd_comp_avg_pred_c(uint8_t *comp_pred, const uint8_t *pred,
}
}
void aom_dist_wtd_comp_avg_upsampled_pred_c(
MACROBLOCKD *xd, const AV1_COMMON *const cm, int mi_row, int mi_col,
const MV *const mv, uint8_t *comp_pred, const uint8_t *pred, int width,
int height, int subpel_x_q3, int subpel_y_q3, const uint8_t *ref,
int ref_stride, const DIST_WTD_COMP_PARAMS *jcp_param, int subpel_search) {
int i, j;
const int fwd_offset = jcp_param->fwd_offset;
const int bck_offset = jcp_param->bck_offset;
aom_upsampled_pred_c(xd, cm, mi_row, mi_col, mv, comp_pred, width, height,
subpel_x_q3, subpel_y_q3, ref, ref_stride,
subpel_search);
for (i = 0; i < height; i++) {
for (j = 0; j < width; j++) {
int tmp = pred[j] * bck_offset + comp_pred[j] * fwd_offset;
tmp = ROUND_POWER_OF_TWO(tmp, DIST_PRECISION_BITS);
comp_pred[j] = (uint8_t)tmp;
}
comp_pred += width;
pred += width;
}
}
#if CONFIG_AV1_HIGHBITDEPTH
static void highbd_variance64(const uint8_t *a8, int a_stride,
const uint8_t *b8, int b_stride, int w, int h,
@ -789,12 +682,16 @@ HIGHBD_VARIANCES(4, 4)
HIGHBD_VARIANCES(4, 2)
HIGHBD_VARIANCES(2, 4)
HIGHBD_VARIANCES(2, 2)
// Realtime mode doesn't use 4x rectangular blocks.
#if !CONFIG_REALTIME_ONLY
HIGHBD_VARIANCES(4, 16)
HIGHBD_VARIANCES(16, 4)
HIGHBD_VARIANCES(8, 32)
HIGHBD_VARIANCES(32, 8)
HIGHBD_VARIANCES(16, 64)
HIGHBD_VARIANCES(64, 16)
#endif
HIGHBD_GET_VAR(8)
HIGHBD_GET_VAR(16)
@ -822,107 +719,6 @@ void aom_highbd_comp_avg_pred_c(uint8_t *comp_pred8, const uint8_t *pred8,
}
}
void aom_highbd_upsampled_pred_c(MACROBLOCKD *xd,
const struct AV1Common *const cm, int mi_row,
int mi_col, const MV *const mv,
uint8_t *comp_pred8, int width, int height,
int subpel_x_q3, int subpel_y_q3,
const uint8_t *ref8, int ref_stride, int bd,
int subpel_search) {
// expect xd == NULL only in tests
if (xd != NULL) {
const MB_MODE_INFO *mi = xd->mi[0];
const int ref_num = 0;
const int is_intrabc = is_intrabc_block(mi);
const struct scale_factors *const sf =
is_intrabc ? &cm->sf_identity : xd->block_ref_scale_factors[ref_num];
const int is_scaled = av1_is_scaled(sf);
if (is_scaled) {
int plane = 0;
const int mi_x = mi_col * MI_SIZE;
const int mi_y = mi_row * MI_SIZE;
const struct macroblockd_plane *const pd = &xd->plane[plane];
const struct buf_2d *const dst_buf = &pd->dst;
const struct buf_2d *const pre_buf =
is_intrabc ? dst_buf : &pd->pre[ref_num];
InterPredParams inter_pred_params;
inter_pred_params.conv_params = get_conv_params(0, plane, xd->bd);
const int_interpfilters filters =
av1_broadcast_interp_filter(EIGHTTAP_REGULAR);
av1_init_inter_params(
&inter_pred_params, width, height, mi_y >> pd->subsampling_y,
mi_x >> pd->subsampling_x, pd->subsampling_x, pd->subsampling_y,
xd->bd, is_cur_buf_hbd(xd), is_intrabc, sf, pre_buf, filters);
av1_enc_build_one_inter_predictor(comp_pred8, width, mv,
&inter_pred_params);
return;
}
}
const InterpFilterParams *filter = av1_get_filter(subpel_search);
if (!subpel_x_q3 && !subpel_y_q3) {
const uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);
uint16_t *comp_pred = CONVERT_TO_SHORTPTR(comp_pred8);
for (int i = 0; i < height; i++) {
memcpy(comp_pred, ref, width * sizeof(*comp_pred));
comp_pred += width;
ref += ref_stride;
}
} else if (!subpel_y_q3) {
const int16_t *const kernel =
av1_get_interp_filter_subpel_kernel(filter, subpel_x_q3 << 1);
aom_highbd_convolve8_horiz_c(ref8, ref_stride, comp_pred8, width, kernel,
16, NULL, -1, width, height, bd);
} else if (!subpel_x_q3) {
const int16_t *const kernel =
av1_get_interp_filter_subpel_kernel(filter, subpel_y_q3 << 1);
aom_highbd_convolve8_vert_c(ref8, ref_stride, comp_pred8, width, NULL, -1,
kernel, 16, width, height, bd);
} else {
DECLARE_ALIGNED(16, uint16_t,
temp[((MAX_SB_SIZE + 16) + 16) * MAX_SB_SIZE]);
const int16_t *const kernel_x =
av1_get_interp_filter_subpel_kernel(filter, subpel_x_q3 << 1);
const int16_t *const kernel_y =
av1_get_interp_filter_subpel_kernel(filter, subpel_y_q3 << 1);
const int intermediate_height =
(((height - 1) * 8 + subpel_y_q3) >> 3) + filter->taps;
assert(intermediate_height <= (MAX_SB_SIZE * 2 + 16) + 16);
aom_highbd_convolve8_horiz_c(ref8 - ref_stride * ((filter->taps >> 1) - 1),
ref_stride, CONVERT_TO_BYTEPTR(temp),
MAX_SB_SIZE, kernel_x, 16, NULL, -1, width,
intermediate_height, bd);
aom_highbd_convolve8_vert_c(
CONVERT_TO_BYTEPTR(temp + MAX_SB_SIZE * ((filter->taps >> 1) - 1)),
MAX_SB_SIZE, comp_pred8, width, NULL, -1, kernel_y, 16, width, height,
bd);
}
}
void aom_highbd_comp_avg_upsampled_pred_c(
MACROBLOCKD *xd, const struct AV1Common *const cm, int mi_row, int mi_col,
const MV *const mv, uint8_t *comp_pred8, const uint8_t *pred8, int width,
int height, int subpel_x_q3, int subpel_y_q3, const uint8_t *ref8,
int ref_stride, int bd, int subpel_search) {
int i, j;
const uint16_t *pred = CONVERT_TO_SHORTPTR(pred8);
uint16_t *comp_pred = CONVERT_TO_SHORTPTR(comp_pred8);
aom_highbd_upsampled_pred(xd, cm, mi_row, mi_col, mv, comp_pred8, width,
height, subpel_x_q3, subpel_y_q3, ref8, ref_stride,
bd, subpel_search);
for (i = 0; i < height; ++i) {
for (j = 0; j < width; ++j) {
comp_pred[j] = ROUND_POWER_OF_TWO(pred[j] + comp_pred[j], 1);
}
comp_pred += width;
pred += width;
}
}
void aom_highbd_dist_wtd_comp_avg_pred_c(
uint8_t *comp_pred8, const uint8_t *pred8, int width, int height,
const uint8_t *ref8, int ref_stride,
@ -945,32 +741,6 @@ void aom_highbd_dist_wtd_comp_avg_pred_c(
ref += ref_stride;
}
}
void aom_highbd_dist_wtd_comp_avg_upsampled_pred_c(
MACROBLOCKD *xd, const struct AV1Common *const cm, int mi_row, int mi_col,
const MV *const mv, uint8_t *comp_pred8, const uint8_t *pred8, int width,
int height, int subpel_x_q3, int subpel_y_q3, const uint8_t *ref8,
int ref_stride, int bd, const DIST_WTD_COMP_PARAMS *jcp_param,
int subpel_search) {
int i, j;
const int fwd_offset = jcp_param->fwd_offset;
const int bck_offset = jcp_param->bck_offset;
const uint16_t *pred = CONVERT_TO_SHORTPTR(pred8);
uint16_t *comp_pred = CONVERT_TO_SHORTPTR(comp_pred8);
aom_highbd_upsampled_pred_c(xd, cm, mi_row, mi_col, mv, comp_pred8, width,
height, subpel_x_q3, subpel_y_q3, ref8,
ref_stride, bd, subpel_search);
for (i = 0; i < height; i++) {
for (j = 0; j < width; j++) {
int tmp = pred[j] * bck_offset + comp_pred[j] * fwd_offset;
tmp = ROUND_POWER_OF_TWO(tmp, DIST_PRECISION_BITS);
comp_pred[j] = (uint16_t)tmp;
}
comp_pred += width;
pred += width;
}
}
#endif // CONFIG_AV1_HIGHBITDEPTH
void aom_comp_mask_pred_c(uint8_t *comp_pred, const uint8_t *pred, int width,
@ -993,25 +763,6 @@ void aom_comp_mask_pred_c(uint8_t *comp_pred, const uint8_t *pred, int width,
}
}
void aom_comp_mask_upsampled_pred_c(MACROBLOCKD *xd, const AV1_COMMON *const cm,
int mi_row, int mi_col, const MV *const mv,
uint8_t *comp_pred, const uint8_t *pred,
int width, int height, int subpel_x_q3,
int subpel_y_q3, const uint8_t *ref,
int ref_stride, const uint8_t *mask,
int mask_stride, int invert_mask,
int subpel_search) {
if (subpel_x_q3 | subpel_y_q3) {
aom_upsampled_pred_c(xd, cm, mi_row, mi_col, mv, comp_pred, width, height,
subpel_x_q3, subpel_y_q3, ref, ref_stride,
subpel_search);
ref = comp_pred;
ref_stride = width;
}
aom_comp_mask_pred_c(comp_pred, pred, width, height, ref, ref_stride, mask,
mask_stride, invert_mask);
}
#define MASK_SUBPIX_VAR(W, H) \
unsigned int aom_masked_sub_pixel_variance##W##x##H##_c( \
const uint8_t *src, int src_stride, int xoffset, int yoffset, \
@ -1048,12 +799,16 @@ MASK_SUBPIX_VAR(64, 64)
MASK_SUBPIX_VAR(64, 128)
MASK_SUBPIX_VAR(128, 64)
MASK_SUBPIX_VAR(128, 128)
// Realtime mode doesn't use 4x rectangular blocks.
#if !CONFIG_REALTIME_ONLY
MASK_SUBPIX_VAR(4, 16)
MASK_SUBPIX_VAR(16, 4)
MASK_SUBPIX_VAR(8, 32)
MASK_SUBPIX_VAR(32, 8)
MASK_SUBPIX_VAR(16, 64)
MASK_SUBPIX_VAR(64, 16)
#endif
#if CONFIG_AV1_HIGHBITDEPTH
void aom_highbd_comp_mask_pred_c(uint8_t *comp_pred8, const uint8_t *pred8,
@ -1078,19 +833,6 @@ void aom_highbd_comp_mask_pred_c(uint8_t *comp_pred8, const uint8_t *pred8,
}
}
void aom_highbd_comp_mask_upsampled_pred(
MACROBLOCKD *xd, const struct AV1Common *const cm, int mi_row, int mi_col,
const MV *const mv, uint8_t *comp_pred8, const uint8_t *pred8, int width,
int height, int subpel_x_q3, int subpel_y_q3, const uint8_t *ref8,
int ref_stride, const uint8_t *mask, int mask_stride, int invert_mask,
int bd, int subpel_search) {
aom_highbd_upsampled_pred(xd, cm, mi_row, mi_col, mv, comp_pred8, width,
height, subpel_x_q3, subpel_y_q3, ref8, ref_stride,
bd, subpel_search);
aom_highbd_comp_mask_pred(comp_pred8, pred8, width, height, comp_pred8, width,
mask, mask_stride, invert_mask);
}
#define HIGHBD_MASK_SUBPIX_VAR(W, H) \
unsigned int aom_highbd_8_masked_sub_pixel_variance##W##x##H##_c( \
const uint8_t *src, int src_stride, int xoffset, int yoffset, \
@ -1174,14 +916,17 @@ HIGHBD_MASK_SUBPIX_VAR(64, 64)
HIGHBD_MASK_SUBPIX_VAR(64, 128)
HIGHBD_MASK_SUBPIX_VAR(128, 64)
HIGHBD_MASK_SUBPIX_VAR(128, 128)
#if !CONFIG_REALTIME_ONLY
HIGHBD_MASK_SUBPIX_VAR(4, 16)
HIGHBD_MASK_SUBPIX_VAR(16, 4)
HIGHBD_MASK_SUBPIX_VAR(8, 32)
HIGHBD_MASK_SUBPIX_VAR(32, 8)
HIGHBD_MASK_SUBPIX_VAR(16, 64)
HIGHBD_MASK_SUBPIX_VAR(64, 16)
#endif
#endif // CONFIG_AV1_HIGHBITDEPTH
#if !CONFIG_REALTIME_ONLY
static INLINE void obmc_variance(const uint8_t *pre, int pre_stride,
const int32_t *wsrc, const int32_t *mask,
int w, int h, unsigned int *sse, int *sum) {
@ -1481,3 +1226,28 @@ HIGHBD_OBMC_SUBPIX_VAR(16, 64)
HIGHBD_OBMC_VAR(64, 16)
HIGHBD_OBMC_SUBPIX_VAR(64, 16)
#endif // CONFIG_AV1_HIGHBITDEPTH
#endif // !CONFIG_REALTIME_ONLY
uint64_t aom_mse_wxh_16bit_c(uint8_t *dst, int dstride, uint16_t *src,
int sstride, int w, int h) {
uint64_t sum = 0;
for (int i = 0; i < h; i++) {
for (int j = 0; j < w; j++) {
int e = (uint16_t)dst[i * dstride + j] - src[i * sstride + j];
sum += e * e;
}
}
return sum;
}
uint64_t aom_mse_wxh_16bit_highbd_c(uint16_t *dst, int dstride, uint16_t *src,
int sstride, int w, int h) {
uint64_t sum = 0;
for (int i = 0; i < h; i++) {
for (int j = 0; j < w; j++) {
int e = dst[i * dstride + j] - src[i * sstride + j];
sum += e * e;
}
}
return sum;
}

View file

@ -69,13 +69,6 @@ typedef unsigned int (*aom_masked_subpixvariance_fn_t)(
const uint8_t *ref, int ref_stride, const uint8_t *second_pred,
const uint8_t *msk, int msk_stride, int invert_mask, unsigned int *sse);
void aom_highbd_comp_mask_upsampled_pred(
MACROBLOCKD *xd, const struct AV1Common *const cm, int mi_row, int mi_col,
const MV *const mv, uint8_t *comp_pred8, const uint8_t *pred8, int width,
int height, int subpel_x_q3, int subpel_y_q3, const uint8_t *ref8,
int ref_stride, const uint8_t *mask, int mask_stride, int invert_mask,
int bd, int subpel_search);
typedef unsigned int (*aom_obmc_sad_fn_t)(const uint8_t *pred, int pred_stride,
const int32_t *wsrc,
const int32_t *msk);
@ -90,11 +83,15 @@ typedef unsigned int (*aom_obmc_subpixvariance_fn_t)(
typedef struct aom_variance_vtable {
aom_sad_fn_t sdf;
// Same as normal sad, but downsample the rows by a factor of 2.
aom_sad_fn_t sdsf;
aom_sad_avg_fn_t sdaf;
aom_variance_fn_t vf;
aom_subpixvariance_fn_t svf;
aom_subp_avg_variance_fn_t svaf;
aom_sad_multi_d_fn_t sdx4df;
// Same as sadx4, but downsample the rows by a factor of 2.
aom_sad_multi_d_fn_t sdsx4df;
aom_masked_sad_fn_t msdf;
aom_masked_subpixvariance_fn_t msvf;
aom_obmc_sad_fn_t osdf;

View file

@ -9,151 +9,184 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include "aom_dsp/vmaf.h"
#include <assert.h>
#include <libvmaf/libvmaf.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#ifdef _WIN32
#include <process.h>
#else
#include <unistd.h>
#endif
#include "aom_dsp/blend.h"
#include "aom_dsp/vmaf.h"
#include "aom_ports/system_state.h"
typedef struct FrameData {
const YV12_BUFFER_CONFIG *source;
const YV12_BUFFER_CONFIG *distorted;
int frame_set;
int bit_depth;
} FrameData;
static void vmaf_fatal_error(const char *message) {
fprintf(stderr, "Fatal error: %s\n", message);
exit(EXIT_FAILURE);
}
// A callback function used to pass data to VMAF.
// Returns 0 after reading a frame.
// Returns 2 when there is no more frame to read.
static int read_frame(float *ref_data, float *main_data, float *temp_data,
int stride, void *user_data) {
FrameData *frames = (FrameData *)user_data;
void aom_init_vmaf_model(VmafModel **vmaf_model, const char *model_path) {
if (*vmaf_model != NULL) return;
VmafModelConfig model_cfg;
model_cfg.flags = VMAF_MODEL_FLAG_DISABLE_CLIP;
model_cfg.name = "vmaf";
if (!frames->frame_set) {
const int width = frames->source->y_width;
const int height = frames->source->y_height;
assert(width == frames->distorted->y_width);
assert(height == frames->distorted->y_height);
if (vmaf_model_load_from_path(vmaf_model, &model_cfg, model_path)) {
vmaf_fatal_error("Failed to load VMAF model.");
}
}
if (frames->bit_depth > 8) {
const float scale_factor = 1.0f / (float)(1 << (frames->bit_depth - 8));
uint16_t *ref_ptr = CONVERT_TO_SHORTPTR(frames->source->y_buffer);
uint16_t *main_ptr = CONVERT_TO_SHORTPTR(frames->distorted->y_buffer);
void aom_close_vmaf_model(VmafModel *vmaf_model) {
vmaf_model_destroy(vmaf_model);
}
for (int row = 0; row < height; ++row) {
for (int col = 0; col < width; ++col) {
ref_data[col] = scale_factor * (float)ref_ptr[col];
}
ref_ptr += frames->source->y_stride;
ref_data += stride / sizeof(*ref_data);
}
static void copy_picture(const int bit_depth, const YV12_BUFFER_CONFIG *src,
VmafPicture *dst) {
const int width = src->y_width;
const int height = src->y_height;
for (int row = 0; row < height; ++row) {
for (int col = 0; col < width; ++col) {
main_data[col] = scale_factor * (float)main_ptr[col];
}
main_ptr += frames->distorted->y_stride;
main_data += stride / sizeof(*main_data);
}
} else {
uint8_t *ref_ptr = frames->source->y_buffer;
uint8_t *main_ptr = frames->distorted->y_buffer;
if (bit_depth > 8) {
uint16_t *src_ptr = CONVERT_TO_SHORTPTR(src->y_buffer);
uint16_t *dst_ptr = dst->data[0];
for (int row = 0; row < height; ++row) {
for (int col = 0; col < width; ++col) {
ref_data[col] = (float)ref_ptr[col];
}
ref_ptr += frames->source->y_stride;
ref_data += stride / sizeof(*ref_data);
}
for (int row = 0; row < height; ++row) {
for (int col = 0; col < width; ++col) {
main_data[col] = (float)main_ptr[col];
}
main_ptr += frames->distorted->y_stride;
main_data += stride / sizeof(*main_data);
}
for (int row = 0; row < height; ++row) {
memcpy(dst_ptr, src_ptr, width * sizeof(dst_ptr[0]));
src_ptr += src->y_stride;
dst_ptr += dst->stride[0] / 2;
}
frames->frame_set = 1;
return 0;
}
} else {
uint8_t *src_ptr = src->y_buffer;
uint8_t *dst_ptr = (uint8_t *)dst->data[0];
(void)temp_data;
return 2;
}
void aom_calc_vmaf(const char *model_path, const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *distorted, const int bit_depth,
double *const vmaf) {
aom_clear_system_state();
const int width = source->y_width;
const int height = source->y_height;
FrameData frames = { source, distorted, 0, bit_depth };
char *fmt = bit_depth == 10 ? "yuv420p10le" : "yuv420p";
double vmaf_score;
const int ret =
compute_vmaf(&vmaf_score, fmt, width, height, read_frame,
/*user_data=*/&frames, (char *)model_path,
/*log_path=*/NULL, /*log_fmt=*/NULL, /*disable_clip=*/1,
/*disable_avx=*/0, /*enable_transform=*/0,
/*phone_model=*/0, /*do_psnr=*/0, /*do_ssim=*/0,
/*do_ms_ssim=*/0, /*pool_method=*/NULL, /*n_thread=*/0,
/*n_subsample=*/1, /*enable_conf_interval=*/0);
if (ret) vmaf_fatal_error("Failed to compute VMAF scores.");
aom_clear_system_state();
*vmaf = vmaf_score;
}
void aom_calc_vmaf_multi_frame(
void *user_data, const char *model_path,
int (*read_frame)(float *ref_data, float *main_data, float *temp_data,
int stride_byte, void *user_data),
int frame_width, int frame_height, int bit_depth, double *vmaf) {
aom_clear_system_state();
char *fmt = bit_depth == 10 ? "yuv420p10le" : "yuv420p";
double vmaf_score;
const int ret = compute_vmaf(
&vmaf_score, fmt, frame_width, frame_height, read_frame,
/*user_data=*/user_data, (char *)model_path,
/*log_path=*/"vmaf_scores.xml", /*log_fmt=*/NULL, /*disable_clip=*/0,
/*disable_avx=*/0, /*enable_transform=*/0,
/*phone_model=*/0, /*do_psnr=*/0, /*do_ssim=*/0,
/*do_ms_ssim=*/0, /*pool_method=*/NULL, /*n_thread=*/0,
/*n_subsample=*/1, /*enable_conf_interval=*/0);
FILE *vmaf_log = fopen("vmaf_scores.xml", "r");
if (vmaf_log == NULL || ret) {
vmaf_fatal_error("Failed to compute VMAF scores.");
}
int frame_index = 0;
char buf[512];
while (fgets(buf, 511, vmaf_log) != NULL) {
if (memcmp(buf, "\t\t<frame ", 9) == 0) {
char *p = strstr(buf, "vmaf=");
if (p != NULL && p[5] == '"') {
char *p2 = strstr(&p[6], "\"");
*p2 = '\0';
const double score = atof(&p[6]);
if (score < 0.0 || score > 100.0) {
vmaf_fatal_error("Failed to compute VMAF scores.");
}
vmaf[frame_index++] = score;
}
for (int row = 0; row < height; ++row) {
memcpy(dst_ptr, src_ptr, width * sizeof(dst_ptr[0]));
src_ptr += src->y_stride;
dst_ptr += dst->stride[0];
}
}
fclose(vmaf_log);
aom_clear_system_state();
}
void aom_init_vmaf_context(VmafContext **vmaf_context, VmafModel *vmaf_model,
bool cal_vmaf_neg) {
// TODO(sdeng): make them CLI arguments.
VmafConfiguration cfg;
cfg.log_level = VMAF_LOG_LEVEL_NONE;
cfg.n_threads = 0;
cfg.n_subsample = 0;
cfg.cpumask = 0;
if (vmaf_init(vmaf_context, cfg)) {
vmaf_fatal_error("Failed to init VMAF context.");
}
if (cal_vmaf_neg) {
VmafFeatureDictionary *vif_feature = NULL;
if (vmaf_feature_dictionary_set(&vif_feature, "vif_enhn_gain_limit",
"1.0")) {
vmaf_fatal_error("Failed to set vif_enhn_gain_limit.");
}
if (vmaf_model_feature_overload(vmaf_model, "float_vif", vif_feature)) {
vmaf_fatal_error("Failed to use feature float_vif.");
}
VmafFeatureDictionary *adm_feature = NULL;
if (vmaf_feature_dictionary_set(&adm_feature, "adm_enhn_gain_limit",
"1.0")) {
vmaf_fatal_error("Failed to set adm_enhn_gain_limit.");
}
if (vmaf_model_feature_overload(vmaf_model, "adm", adm_feature)) {
vmaf_fatal_error("Failed to use feature float_adm.");
}
}
VmafFeatureDictionary *motion_force_zero = NULL;
if (vmaf_feature_dictionary_set(&motion_force_zero, "motion_force_zero",
"1")) {
vmaf_fatal_error("Failed to set motion_force_zero.");
}
if (vmaf_model_feature_overload(vmaf_model, "float_motion",
motion_force_zero)) {
vmaf_fatal_error("Failed to use feature float_motion.");
}
if (vmaf_use_features_from_model(*vmaf_context, vmaf_model)) {
vmaf_fatal_error("Failed to load feature extractors from VMAF model.");
}
}
void aom_close_vmaf_context(VmafContext *vmaf_context) {
if (vmaf_close(vmaf_context)) {
vmaf_fatal_error("Failed to close VMAF context.");
}
}
void aom_calc_vmaf(VmafModel *vmaf_model, const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *distorted, int bit_depth,
bool cal_vmaf_neg, double *vmaf) {
VmafContext *vmaf_context;
aom_init_vmaf_context(&vmaf_context, vmaf_model, cal_vmaf_neg);
const int frame_index = 0;
VmafPicture ref, dist;
if (vmaf_picture_alloc(&ref, VMAF_PIX_FMT_YUV420P, bit_depth, source->y_width,
source->y_height) ||
vmaf_picture_alloc(&dist, VMAF_PIX_FMT_YUV420P, bit_depth,
source->y_width, source->y_height)) {
vmaf_fatal_error("Failed to alloc VMAF pictures.");
}
copy_picture(bit_depth, source, &ref);
copy_picture(bit_depth, distorted, &dist);
if (vmaf_read_pictures(vmaf_context, &ref, &dist,
/*picture index=*/frame_index)) {
vmaf_fatal_error("Failed to read VMAF pictures.");
}
if (vmaf_read_pictures(vmaf_context, NULL, NULL, 0)) {
vmaf_fatal_error("Failed to flush context.");
}
vmaf_picture_unref(&ref);
vmaf_picture_unref(&dist);
vmaf_score_at_index(vmaf_context, vmaf_model, vmaf, frame_index);
aom_close_vmaf_context(vmaf_context);
}
void aom_read_vmaf_image(VmafContext *vmaf_context,
const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *distorted, int bit_depth,
int frame_index) {
VmafPicture ref, dist;
if (vmaf_picture_alloc(&ref, VMAF_PIX_FMT_YUV420P, bit_depth, source->y_width,
source->y_height) ||
vmaf_picture_alloc(&dist, VMAF_PIX_FMT_YUV420P, bit_depth,
source->y_width, source->y_height)) {
vmaf_fatal_error("Failed to alloc VMAF pictures.");
}
copy_picture(bit_depth, source, &ref);
copy_picture(bit_depth, distorted, &dist);
if (vmaf_read_pictures(vmaf_context, &ref, &dist,
/*picture index=*/frame_index)) {
vmaf_fatal_error("Failed to read VMAF pictures.");
}
vmaf_picture_unref(&ref);
vmaf_picture_unref(&dist);
}
double aom_calc_vmaf_at_index(VmafContext *vmaf_context, VmafModel *vmaf_model,
int frame_index) {
double vmaf;
if (vmaf_score_at_index(vmaf_context, vmaf_model, &vmaf, frame_index)) {
vmaf_fatal_error("Failed to calc VMAF scores.");
}
return vmaf;
}
void aom_flush_vmaf_context(VmafContext *vmaf_context) {
if (vmaf_read_pictures(vmaf_context, NULL, NULL, 0)) {
vmaf_fatal_error("Failed to flush context.");
}
}

View file

@ -12,16 +12,30 @@
#ifndef AOM_AOM_DSP_VMAF_H_
#define AOM_AOM_DSP_VMAF_H_
#include <libvmaf/libvmaf.h>
#include <stdbool.h>
#include "aom_scale/yv12config.h"
void aom_calc_vmaf(const char *model_path, const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *distorted, int bit_depth,
double *vmaf);
void aom_init_vmaf_context(VmafContext **vmaf_context, VmafModel *vmaf_model,
bool cal_vmaf_neg);
void aom_close_vmaf_context(VmafContext *vmaf_context);
void aom_calc_vmaf_multi_frame(
void *user_data, const char *model_path,
int (*read_frame)(float *ref_data, float *main_data, float *temp_data,
int stride_byte, void *user_data),
int frame_width, int frame_height, int bit_depth, double *vmaf);
void aom_init_vmaf_model(VmafModel **vmaf_model, const char *model_path);
void aom_close_vmaf_model(VmafModel *vmaf_model);
void aom_calc_vmaf(VmafModel *vmaf_model, const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *distorted, int bit_depth,
bool cal_vmaf_neg, double *vmaf);
void aom_read_vmaf_image(VmafContext *vmaf_context,
const YV12_BUFFER_CONFIG *source,
const YV12_BUFFER_CONFIG *distorted, int bit_depth,
int frame_index);
double aom_calc_vmaf_at_index(VmafContext *vmaf_context, VmafModel *vmaf_model,
int frame_index);
void aom_flush_vmaf_context(VmafContext *vmaf_context);
#endif // AOM_AOM_DSP_VMAF_H_

View file

@ -12,7 +12,7 @@
#include <immintrin.h>
#include "config/aom_dsp_rtcd.h"
#include "aom/aom_integer.h"
#include "av1/encoder/av1_quantize.h"
#include "aom_dsp/quantize.h"
#include "aom_dsp/x86/quantize_x86.h"
static INLINE void load_b_values_avx2(const int16_t *zbin_ptr, __m256i *zbin,

View file

@ -13,7 +13,7 @@
#include <emmintrin.h>
#include "config/aom_dsp_rtcd.h"
#include "aom/aom_integer.h"
#include "av1/encoder/av1_quantize.h"
#include "aom_dsp/quantize.h"
#include "aom_dsp/x86/quantize_x86.h"
void aom_quantize_b_adaptive_sse2(

View file

@ -46,8 +46,8 @@ filter8_1dfunction aom_filter_block1d4_h2_sse2;
// const int16_t *filter_x, int x_step_q4,
// const int16_t *filter_y, int y_step_q4,
// int w, int h);
FUN_CONV_1D(horiz, x_step_q4, filter_x, h, src, , sse2);
FUN_CONV_1D(vert, y_step_q4, filter_y, v, src - src_stride * 3, , sse2);
FUN_CONV_1D(horiz, x_step_q4, filter_x, h, src, , sse2)
FUN_CONV_1D(vert, y_step_q4, filter_y, v, src - src_stride * 3, , sse2)
#if CONFIG_AV1_HIGHBITDEPTH
highbd_filter8_1dfunction aom_highbd_filter_block1d16_v8_sse2;
@ -89,7 +89,7 @@ highbd_filter8_1dfunction aom_highbd_filter_block1d4_h2_sse2;
// const int16_t *filter_y,
// int y_step_q4,
// int w, int h, int bd);
HIGH_FUN_CONV_1D(horiz, x_step_q4, filter_x, h, src, , sse2);
HIGH_FUN_CONV_1D(vert, y_step_q4, filter_y, v, src - src_stride * 3, , sse2);
HIGH_FUN_CONV_1D(horiz, x_step_q4, filter_x, h, src, , sse2)
HIGH_FUN_CONV_1D(vert, y_step_q4, filter_y, v, src - src_stride * 3, , sse2)
#endif
#endif // HAVE_SSE2

View file

@ -0,0 +1,256 @@
/*
* Copyright (c) 2020, Alliance for Open Media. All Rights Reserved.
*
* Use of this source code is governed by a BSD-style license
* that can be found in the LICENSE file in the root of the source
* tree. An additional intellectual property rights grant can be found
* in the file PATENTS. All contributing project authors may
* be found in the AUTHORS file in the root of the source tree.
*/
#include <immintrin.h>
#include "config/aom_dsp_rtcd.h"
static INLINE void copy_128(const uint8_t *src, uint8_t *dst) {
__m256i s[4];
s[0] = _mm256_loadu_si256((__m256i *)(src + 0 * 32));
s[1] = _mm256_loadu_si256((__m256i *)(src + 1 * 32));
s[2] = _mm256_loadu_si256((__m256i *)(src + 2 * 32));
s[3] = _mm256_loadu_si256((__m256i *)(src + 3 * 32));
_mm256_storeu_si256((__m256i *)(dst + 0 * 32), s[0]);
_mm256_storeu_si256((__m256i *)(dst + 1 * 32), s[1]);
_mm256_storeu_si256((__m256i *)(dst + 2 * 32), s[2]);
_mm256_storeu_si256((__m256i *)(dst + 3 * 32), s[3]);
}
void aom_convolve_copy_avx2(const uint8_t *src, ptrdiff_t src_stride,
uint8_t *dst, ptrdiff_t dst_stride, int w, int h) {
if (w >= 16) {
assert(!((intptr_t)dst % 16));
assert(!(dst_stride % 16));
}
if (w == 2) {
do {
memmove(dst, src, 2 * sizeof(*src));
src += src_stride;
dst += dst_stride;
memmove(dst, src, 2 * sizeof(*src));
src += src_stride;
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 4) {
do {
memmove(dst, src, 4 * sizeof(*src));
src += src_stride;
dst += dst_stride;
memmove(dst, src, 4 * sizeof(*src));
src += src_stride;
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 8) {
do {
__m128i s[2];
s[0] = _mm_loadl_epi64((__m128i *)src);
src += src_stride;
s[1] = _mm_loadl_epi64((__m128i *)src);
src += src_stride;
_mm_storel_epi64((__m128i *)dst, s[0]);
dst += dst_stride;
_mm_storel_epi64((__m128i *)dst, s[1]);
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 16) {
do {
__m128i s[2];
s[0] = _mm_loadu_si128((__m128i *)src);
src += src_stride;
s[1] = _mm_loadu_si128((__m128i *)src);
src += src_stride;
_mm_store_si128((__m128i *)dst, s[0]);
dst += dst_stride;
_mm_store_si128((__m128i *)dst, s[1]);
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 32) {
do {
__m256i s[2];
s[0] = _mm256_loadu_si256((__m256i *)src);
src += src_stride;
s[1] = _mm256_loadu_si256((__m256i *)src);
src += src_stride;
_mm256_storeu_si256((__m256i *)dst, s[0]);
dst += dst_stride;
_mm256_storeu_si256((__m256i *)dst, s[1]);
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 64) {
do {
__m256i s[4];
s[0] = _mm256_loadu_si256((__m256i *)(src + 0 * 32));
s[1] = _mm256_loadu_si256((__m256i *)(src + 1 * 32));
src += src_stride;
s[2] = _mm256_loadu_si256((__m256i *)(src + 0 * 32));
s[3] = _mm256_loadu_si256((__m256i *)(src + 1 * 32));
src += src_stride;
_mm256_storeu_si256((__m256i *)(dst + 0 * 32), s[0]);
_mm256_storeu_si256((__m256i *)(dst + 1 * 32), s[1]);
dst += dst_stride;
_mm256_storeu_si256((__m256i *)(dst + 0 * 32), s[2]);
_mm256_storeu_si256((__m256i *)(dst + 1 * 32), s[3]);
dst += dst_stride;
h -= 2;
} while (h);
} else {
do {
copy_128(src, dst);
src += src_stride;
dst += dst_stride;
copy_128(src, dst);
src += src_stride;
dst += dst_stride;
h -= 2;
} while (h);
}
}
#if CONFIG_AV1_HIGHBITDEPTH
static INLINE void highbd_copy_64(const uint16_t *src, uint16_t *dst) {
__m256i s[4];
s[0] = _mm256_loadu_si256((__m256i *)(src + 0 * 16));
s[1] = _mm256_loadu_si256((__m256i *)(src + 1 * 16));
s[2] = _mm256_loadu_si256((__m256i *)(src + 2 * 16));
s[3] = _mm256_loadu_si256((__m256i *)(src + 3 * 16));
_mm256_storeu_si256((__m256i *)(dst + 0 * 16), s[0]);
_mm256_storeu_si256((__m256i *)(dst + 1 * 16), s[1]);
_mm256_storeu_si256((__m256i *)(dst + 2 * 16), s[2]);
_mm256_storeu_si256((__m256i *)(dst + 3 * 16), s[3]);
}
static INLINE void highbd_copy_128(const uint16_t *src, uint16_t *dst) {
__m256i s[8];
s[0] = _mm256_loadu_si256((__m256i *)(src + 0 * 16));
s[1] = _mm256_loadu_si256((__m256i *)(src + 1 * 16));
s[2] = _mm256_loadu_si256((__m256i *)(src + 2 * 16));
s[3] = _mm256_loadu_si256((__m256i *)(src + 3 * 16));
s[4] = _mm256_loadu_si256((__m256i *)(src + 4 * 16));
s[5] = _mm256_loadu_si256((__m256i *)(src + 5 * 16));
s[6] = _mm256_loadu_si256((__m256i *)(src + 6 * 16));
s[7] = _mm256_loadu_si256((__m256i *)(src + 7 * 16));
_mm256_storeu_si256((__m256i *)(dst + 0 * 16), s[0]);
_mm256_storeu_si256((__m256i *)(dst + 1 * 16), s[1]);
_mm256_storeu_si256((__m256i *)(dst + 2 * 16), s[2]);
_mm256_storeu_si256((__m256i *)(dst + 3 * 16), s[3]);
_mm256_storeu_si256((__m256i *)(dst + 4 * 16), s[4]);
_mm256_storeu_si256((__m256i *)(dst + 5 * 16), s[5]);
_mm256_storeu_si256((__m256i *)(dst + 6 * 16), s[6]);
_mm256_storeu_si256((__m256i *)(dst + 7 * 16), s[7]);
}
void aom_highbd_convolve_copy_avx2(const uint16_t *src, ptrdiff_t src_stride,
uint16_t *dst, ptrdiff_t dst_stride, int w,
int h) {
if (w >= 16) {
assert(!((intptr_t)dst % 16));
assert(!(dst_stride % 16));
}
if (w == 2) {
do {
memmove(dst, src, 2 * sizeof(*src));
src += src_stride;
dst += dst_stride;
memmove(dst, src, 2 * sizeof(*src));
src += src_stride;
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 4) {
do {
__m128i s[2];
s[0] = _mm_loadl_epi64((__m128i *)src);
src += src_stride;
s[1] = _mm_loadl_epi64((__m128i *)src);
src += src_stride;
_mm_storel_epi64((__m128i *)dst, s[0]);
dst += dst_stride;
_mm_storel_epi64((__m128i *)dst, s[1]);
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 8) {
do {
__m128i s[2];
s[0] = _mm_loadu_si128((__m128i *)src);
src += src_stride;
s[1] = _mm_loadu_si128((__m128i *)src);
src += src_stride;
_mm_store_si128((__m128i *)dst, s[0]);
dst += dst_stride;
_mm_store_si128((__m128i *)dst, s[1]);
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 16) {
do {
__m256i s[2];
s[0] = _mm256_loadu_si256((__m256i *)src);
src += src_stride;
s[1] = _mm256_loadu_si256((__m256i *)src);
src += src_stride;
_mm256_storeu_si256((__m256i *)dst, s[0]);
dst += dst_stride;
_mm256_storeu_si256((__m256i *)dst, s[1]);
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 32) {
do {
__m256i s[4];
s[0] = _mm256_loadu_si256((__m256i *)(src + 0 * 16));
s[1] = _mm256_loadu_si256((__m256i *)(src + 1 * 16));
src += src_stride;
s[2] = _mm256_loadu_si256((__m256i *)(src + 0 * 16));
s[3] = _mm256_loadu_si256((__m256i *)(src + 1 * 16));
src += src_stride;
_mm256_storeu_si256((__m256i *)(dst + 0 * 16), s[0]);
_mm256_storeu_si256((__m256i *)(dst + 1 * 16), s[1]);
dst += dst_stride;
_mm256_storeu_si256((__m256i *)(dst + 0 * 16), s[2]);
_mm256_storeu_si256((__m256i *)(dst + 1 * 16), s[3]);
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 64) {
do {
highbd_copy_64(src, dst);
src += src_stride;
dst += dst_stride;
highbd_copy_64(src, dst);
src += src_stride;
dst += dst_stride;
h -= 2;
} while (h);
} else {
assert(w == 128);
do {
highbd_copy_128(src, dst);
src += src_stride;
dst += dst_stride;
highbd_copy_128(src, dst);
src += src_stride;
dst += dst_stride;
h -= 2;
} while (h);
}
}
#endif // CONFIG_AV1_HIGHBITDEPTH

View file

@ -1,297 +0,0 @@
;
; Copyright (c) 2016, Alliance for Open Media. All rights reserved
;
; This source code is subject to the terms of the BSD 2 Clause License and
; the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
; was not distributed with this source code in the LICENSE file, you can
; obtain it at www.aomedia.org/license/software. If the Alliance for Open
; Media Patent License 1.0 was not distributed with this source code in the
; PATENTS file, you can obtain it at www.aomedia.org/license/patent.
;
;
%include "third_party/x86inc/x86inc.asm"
SECTION .text
%macro convolve_fn 1-2
%ifidn %1, avg
%define AUX_XMM_REGS 4
%else
%define AUX_XMM_REGS 0
%endif
%ifidn %2, highbd
%define pavg pavgw
cglobal %2_convolve_%1, 4, 7, 4+AUX_XMM_REGS, src, src_stride, \
dst, dst_stride, \
fx, fxs, fy, fys, w, h, bd
%else
%define pavg pavgb
cglobal convolve_%1, 4, 7, 4+AUX_XMM_REGS, src, src_stride, \
dst, dst_stride, \
fx, fxs, fy, fys, w, h
%endif
mov r4d, dword wm
%ifidn %2, highbd
shl r4d, 1
shl srcq, 1
shl src_strideq, 1
shl dstq, 1
shl dst_strideq, 1
%else
cmp r4d, 4
je .w4
%endif
cmp r4d, 8
je .w8
cmp r4d, 16
je .w16
cmp r4d, 32
je .w32
cmp r4d, 64
je .w64
%ifidn %2, highbd
cmp r4d, 128
je .w128
.w256:
mov r4d, dword hm
.loop256:
movu m0, [srcq]
movu m1, [srcq+16]
movu m2, [srcq+32]
movu m3, [srcq+48]
%ifidn %1, avg
pavg m0, [dstq]
pavg m1, [dstq+16]
pavg m2, [dstq+32]
pavg m3, [dstq+48]
%endif
mova [dstq ], m0
mova [dstq+16], m1
mova [dstq+32], m2
mova [dstq+48], m3
movu m0, [srcq+64]
movu m1, [srcq+80]
movu m2, [srcq+96]
movu m3, [srcq+112]
%ifidn %1, avg
pavg m0, [dstq+64]
pavg m1, [dstq+80]
pavg m2, [dstq+96]
pavg m3, [dstq+112]
%endif
mova [dstq+64], m0
mova [dstq+80], m1
mova [dstq+96], m2
mova [dstq+112], m3
movu m0, [srcq+128]
movu m1, [srcq+128+16]
movu m2, [srcq+128+32]
movu m3, [srcq+128+48]
%ifidn %1, avg
pavg m0, [dstq+128]
pavg m1, [dstq+128+16]
pavg m2, [dstq+128+32]
pavg m3, [dstq+128+48]
%endif
mova [dstq+128 ], m0
mova [dstq+128+16], m1
mova [dstq+128+32], m2
mova [dstq+128+48], m3
movu m0, [srcq+128+64]
movu m1, [srcq+128+80]
movu m2, [srcq+128+96]
movu m3, [srcq+128+112]
add srcq, src_strideq
%ifidn %1, avg
pavg m0, [dstq+128+64]
pavg m1, [dstq+128+80]
pavg m2, [dstq+128+96]
pavg m3, [dstq+128+112]
%endif
mova [dstq+128+64], m0
mova [dstq+128+80], m1
mova [dstq+128+96], m2
mova [dstq+128+112], m3
add dstq, dst_strideq
sub r4d, 1
jnz .loop256
RET
%endif
.w128:
mov r4d, dword hm
.loop128:
movu m0, [srcq]
movu m1, [srcq+16]
movu m2, [srcq+32]
movu m3, [srcq+48]
%ifidn %1, avg
pavg m0, [dstq]
pavg m1, [dstq+16]
pavg m2, [dstq+32]
pavg m3, [dstq+48]
%endif
mova [dstq ], m0
mova [dstq+16], m1
mova [dstq+32], m2
mova [dstq+48], m3
movu m0, [srcq+64]
movu m1, [srcq+80]
movu m2, [srcq+96]
movu m3, [srcq+112]
add srcq, src_strideq
%ifidn %1, avg
pavg m0, [dstq+64]
pavg m1, [dstq+80]
pavg m2, [dstq+96]
pavg m3, [dstq+112]
%endif
mova [dstq+64], m0
mova [dstq+80], m1
mova [dstq+96], m2
mova [dstq+112], m3
add dstq, dst_strideq
sub r4d, 1
jnz .loop128
RET
.w64:
mov r4d, dword hm
.loop64:
movu m0, [srcq]
movu m1, [srcq+16]
movu m2, [srcq+32]
movu m3, [srcq+48]
add srcq, src_strideq
%ifidn %1, avg
pavg m0, [dstq]
pavg m1, [dstq+16]
pavg m2, [dstq+32]
pavg m3, [dstq+48]
%endif
mova [dstq ], m0
mova [dstq+16], m1
mova [dstq+32], m2
mova [dstq+48], m3
add dstq, dst_strideq
sub r4d, 1
jnz .loop64
RET
.w32:
mov r4d, dword hm
.loop32:
movu m0, [srcq]
movu m1, [srcq+16]
movu m2, [srcq+src_strideq]
movu m3, [srcq+src_strideq+16]
lea srcq, [srcq+src_strideq*2]
%ifidn %1, avg
pavg m0, [dstq]
pavg m1, [dstq +16]
pavg m2, [dstq+dst_strideq]
pavg m3, [dstq+dst_strideq+16]
%endif
mova [dstq ], m0
mova [dstq +16], m1
mova [dstq+dst_strideq ], m2
mova [dstq+dst_strideq+16], m3
lea dstq, [dstq+dst_strideq*2]
sub r4d, 2
jnz .loop32
RET
.w16:
mov r4d, dword hm
lea r5q, [src_strideq*3]
lea r6q, [dst_strideq*3]
.loop16:
movu m0, [srcq]
movu m1, [srcq+src_strideq]
movu m2, [srcq+src_strideq*2]
movu m3, [srcq+r5q]
lea srcq, [srcq+src_strideq*4]
%ifidn %1, avg
pavg m0, [dstq]
pavg m1, [dstq+dst_strideq]
pavg m2, [dstq+dst_strideq*2]
pavg m3, [dstq+r6q]
%endif
mova [dstq ], m0
mova [dstq+dst_strideq ], m1
mova [dstq+dst_strideq*2], m2
mova [dstq+r6q ], m3
lea dstq, [dstq+dst_strideq*4]
sub r4d, 4
jnz .loop16
RET
.w8:
mov r4d, dword hm
lea r5q, [src_strideq*3]
lea r6q, [dst_strideq*3]
.loop8:
movh m0, [srcq]
movh m1, [srcq+src_strideq]
movh m2, [srcq+src_strideq*2]
movh m3, [srcq+r5q]
lea srcq, [srcq+src_strideq*4]
%ifidn %1, avg
movh m4, [dstq]
movh m5, [dstq+dst_strideq]
movh m6, [dstq+dst_strideq*2]
movh m7, [dstq+r6q]
pavg m0, m4
pavg m1, m5
pavg m2, m6
pavg m3, m7
%endif
movh [dstq ], m0
movh [dstq+dst_strideq ], m1
movh [dstq+dst_strideq*2], m2
movh [dstq+r6q ], m3
lea dstq, [dstq+dst_strideq*4]
sub r4d, 4
jnz .loop8
RET
%ifnidn %2, highbd
.w4:
mov r4d, dword hm
lea r5q, [src_strideq*3]
lea r6q, [dst_strideq*3]
.loop4:
movd m0, [srcq]
movd m1, [srcq+src_strideq]
movd m2, [srcq+src_strideq*2]
movd m3, [srcq+r5q]
lea srcq, [srcq+src_strideq*4]
%ifidn %1, avg
movd m4, [dstq]
movd m5, [dstq+dst_strideq]
movd m6, [dstq+dst_strideq*2]
movd m7, [dstq+r6q]
pavg m0, m4
pavg m1, m5
pavg m2, m6
pavg m3, m7
%endif
movd [dstq ], m0
movd [dstq+dst_strideq ], m1
movd [dstq+dst_strideq*2], m2
movd [dstq+r6q ], m3
lea dstq, [dstq+dst_strideq*4]
sub r4d, 4
jnz .loop4
RET
%endif
%endmacro
INIT_XMM sse2
convolve_fn copy
convolve_fn avg
convolve_fn copy, highbd

View file

@ -1,21 +1,146 @@
/*
* Copyright (c) 2018, Alliance for Open Media. All rights reserved
* Copyright (c) 2020, Alliance for Open Media. All Rights Reserved.
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
* Use of this source code is governed by a BSD-style license
* that can be found in the LICENSE file in the root of the source
* tree. An additional intellectual property rights grant can be found
* in the file PATENTS. All contributing project authors may
* be found in the AUTHORS file in the root of the source tree.
*/
#include <emmintrin.h>
#include <assert.h>
#include <immintrin.h>
#include "config/aom_dsp_rtcd.h"
#include "aom_dsp/aom_filter.h"
static INLINE void copy_128(const uint8_t *src, uint8_t *dst) {
__m128i s[8];
s[0] = _mm_loadu_si128((__m128i *)(src + 0 * 16));
s[1] = _mm_loadu_si128((__m128i *)(src + 1 * 16));
s[2] = _mm_loadu_si128((__m128i *)(src + 2 * 16));
s[3] = _mm_loadu_si128((__m128i *)(src + 3 * 16));
s[4] = _mm_loadu_si128((__m128i *)(src + 4 * 16));
s[5] = _mm_loadu_si128((__m128i *)(src + 5 * 16));
s[6] = _mm_loadu_si128((__m128i *)(src + 6 * 16));
s[7] = _mm_loadu_si128((__m128i *)(src + 7 * 16));
_mm_store_si128((__m128i *)(dst + 0 * 16), s[0]);
_mm_store_si128((__m128i *)(dst + 1 * 16), s[1]);
_mm_store_si128((__m128i *)(dst + 2 * 16), s[2]);
_mm_store_si128((__m128i *)(dst + 3 * 16), s[3]);
_mm_store_si128((__m128i *)(dst + 4 * 16), s[4]);
_mm_store_si128((__m128i *)(dst + 5 * 16), s[5]);
_mm_store_si128((__m128i *)(dst + 6 * 16), s[6]);
_mm_store_si128((__m128i *)(dst + 7 * 16), s[7]);
}
static INLINE void copy_64(const uint16_t *src, uint16_t *dst) {
void aom_convolve_copy_sse2(const uint8_t *src, ptrdiff_t src_stride,
uint8_t *dst, ptrdiff_t dst_stride, int w, int h) {
if (w >= 16) {
assert(!((intptr_t)dst % 16));
assert(!(dst_stride % 16));
}
if (w == 2) {
do {
memmove(dst, src, 2 * sizeof(*src));
src += src_stride;
dst += dst_stride;
memmove(dst, src, 2 * sizeof(*src));
src += src_stride;
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 4) {
do {
memmove(dst, src, 4 * sizeof(*src));
src += src_stride;
dst += dst_stride;
memmove(dst, src, 4 * sizeof(*src));
src += src_stride;
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 8) {
do {
__m128i s[2];
s[0] = _mm_loadl_epi64((__m128i *)src);
src += src_stride;
s[1] = _mm_loadl_epi64((__m128i *)src);
src += src_stride;
_mm_storel_epi64((__m128i *)dst, s[0]);
dst += dst_stride;
_mm_storel_epi64((__m128i *)dst, s[1]);
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 16) {
do {
__m128i s[2];
s[0] = _mm_loadu_si128((__m128i *)src);
src += src_stride;
s[1] = _mm_loadu_si128((__m128i *)src);
src += src_stride;
_mm_store_si128((__m128i *)dst, s[0]);
dst += dst_stride;
_mm_store_si128((__m128i *)dst, s[1]);
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 32) {
do {
__m128i s[4];
s[0] = _mm_loadu_si128((__m128i *)(src + 0 * 16));
s[1] = _mm_loadu_si128((__m128i *)(src + 1 * 16));
src += src_stride;
s[2] = _mm_loadu_si128((__m128i *)(src + 0 * 16));
s[3] = _mm_loadu_si128((__m128i *)(src + 1 * 16));
src += src_stride;
_mm_store_si128((__m128i *)(dst + 0 * 16), s[0]);
_mm_store_si128((__m128i *)(dst + 1 * 16), s[1]);
dst += dst_stride;
_mm_store_si128((__m128i *)(dst + 0 * 16), s[2]);
_mm_store_si128((__m128i *)(dst + 1 * 16), s[3]);
dst += dst_stride;
h -= 2;
} while (h);
} else if (w == 64) {
do {
__m128i s[8];
s[0] = _mm_loadu_si128((__m128i *)(src + 0 * 16));
s[1] = _mm_loadu_si128((__m128i *)(src + 1 * 16));
s[2] = _mm_loadu_si128((__m128i *)(src + 2 * 16));
s[3] = _mm_loadu_si128((__m128i *)(src + 3 * 16));
src += src_stride;
s[4] = _mm_loadu_si128((__m128i *)(src + 0 * 16));
s[5] = _mm_loadu_si128((__m128i *)(src + 1 * 16));
s[6] = _mm_loadu_si128((__m128i *)(src + 2 * 16));
s[7] = _mm_loadu_si128((__m128i *)(src + 3 * 16));
src += src_stride;
_mm_store_si128((__m128i *)(dst + 0 * 16), s[0]);
_mm_store_si128((__m128i *)(dst + 1 * 16), s[1]);
_mm_store_si128((__m128i *)(dst + 2 * 16), s[2]);
_mm_store_si128((__m128i *)(dst + 3 * 16), s[3]);
dst += dst_stride;
_mm_store_si128((__m128i *)(dst + 0 * 16), s[4]);
_mm_store_si128((__m128i *)(dst + 1 * 16), s[5]);
_mm_store_si128((__m128i *)(dst + 2 * 16), s[6]);
_mm_store_si128((__m128i *)(dst + 3 * 16), s[7]);
dst += dst_stride;
h -= 2;
} while (h);
} else {
do {
copy_128(src, dst);
src += src_stride;
dst += dst_stride;
copy_128(src, dst);
src += src_stride;
dst += dst_stride;
h -= 2;
} while (h);
}
}
static INLINE void highbd_copy_64(const uint16_t *src, uint16_t *dst) {
__m128i s[8];
s[0] = _mm_loadu_si128((__m128i *)(src + 0 * 8));
s[1] = _mm_loadu_si128((__m128i *)(src + 1 * 8));
@ -35,7 +160,7 @@ static INLINE void copy_64(const uint16_t *src, uint16_t *dst) {
_mm_store_si128((__m128i *)(dst + 7 * 8), s[7]);
}
static INLINE void copy_128(const uint16_t *src, uint16_t *dst) {
static INLINE void highbd_copy_128(const uint16_t *src, uint16_t *dst) {
__m128i s[16];
s[0] = _mm_loadu_si128((__m128i *)(src + 0 * 8));
s[1] = _mm_loadu_si128((__m128i *)(src + 1 * 8));
@ -71,17 +196,9 @@ static INLINE void copy_128(const uint16_t *src, uint16_t *dst) {
_mm_store_si128((__m128i *)(dst + 15 * 8), s[15]);
}
void av1_highbd_convolve_2d_copy_sr_sse2(
const uint16_t *src, int src_stride, uint16_t *dst, int dst_stride, int w,
int h, const InterpFilterParams *filter_params_x,
const InterpFilterParams *filter_params_y, const int subpel_x_qn,
const int subpel_y_qn, ConvolveParams *conv_params, int bd) {
(void)filter_params_x;
(void)filter_params_y;
(void)subpel_x_qn;
(void)subpel_y_qn;
(void)conv_params;
(void)bd;
void aom_highbd_convolve_copy_sse2(const uint16_t *src, ptrdiff_t src_stride,
uint16_t *dst, ptrdiff_t dst_stride, int w,
int h) {
if (w >= 16) {
assert(!((intptr_t)dst % 16));
assert(!(dst_stride % 16));
@ -169,20 +286,20 @@ void av1_highbd_convolve_2d_copy_sr_sse2(
} while (h);
} else if (w == 64) {
do {
copy_64(src, dst);
highbd_copy_64(src, dst);
src += src_stride;
dst += dst_stride;
copy_64(src, dst);
highbd_copy_64(src, dst);
src += src_stride;
dst += dst_stride;
h -= 2;
} while (h);
} else {
do {
copy_128(src, dst);
highbd_copy_128(src, dst);
src += src_stride;
dst += dst_stride;
copy_128(src, dst);
highbd_copy_128(src, dst);
src += src_stride;
dst += dst_stride;
h -= 2;

View file

@ -211,7 +211,7 @@ SECTION .text
; unsigned int output_height,
; short *filter
;)
global sym(aom_highbd_filter_block1d4_v8_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d4_v8_sse2)
sym(aom_highbd_filter_block1d4_v8_sse2):
push rbp
mov rbp, rsp
@ -281,7 +281,7 @@ sym(aom_highbd_filter_block1d4_v8_sse2):
; unsigned int output_height,
; short *filter
;)
global sym(aom_highbd_filter_block1d8_v8_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d8_v8_sse2)
sym(aom_highbd_filter_block1d8_v8_sse2):
push rbp
mov rbp, rsp
@ -340,7 +340,7 @@ sym(aom_highbd_filter_block1d8_v8_sse2):
; unsigned int output_height,
; short *filter
;)
global sym(aom_highbd_filter_block1d16_v8_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d16_v8_sse2)
sym(aom_highbd_filter_block1d16_v8_sse2):
push rbp
mov rbp, rsp
@ -403,7 +403,7 @@ sym(aom_highbd_filter_block1d16_v8_sse2):
; unsigned int output_height,
; short *filter
;)
global sym(aom_highbd_filter_block1d4_h8_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d4_h8_sse2)
sym(aom_highbd_filter_block1d4_h8_sse2):
push rbp
mov rbp, rsp
@ -478,7 +478,7 @@ sym(aom_highbd_filter_block1d4_h8_sse2):
; unsigned int output_height,
; short *filter
;)
global sym(aom_highbd_filter_block1d8_h8_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d8_h8_sse2)
sym(aom_highbd_filter_block1d8_h8_sse2):
push rbp
mov rbp, rsp
@ -544,7 +544,7 @@ sym(aom_highbd_filter_block1d8_h8_sse2):
; unsigned int output_height,
; short *filter
;)
global sym(aom_highbd_filter_block1d16_h8_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d16_h8_sse2)
sym(aom_highbd_filter_block1d16_h8_sse2):
push rbp
mov rbp, rsp

View file

@ -177,7 +177,7 @@
SECTION .text
global sym(aom_highbd_filter_block1d4_v2_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d4_v2_sse2)
sym(aom_highbd_filter_block1d4_v2_sse2):
push rbp
mov rbp, rsp
@ -201,7 +201,7 @@ sym(aom_highbd_filter_block1d4_v2_sse2):
pop rbp
ret
global sym(aom_highbd_filter_block1d8_v2_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d8_v2_sse2)
sym(aom_highbd_filter_block1d8_v2_sse2):
push rbp
mov rbp, rsp
@ -235,7 +235,7 @@ sym(aom_highbd_filter_block1d8_v2_sse2):
pop rbp
ret
global sym(aom_highbd_filter_block1d16_v2_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d16_v2_sse2)
sym(aom_highbd_filter_block1d16_v2_sse2):
push rbp
mov rbp, rsp
@ -271,7 +271,7 @@ sym(aom_highbd_filter_block1d16_v2_sse2):
pop rbp
ret
global sym(aom_highbd_filter_block1d4_h2_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d4_h2_sse2)
sym(aom_highbd_filter_block1d4_h2_sse2):
push rbp
mov rbp, rsp
@ -296,7 +296,7 @@ sym(aom_highbd_filter_block1d4_h2_sse2):
pop rbp
ret
global sym(aom_highbd_filter_block1d8_h2_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d8_h2_sse2)
sym(aom_highbd_filter_block1d8_h2_sse2):
push rbp
mov rbp, rsp
@ -330,7 +330,7 @@ sym(aom_highbd_filter_block1d8_h2_sse2):
pop rbp
ret
global sym(aom_highbd_filter_block1d16_h2_sse2) PRIVATE
globalsym(aom_highbd_filter_block1d16_h2_sse2)
sym(aom_highbd_filter_block1d16_h2_sse2):
push rbp
mov rbp, rsp

View file

@ -0,0 +1,282 @@
/*
* Copyright (c) 2020, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include <immintrin.h>
#include "config/aom_dsp_rtcd.h"
#include "aom/aom_integer.h"
#include "aom_dsp/x86/bitdepth_conversion_sse2.h"
#include "aom_dsp/x86/quantize_x86.h"
static INLINE void calculate_dqcoeff_and_store(__m128i qcoeff, __m128i dequant,
tran_low_t *dqcoeff) {
const __m128i low = _mm_mullo_epi16(qcoeff, dequant);
const __m128i high = _mm_mulhi_epi16(qcoeff, dequant);
const __m128i dqcoeff32_0 = _mm_unpacklo_epi16(low, high);
const __m128i dqcoeff32_1 = _mm_unpackhi_epi16(low, high);
_mm_store_si128((__m128i *)(dqcoeff), dqcoeff32_0);
_mm_store_si128((__m128i *)(dqcoeff + 4), dqcoeff32_1);
}
void aom_quantize_b_avx(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
const int16_t *zbin_ptr, const int16_t *round_ptr,
const int16_t *quant_ptr,
const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr,
uint16_t *eob_ptr, const int16_t *scan,
const int16_t *iscan) {
const __m128i zero = _mm_setzero_si128();
const __m256i big_zero = _mm256_setzero_si256();
int index;
__m128i zbin, round, quant, dequant, shift;
__m128i coeff0, coeff1;
__m128i qcoeff0, qcoeff1;
__m128i cmp_mask0, cmp_mask1;
__m128i all_zero;
__m128i eob = zero, eob0;
(void)scan;
*eob_ptr = 0;
load_b_values(zbin_ptr, &zbin, round_ptr, &round, quant_ptr, &quant,
dequant_ptr, &dequant, quant_shift_ptr, &shift);
// Do DC and first 15 AC.
coeff0 = load_tran_low(coeff_ptr);
coeff1 = load_tran_low(coeff_ptr + 8);
qcoeff0 = _mm_abs_epi16(coeff0);
qcoeff1 = _mm_abs_epi16(coeff1);
cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin);
zbin = _mm_unpackhi_epi64(zbin, zbin); // Switch DC to AC
cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin);
all_zero = _mm_or_si128(cmp_mask0, cmp_mask1);
if (_mm_test_all_zeros(all_zero, all_zero)) {
_mm256_store_si256((__m256i *)(qcoeff_ptr), big_zero);
_mm256_store_si256((__m256i *)(dqcoeff_ptr), big_zero);
_mm256_store_si256((__m256i *)(qcoeff_ptr + 8), big_zero);
_mm256_store_si256((__m256i *)(dqcoeff_ptr + 8), big_zero);
if (n_coeffs == 16) return;
round = _mm_unpackhi_epi64(round, round);
quant = _mm_unpackhi_epi64(quant, quant);
shift = _mm_unpackhi_epi64(shift, shift);
dequant = _mm_unpackhi_epi64(dequant, dequant);
} else {
calculate_qcoeff(&qcoeff0, round, quant, shift);
round = _mm_unpackhi_epi64(round, round);
quant = _mm_unpackhi_epi64(quant, quant);
shift = _mm_unpackhi_epi64(shift, shift);
calculate_qcoeff(&qcoeff1, round, quant, shift);
// Reinsert signs
qcoeff0 = _mm_sign_epi16(qcoeff0, coeff0);
qcoeff1 = _mm_sign_epi16(qcoeff1, coeff1);
// Mask out zbin threshold coeffs
qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0);
qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1);
store_tran_low(qcoeff0, qcoeff_ptr);
store_tran_low(qcoeff1, qcoeff_ptr + 8);
calculate_dqcoeff_and_store(qcoeff0, dequant, dqcoeff_ptr);
dequant = _mm_unpackhi_epi64(dequant, dequant);
calculate_dqcoeff_and_store(qcoeff1, dequant, dqcoeff_ptr + 8);
eob =
scan_for_eob(&qcoeff0, &qcoeff1, cmp_mask0, cmp_mask1, iscan, 0, zero);
}
// AC only loop.
for (index = 16; index < n_coeffs; index += 16) {
coeff0 = load_tran_low(coeff_ptr + index);
coeff1 = load_tran_low(coeff_ptr + index + 8);
qcoeff0 = _mm_abs_epi16(coeff0);
qcoeff1 = _mm_abs_epi16(coeff1);
cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin);
cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin);
all_zero = _mm_or_si128(cmp_mask0, cmp_mask1);
if (_mm_test_all_zeros(all_zero, all_zero)) {
_mm256_store_si256((__m256i *)(qcoeff_ptr + index), big_zero);
_mm256_store_si256((__m256i *)(dqcoeff_ptr + index), big_zero);
_mm256_store_si256((__m256i *)(qcoeff_ptr + index + 8), big_zero);
_mm256_store_si256((__m256i *)(dqcoeff_ptr + index + 8), big_zero);
continue;
}
calculate_qcoeff(&qcoeff0, round, quant, shift);
calculate_qcoeff(&qcoeff1, round, quant, shift);
qcoeff0 = _mm_sign_epi16(qcoeff0, coeff0);
qcoeff1 = _mm_sign_epi16(qcoeff1, coeff1);
qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0);
qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1);
store_tran_low(qcoeff0, qcoeff_ptr + index);
store_tran_low(qcoeff1, qcoeff_ptr + index + 8);
calculate_dqcoeff_and_store(qcoeff0, dequant, dqcoeff_ptr + index);
calculate_dqcoeff_and_store(qcoeff1, dequant, dqcoeff_ptr + index + 8);
eob0 = scan_for_eob(&qcoeff0, &qcoeff1, cmp_mask0, cmp_mask1, iscan, index,
zero);
eob = _mm_max_epi16(eob, eob0);
}
*eob_ptr = accumulate_eob(eob);
}
void aom_quantize_b_32x32_avx(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
const int16_t *zbin_ptr, const int16_t *round_ptr,
const int16_t *quant_ptr,
const int16_t *quant_shift_ptr,
tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr,
const int16_t *dequant_ptr, uint16_t *eob_ptr,
const int16_t *scan, const int16_t *iscan) {
const __m128i zero = _mm_setzero_si128();
const __m128i one = _mm_set1_epi16(1);
const __m256i big_zero = _mm256_setzero_si256();
int index;
const int log_scale = 1;
__m128i zbin, round, quant, dequant, shift;
__m128i coeff0, coeff1;
__m128i qcoeff0, qcoeff1;
__m128i cmp_mask0, cmp_mask1;
__m128i all_zero;
__m128i eob = zero, eob0;
(void)scan;
// Setup global values.
// The 32x32 halves zbin and round.
zbin = _mm_load_si128((const __m128i *)zbin_ptr);
// Shift with rounding.
zbin = _mm_add_epi16(zbin, one);
zbin = _mm_srli_epi16(zbin, 1);
// x86 has no "greater *or equal*" comparison. Subtract 1 from zbin so
// it is a strict "greater" comparison.
zbin = _mm_sub_epi16(zbin, one);
round = _mm_load_si128((const __m128i *)round_ptr);
round = _mm_add_epi16(round, one);
round = _mm_srli_epi16(round, 1);
quant = _mm_load_si128((const __m128i *)quant_ptr);
dequant = _mm_load_si128((const __m128i *)dequant_ptr);
shift = _mm_load_si128((const __m128i *)quant_shift_ptr);
// Do DC and first 15 AC.
coeff0 = load_tran_low(coeff_ptr);
coeff1 = load_tran_low(coeff_ptr + 8);
qcoeff0 = _mm_abs_epi16(coeff0);
qcoeff1 = _mm_abs_epi16(coeff1);
cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin);
zbin = _mm_unpackhi_epi64(zbin, zbin); // Switch DC to AC.
cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin);
all_zero = _mm_or_si128(cmp_mask0, cmp_mask1);
if (_mm_test_all_zeros(all_zero, all_zero)) {
_mm256_store_si256((__m256i *)(qcoeff_ptr), big_zero);
_mm256_store_si256((__m256i *)(dqcoeff_ptr), big_zero);
_mm256_store_si256((__m256i *)(qcoeff_ptr + 8), big_zero);
_mm256_store_si256((__m256i *)(dqcoeff_ptr + 8), big_zero);
round = _mm_unpackhi_epi64(round, round);
quant = _mm_unpackhi_epi64(quant, quant);
shift = _mm_unpackhi_epi64(shift, shift);
dequant = _mm_unpackhi_epi64(dequant, dequant);
} else {
calculate_qcoeff_log_scale(&qcoeff0, round, quant, &shift, &log_scale);
round = _mm_unpackhi_epi64(round, round);
quant = _mm_unpackhi_epi64(quant, quant);
shift = _mm_unpackhi_epi64(shift, shift);
calculate_qcoeff_log_scale(&qcoeff1, round, quant, &shift, &log_scale);
// Reinsert signs.
qcoeff0 = _mm_sign_epi16(qcoeff0, coeff0);
qcoeff1 = _mm_sign_epi16(qcoeff1, coeff1);
// Mask out zbin threshold coeffs.
qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0);
qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1);
store_tran_low(qcoeff0, qcoeff_ptr);
store_tran_low(qcoeff1, qcoeff_ptr + 8);
calculate_dqcoeff_and_store_log_scale(qcoeff0, dequant, zero, dqcoeff_ptr,
&log_scale);
dequant = _mm_unpackhi_epi64(dequant, dequant);
calculate_dqcoeff_and_store_log_scale(qcoeff1, dequant, zero,
dqcoeff_ptr + 8, &log_scale);
eob =
scan_for_eob(&qcoeff0, &qcoeff1, cmp_mask0, cmp_mask1, iscan, 0, zero);
}
// AC only loop.
for (index = 16; index < n_coeffs; index += 16) {
coeff0 = load_tran_low(coeff_ptr + index);
coeff1 = load_tran_low(coeff_ptr + index + 8);
qcoeff0 = _mm_abs_epi16(coeff0);
qcoeff1 = _mm_abs_epi16(coeff1);
cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin);
cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin);
all_zero = _mm_or_si128(cmp_mask0, cmp_mask1);
if (_mm_test_all_zeros(all_zero, all_zero)) {
_mm256_store_si256((__m256i *)(qcoeff_ptr + index), big_zero);
_mm256_store_si256((__m256i *)(dqcoeff_ptr + index), big_zero);
_mm256_store_si256((__m256i *)(qcoeff_ptr + index + 8), big_zero);
_mm256_store_si256((__m256i *)(dqcoeff_ptr + index + 8), big_zero);
continue;
}
calculate_qcoeff_log_scale(&qcoeff0, round, quant, &shift, &log_scale);
calculate_qcoeff_log_scale(&qcoeff1, round, quant, &shift, &log_scale);
qcoeff0 = _mm_sign_epi16(qcoeff0, coeff0);
qcoeff1 = _mm_sign_epi16(qcoeff1, coeff1);
qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0);
qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1);
store_tran_low(qcoeff0, qcoeff_ptr + index);
store_tran_low(qcoeff1, qcoeff_ptr + index + 8);
calculate_dqcoeff_and_store_log_scale(qcoeff0, dequant, zero,
dqcoeff_ptr + index, &log_scale);
calculate_dqcoeff_and_store_log_scale(qcoeff1, dequant, zero,
dqcoeff_ptr + index + 8, &log_scale);
eob0 = scan_for_eob(&qcoeff0, &qcoeff1, cmp_mask0, cmp_mask1, iscan, index,
zero);
eob = _mm_max_epi16(eob, eob0);
}
*eob_ptr = accumulate_eob(eob);
}

View file

@ -1435,7 +1435,7 @@ filter8_1dfunction aom_filter_block1d4_h2_ssse3;
// const int16_t *filter_x, int x_step_q4,
// const int16_t *filter_y, int y_step_q4,
// int w, int h);
FUN_CONV_1D(horiz, x_step_q4, filter_x, h, src, , avx2);
FUN_CONV_1D(vert, y_step_q4, filter_y, v, src - src_stride * 3, , avx2);
FUN_CONV_1D(horiz, x_step_q4, filter_x, h, src, , avx2)
FUN_CONV_1D(vert, y_step_q4, filter_y, v, src - src_stride * 3, , avx2)
#endif // HAVE_AX2 && HAVE_SSSE3

Some files were not shown because too many files have changed in this diff Show more