Compare commits
306 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c150ba6c9b | |||
| b62f72fc26 | |||
| 7de554e025 | |||
| 29c80ca095 | |||
| f009daed89 | |||
| 97c431bddb | |||
| 77ef66354d | |||
| 953c7139a5 | |||
| ee0255b3ce | |||
| a717c62dc3 | |||
| 14c73c7db3 | |||
| c98979434b | |||
| 6b30911e2d | |||
| bd95380139 | |||
| e9c5800ace | |||
| 6d681d5144 | |||
| 46e9017355 | |||
| 720adf9b5b | |||
| beda1b3cda | |||
| 62501cc7db | |||
| dbed372b70 | |||
| a38236491a | |||
| 1718ff9ade | |||
| c85856e360 | |||
| 1cfad6dbca | |||
| de223ad6ab | |||
| 51b67010e2 | |||
| 4db1a05aad | |||
| 037430193a | |||
| 7b9ab8373e | |||
| ac5dfb0a85 | |||
| 556c5202b4 | |||
| e1bd6f76c2 | |||
| 5cfc383268 | |||
| 727d0f984b | |||
| f995e1fbf9 | |||
| c0f2fe3135 | |||
| c5726277ff | |||
| 47e2607e6c | |||
| aa4504729c | |||
| d69235dd80 | |||
| bfbd7d4677 | |||
| afcdb781cb | |||
| 42ac98706a | |||
| 8feb8526bb | |||
| 594d1601ff | |||
| 6894b7f2d0 | |||
| 241a6b236a | |||
| 4fd22e97d8 | |||
| 1dcfc90757 | |||
| daef396277 | |||
| de4ce4f32d | |||
| 9c13b5fbd0 | |||
| 60507bffc0 | |||
| 4264096b72 | |||
| 2272a19ab0 | |||
| b29c5782e7 | |||
| 8674770e4b | |||
| f0b233fd09 | |||
| 50015c2ec9 | |||
| 2bb7e63266 | |||
| a44b589872 | |||
| 04b69f93e5 | |||
| afd13d8906 | |||
| 574e7f4727 | |||
| b2f9c10670 | |||
| 3a2a874994 | |||
| 3374404179 | |||
| b546257f77 | |||
| 844510cdb4 | |||
| 5e8c380e4b | |||
| d2fa9466be | |||
| 5f7f8ff5b6 | |||
| a31e4bd757 | |||
| bf82cfa74e | |||
| 549a6a0983 | |||
| 38dd16e108 | |||
| 43f3b8d33b | |||
| 165e9e251b | |||
| 84792e61c8 | |||
| e60603a9f2 | |||
| f3a1070f25 | |||
| 28b165940d | |||
| 04d588ee94 | |||
| e7c280e4cd | |||
| 2eac05d648 | |||
| 6deac59d1e | |||
| 7ba6452b09 | |||
| 7c04792480 | |||
| d26f298ce2 | |||
| f979959367 | |||
| 43a7ac13a5 | |||
| c720f4d355 | |||
| fcbc3d1b93 | |||
| f6965b7f12 | |||
| 0bc6bd9341 | |||
| af5cf2b1e7 | |||
| 0558c332ca | |||
| 04faac6900 | |||
| 716164239a | |||
| fa30043ba0 | |||
| c3f3a7e567 | |||
| 583e8e02eb | |||
| a86d561b79 | |||
| 9eea4fe842 | |||
| b3c5848f7f | |||
| 63bf075aad | |||
| fe0ab51460 | |||
| 29efbb9496 | |||
| 68dc20035b | |||
| 7889ac7603 | |||
| 8d95618093 | |||
| edeac873c4 | |||
| 1d0cda02a6 | |||
| caef968117 | |||
| 7d4b789f55 | |||
| 42b2b24fb8 | |||
| 40ff2a1251 | |||
| edb16889d1 | |||
| b129d9f2cb | |||
| cd5bfa124a | |||
| 5ea4939a1d | |||
| 2ba57aa535 | |||
| 8291a66e50 | |||
| a149f5c3c0 | |||
| 9da303e989 | |||
| d242c47b43 | |||
| 9a75cebc36 | |||
| 575af25859 | |||
| 767efeca06 | |||
| f8d2620d82 | |||
| f15666b703 | |||
| 30c3dd8edd | |||
| 1b7f126361 | |||
| c43debf1b1 | |||
| 1c7433a5eb | |||
| f32b314616 | |||
| 01d417c2fa | |||
| 847eece170 | |||
| 56a55933b3 | |||
| 9db59d8904 | |||
| c8fdaa8611 | |||
| f772f3e678 | |||
| 72b5380757 | |||
| bed3a34365 | |||
| 718b62c8cd | |||
| 1648c232ee | |||
| ce80e6daf6 | |||
| 8bd31a92a5 | |||
| 330e20672e | |||
| f966172feb | |||
| a403b575b1 | |||
| ac1fa6cbca | |||
| 4f088e42cb | |||
| 423cf6e2bf | |||
| 08681fdf13 | |||
| 93f12c117a | |||
| a17c862576 | |||
| 907dd87191 | |||
| 9710e7de9c | |||
| 28d1c21779 | |||
| aa2deb898e | |||
| c783088fe7 | |||
| 67c60d76e1 | |||
| 3437a26b3d | |||
| e542f661d0 | |||
| ca489d8aab | |||
| 22e9c0fee3 | |||
| ef4aff75b0 | |||
| 55fb9433b7 | |||
| 23f2769266 | |||
| b13d1bc2bb | |||
| 32cf02af50 | |||
| c3fa1db301 | |||
| 789a1f652b | |||
| 389450f61e | |||
| 79f7188c25 | |||
| 572c5a669d | |||
| 257b04f91c | |||
| b2e7f06c72 | |||
| 56f6d16602 | |||
| 3d12677c54 | |||
| 50ac82603a | |||
| cc7d8773ee | |||
| 2da8107ec1 | |||
| b374b24c0f | |||
| 0e3f70e898 | |||
| a5b9544866 | |||
| 83485c5092 | |||
| 7f2bb2fbc9 | |||
| 01da36ebdf | |||
| d3a94f1194 | |||
| f851fcd0b4 | |||
| 848c5a2dbb | |||
| a0a08d8543 | |||
| c8749f06e5 | |||
| 0cdf1b4be5 | |||
| b830ac82bb | |||
| 44541dfa6b | |||
| d711f974eb | |||
| 2f5bfc37b0 | |||
| f223436bb6 | |||
| 38f74bdc46 | |||
| 7072e79faa | |||
| 21d9f29d38 | |||
| ed004fe95d | |||
| 62a51df14e | |||
| 757f294a49 | |||
| 3d96175df2 | |||
| 70582027e7 | |||
| 96d6e472ad | |||
| b9e9a0ef79 | |||
| af11a10a4b | |||
| 90a9549b4e | |||
| 411fc219a7 | |||
| 7c63bb1b6e | |||
| e3101ddc8b | |||
| 7f891597bf | |||
| f398bf968c | |||
| 13a857d056 | |||
| 843f00e531 | |||
| 083cf424ff | |||
| 3f6c845d81 | |||
| b26f315d00 | |||
| ce45ebdef4 | |||
| 5319278dbe | |||
| 0b9c756f42 | |||
| 7463c2af64 | |||
| 3e9d80d831 | |||
| 2a9cbcc2f3 | |||
| 62c47f3558 | |||
| fa7b72d082 | |||
| 02309b9f60 | |||
| 2154425f70 | |||
| f6ffdc90b3 | |||
| 5de878a4e1 | |||
| 2fc656604b | |||
| 643ae52baa | |||
| d60d93a55c | |||
| 74e0eeb5ec | |||
| f2c3ccd6a6 | |||
| a7a40a3fde | |||
| 8e993f4d0b | |||
| 8d9b1e26b3 | |||
| 75d3ad14f2 | |||
| 0bf331a1bb | |||
| 19e122ee38 | |||
| b1d847beb5 | |||
| da51b12322 | |||
| 33b9d5141f | |||
| 212359662d | |||
| bd875480a9 | |||
| dd32cd5027 | |||
| 82e9155c75 | |||
| f4a0d7cb70 | |||
| 74ccc93687 | |||
| 4385e7e161 | |||
| 166e1df543 | |||
| 79db162487 | |||
| ec5c3052cf | |||
| a992a9bede | |||
| 2d808de191 | |||
| 93339ce857 | |||
| 109b24277b | |||
| d268788467 | |||
| 7629402bbd | |||
| 507b697ec0 | |||
| 312972d69b | |||
| b9cc27d5ff | |||
| 2f9fc727e1 | |||
| 4e1a8b4510 | |||
| cc6eb3d53d | |||
| bdef29970a | |||
| 6b3c489a2e | |||
| 7490d98654 | |||
| a796f66e0a | |||
| 4104018949 | |||
| 41511bf12e | |||
| 0d8abee540 | |||
| 0255c2b227 | |||
| 033a090923 | |||
| ccb02ddf8d | |||
| 27491dd953 | |||
| e560d2ba08 | |||
| 5a33c5c628 | |||
| 472b31f838 | |||
| 3329f8d139 | |||
| 01558f3f66 | |||
| 713c076d80 | |||
| 287e90a3a6 | |||
| 5ef6b241f0 | |||
| 2355eeb8f2 | |||
| 7fbcdc6d04 | |||
| 431f4fb242 | |||
| 32bf6cde06 | |||
| ca83ee6d9d | |||
| 01b94cc33b | |||
| 92f592ed10 | |||
| da2cc7817c | |||
| 26a2744eae | |||
| 54801d0734 | |||
| 89a200c82e | |||
| ca156d90b8 | |||
| 8afbd4f68a | |||
| 85c1639170 | |||
| d3997acbeb |
+41
-30
@@ -4,56 +4,56 @@ stages:
|
||||
- test
|
||||
|
||||
.debian-amd64-common:
|
||||
image: registry.videolan.org/dav1d-debian-unstable:20240406142551
|
||||
image: registry.videolan.org/dav1d-debian-unstable:20260622120900
|
||||
stage: build
|
||||
tags:
|
||||
- docker
|
||||
- amd64
|
||||
|
||||
.debian-amd64-minimum:
|
||||
image: registry.videolan.org/dav1d-debian-minimum:20240406142551
|
||||
image: registry.videolan.org/dav1d-debian-minimum:20260422185248
|
||||
stage: build
|
||||
tags:
|
||||
- docker
|
||||
- amd64
|
||||
|
||||
.debian-llvm-mingw-common:
|
||||
image: registry.videolan.org/vlc-debian-llvm-msvcrt:20240415145055
|
||||
image: registry.videolan.org/vlc-debian-llvm-ucrt:20260121170706
|
||||
stage: build
|
||||
tags:
|
||||
- docker
|
||||
- amd64
|
||||
|
||||
.debian-aarch64-common:
|
||||
image: registry.videolan.org/dav1d-debian-bookworm-aarch64:20240401050239
|
||||
image: registry.videolan.org/dav1d-debian-bookworm-aarch64:20260228141555
|
||||
stage: build
|
||||
tags:
|
||||
- docker
|
||||
- aarch64
|
||||
|
||||
.debian-armv7-common:
|
||||
image: registry.videolan.org/dav1d-debian-bookworm-armv7:20240401050040
|
||||
image: registry.videolan.org/dav1d-debian-bookworm-armv7:20260217211228
|
||||
stage: build
|
||||
tags:
|
||||
- docker
|
||||
- armv7
|
||||
|
||||
.debian-ppc64le-common:
|
||||
image: registry.videolan.org/dav1d-debian-unstable-ppc64le:20240401050321
|
||||
image: registry.videolan.org/dav1d-debian-unstable-ppc64le:20260210052452
|
||||
stage: build
|
||||
tags:
|
||||
- docker
|
||||
- ppc64le
|
||||
|
||||
.android-common:
|
||||
image: registry.videolan.org/vlc-debian-android:20240406142551
|
||||
image: registry.videolan.org/vlc-debian-android:20260120134731
|
||||
stage: build
|
||||
tags:
|
||||
- docker
|
||||
- amd64
|
||||
|
||||
.debian-wasm-emscripten-common:
|
||||
image: registry.videolan.org/vlc-debian-wasm-emscripten:20240313095757
|
||||
image: registry.videolan.org/vlc-debian-wasm-emscripten:20260120134731
|
||||
stage: build
|
||||
tags:
|
||||
- docker
|
||||
@@ -192,7 +192,7 @@ build-debian-avx:
|
||||
variables:
|
||||
CFLAGS: '-mavx'
|
||||
script:
|
||||
- meson setup build --buildtype debug
|
||||
- meson setup build --buildtype debugoptimized
|
||||
--werror
|
||||
- ninja -C build
|
||||
- cd build
|
||||
@@ -215,7 +215,7 @@ build-debian-avx512:
|
||||
variables:
|
||||
CFLAGS: '-mavx'
|
||||
script:
|
||||
- meson setup build --buildtype debug
|
||||
- meson setup build --buildtype debugoptimized
|
||||
--werror
|
||||
- ninja -C build
|
||||
- cd build
|
||||
@@ -342,6 +342,7 @@ build-debian-aarch64:
|
||||
extends: .debian-aarch64-common
|
||||
script:
|
||||
- meson setup build --buildtype debugoptimized
|
||||
-Dtrim_dsp=false
|
||||
--werror
|
||||
- ninja -C build
|
||||
- cd build && meson test -v
|
||||
@@ -353,13 +354,12 @@ build-debian-aarch64-clang-5:
|
||||
CFLAGS: '-integrated-as'
|
||||
script:
|
||||
- meson setup build --buildtype release
|
||||
-Dtrim_dsp=false
|
||||
- ninja -C build
|
||||
- cd build && meson test -v
|
||||
|
||||
build-debian-aarch64-clang-18:
|
||||
build-debian-aarch64-clang:
|
||||
extends: .debian-amd64-common
|
||||
variables:
|
||||
QEMU_LD_PREFIX: /usr/aarch64-linux-gnu/
|
||||
script:
|
||||
- meson setup build --buildtype release
|
||||
-Dtrim_dsp=false
|
||||
@@ -368,11 +368,8 @@ build-debian-aarch64-clang-18:
|
||||
- ninja -C build
|
||||
- cd build && meson test -v
|
||||
|
||||
build-macos:
|
||||
.build-macos-common:
|
||||
stage: build
|
||||
tags:
|
||||
- amd64
|
||||
- macos
|
||||
script:
|
||||
- meson setup build --buildtype release
|
||||
-Ddefault_library=both
|
||||
@@ -381,6 +378,17 @@ build-macos:
|
||||
- ninja -C build
|
||||
- cd build && meson test -v
|
||||
|
||||
build-macos-x86_64:
|
||||
extends: .build-macos-common
|
||||
tags:
|
||||
- amd64
|
||||
- macos
|
||||
|
||||
build-macos-arm64:
|
||||
extends: .build-macos-common
|
||||
tags:
|
||||
- macos-xcode26
|
||||
|
||||
build-debian-werror:
|
||||
extends: .debian-aarch64-common
|
||||
variables:
|
||||
@@ -394,6 +402,7 @@ build-debian-armv7:
|
||||
extends: .debian-armv7-common
|
||||
script:
|
||||
- linux32 meson setup build --buildtype debugoptimized
|
||||
-Dtrim_dsp=false
|
||||
--werror
|
||||
- ninja -C build
|
||||
- cd build && meson test -v
|
||||
@@ -405,11 +414,14 @@ build-debian-armv7-clang-5:
|
||||
CFLAGS: '-integrated-as'
|
||||
script:
|
||||
- linux32 meson setup build --buildtype release
|
||||
-Dtrim_dsp=false
|
||||
- ninja -C build
|
||||
- cd build && meson test -v
|
||||
|
||||
build-debian-ppc64le:
|
||||
extends: .debian-ppc64le-common
|
||||
variables:
|
||||
CC: gcc-13
|
||||
script:
|
||||
- meson setup build --buildtype release
|
||||
-Dtrim_dsp=false
|
||||
@@ -435,7 +447,6 @@ build-debian-riscv64:
|
||||
extends: .debian-amd64-common
|
||||
variables:
|
||||
QEMU_CPU: rv64,v=true,vext_spec=v1.0,vlen=256,elen=64
|
||||
QEMU_LD_PREFIX: /usr/riscv64-linux-gnu/
|
||||
script:
|
||||
- meson setup build --buildtype release
|
||||
-Dtrim_dsp=false
|
||||
@@ -450,8 +461,7 @@ build-debian-riscv64:
|
||||
build-debian-loongarch64:
|
||||
extends: .debian-amd64-common
|
||||
variables:
|
||||
QEMU_CPU: max-loongarch-cpu
|
||||
QEMU_LD_PREFIX: /opt/cross-tools/target/
|
||||
QEMU_CPU: max
|
||||
script:
|
||||
- meson setup build --buildtype release
|
||||
-Dtrim_dsp=false
|
||||
@@ -636,7 +646,11 @@ test-debian-msan:
|
||||
-Db_lundef=false
|
||||
-Denable_asm=false
|
||||
- ninja -C build
|
||||
- cd build && time meson test -v --setup=sanitizer
|
||||
- cd build
|
||||
- exit_code=0
|
||||
- time meson test -v --setup=sanitizer || exit_code=$((exit_code + $?))
|
||||
- time meson test -v --setup=sanitizer --suite testdata --test-args "--frametimes /dev/null" || exit_code=$((exit_code + $?))
|
||||
- if [ $exit_code -ne 0 ]; then exit $exit_code; fi
|
||||
|
||||
test-debian-ubsan:
|
||||
extends:
|
||||
@@ -719,6 +733,8 @@ test-debian-ppc64le:
|
||||
extends:
|
||||
- .debian-ppc64le-common
|
||||
- .test-common
|
||||
variables:
|
||||
CC: gcc-13
|
||||
needs: ["build-debian-ppc64le"]
|
||||
script:
|
||||
- meson setup build --buildtype release
|
||||
@@ -740,9 +756,7 @@ test-debian-riscv64:
|
||||
-Dtrim_dsp=false
|
||||
--cross-file package/crossfiles/riscv64-linux.meson
|
||||
- ninja -C build
|
||||
- cd build && time meson test -v --timeout-multiplier 4
|
||||
variables:
|
||||
QEMU_LD_PREFIX: /usr/riscv64-linux-gnu/
|
||||
- cd build && time meson test -v --timeout-multiplier 10
|
||||
parallel:
|
||||
matrix:
|
||||
- QEMU_CPU: [ "rv64,v=true,vext_spec=v1.0,vlen=128,elen=64",
|
||||
@@ -762,9 +776,7 @@ test-debian-aarch64-qemu:
|
||||
-Dtrim_dsp=false
|
||||
--cross-file package/crossfiles/aarch64-linux.meson
|
||||
- ninja -C build
|
||||
- cd build && time meson test -v --timeout-multiplier 4
|
||||
variables:
|
||||
QEMU_LD_PREFIX: /usr/aarch64-linux-gnu/
|
||||
- cd build && time meson test -v --timeout-multiplier 10
|
||||
parallel:
|
||||
matrix:
|
||||
# sve-default-vector-length sets the max vector length in bytes;
|
||||
@@ -799,8 +811,7 @@ test-debian-loongarch64:
|
||||
- .test-common
|
||||
needs: ["build-debian-loongarch64"]
|
||||
variables:
|
||||
QEMU_CPU: max-loongarch-cpu
|
||||
QEMU_LD_PREFIX: /opt/cross-tools/target/
|
||||
QEMU_CPU: max
|
||||
script:
|
||||
- meson setup build --buildtype release
|
||||
-Dtestdata_tests=true
|
||||
@@ -808,7 +819,7 @@ test-debian-loongarch64:
|
||||
-Dtrim_dsp=false
|
||||
--cross-file package/crossfiles/loongarch64-linux.meson
|
||||
- ninja -C build
|
||||
- cd build && time meson test -v --timeout-multiplier 4
|
||||
- cd build && time meson test -v --timeout-multiplier 10
|
||||
|
||||
.test-argon-script: &test-argon-script
|
||||
- meson setup build --buildtype release
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
Copyright © 2018-2019, VideoLAN and dav1d authors
|
||||
Copyright © 2018-2025, VideoLAN and dav1d authors
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
|
||||
@@ -1,3 +1,84 @@
|
||||
Changes for 1.5.4 'Sonic':
|
||||
--------------------------
|
||||
|
||||
1.5.4 is a minor release of dav1d, focused maintenance:
|
||||
- Support for OS/2
|
||||
- Switch to external checkasm
|
||||
- Add Armv9.3-A GCS support
|
||||
- AArch64: optimize ipred_*_8bpc & ipred_smooth_*_8bpc functions
|
||||
- ARM32: optimize prep_neon
|
||||
- RISC-V: add ipred_v,_h,_pal,_dc optimizations
|
||||
|
||||
|
||||
Changes for 1.5.3 'Sonic':
|
||||
--------------------------
|
||||
|
||||
1.5.3 is a minor release of dav1d, focused on RISC-V and maintenance:
|
||||
- Misc small optimizations
|
||||
- RISC-V assembly optimizations for ipred, emu_edge and w_mask,
|
||||
and VLEN 512 for blend functions
|
||||
- Fix issue with ivf files with 0 frames in tools
|
||||
|
||||
|
||||
Changes for 1.5.2 'Sonic':
|
||||
--------------------------
|
||||
|
||||
1.5.2 is a minor release of dav1d, focused on maintenance:
|
||||
- minor speed improvement in recon
|
||||
- improvements on loongarch symboles visibility and asm
|
||||
- mark C globals with small code model
|
||||
- reduce the code size of the frame header parsing (OBU)
|
||||
- minor fixes on tools and CI
|
||||
- fix compilation with nasm 3.00
|
||||
|
||||
|
||||
Changes for 1.5.1 'Sonic':
|
||||
--------------------------
|
||||
|
||||
1.5.1 is a minor release of dav1d, focusing on optimizations and stack reduction:
|
||||
|
||||
- Rewrite of the looprestoration (SGR, wiener) to reduce stack usage
|
||||
- Rewrite of {put,prep}_scaled functions
|
||||
|
||||
Now, the required stack space for dav1d should be: 62 KB on x86_64 and
|
||||
58KB on arm and aarch64.
|
||||
|
||||
- Improvements on the SSSE3 SGR
|
||||
- Improvements on ARM32/ARM64 looprestoration optimizations
|
||||
- RISC-V: blend optimizations for high bitdepth
|
||||
- Power9: blend optimizations for 8bpc
|
||||
- Port RISC-V to POSIX/non-Linux OS
|
||||
- AArch64: Add Neon implementation of load_tmvs
|
||||
- Fix a rare, but possible deadlock, in flush()
|
||||
|
||||
|
||||
Changes for 1.5.0 'Sonic':
|
||||
--------------------------
|
||||
|
||||
1.5.0 is a major release of dav1d, that:
|
||||
- WARNING: we removed some of the SSE2 optimizations, so if you care about
|
||||
systems without SSSE3, you should be careful when updating!
|
||||
- Add Arm OpenBSD run-time CPU feature
|
||||
- Optimize index offset calculations for decode_coefs
|
||||
- picture: copy HDR10+ and T35 metadata only to visible frames
|
||||
- SSSE3 new optimizations for 6-tap (8bit and hbd)
|
||||
- AArch64/SVE: Add HBD subpel filters using 128-bit SVE2
|
||||
- AArch64: Add USMMLA Implementation for 6-tap H/HV
|
||||
- AArch64: Optimize Armv8.0 NEON for HBD horizontal filters and 6-tap filters
|
||||
- Power9: Optimized ITX till 16x4.
|
||||
- Loongarch: numerous optimizations
|
||||
- RISC-V optimizations for pal, cdef_filter, ipred, mc_blend, mc_bdir, itx
|
||||
- Allow playing videos in full-screen mode in dav1dplay
|
||||
|
||||
|
||||
Changes for 1.4.3 'Road Runner':
|
||||
--------------------------------
|
||||
|
||||
1.4.3 is a small release focused on security issues
|
||||
- AArch64: Fix potential out of bounds access in DotProd H/HV filters
|
||||
- cli: Prevent buffer over-read
|
||||
|
||||
|
||||
Changes for 1.4.2 'Road Runner':
|
||||
--------------------------------
|
||||
|
||||
|
||||
@@ -79,24 +79,27 @@ VideoLAN will only have the collective work rights.
|
||||
The [VideoLAN Code of Conduct](https://wiki.videolan.org/CoC) applies to this project.
|
||||
|
||||
# Compile
|
||||
## General compilation steps
|
||||
|
||||
1. Install [Meson](https://mesonbuild.com/) (0.49 or higher), [Ninja](https://ninja-build.org/), and, for x86\* targets, [nasm](https://nasm.us/) (2.14 or higher)
|
||||
1. Install [Meson](https://mesonbuild.com/) (0.54 or higher), [Ninja](https://ninja-build.org/), and, for x86\* targets, [nasm](https://nasm.us/) (2.14 or higher)
|
||||
2. Run `mkdir build && cd build` to create a build directory and enter it
|
||||
3. Run `meson setup ..` to configure meson, add `--default-library=static` if static linking is desired
|
||||
4. Run `ninja` to compile
|
||||
|
||||
Following are modification of step 3 and 4, for specific purpose.
|
||||
|
||||
## Cross-Compilation for 32- or 64-bit Windows, 32-bit Linux
|
||||
|
||||
If you're on a linux build machine trying to compile .exe for a Windows target/host machine, run
|
||||
If you're on a linux build machine trying to compile .exe for a Windows target/host machine, configure meson like this
|
||||
|
||||
```
|
||||
meson setup build --cross-file=package/crossfiles/x86_64-w64-mingw32.meson
|
||||
meson setup .. --cross-file=../package/crossfiles/x86_64-w64-mingw32.meson
|
||||
```
|
||||
|
||||
or, for 32-bit:
|
||||
|
||||
```
|
||||
meson setup build --cross-file=package/crossfiles/i686-w64-mingw32.meson
|
||||
meson setup .. --cross-file=../package/crossfiles/i686-w64-mingw32.meson
|
||||
```
|
||||
|
||||
`mingw-w64` is a pre-requisite and should be installed on your linux machine via your preferred method or package manager. Note the binary name formats may differ between distributions. Verify the names, and use `alias` if certain binaries cannot be found.
|
||||
@@ -104,14 +107,14 @@ meson setup build --cross-file=package/crossfiles/i686-w64-mingw32.meson
|
||||
For 32-bit linux, run
|
||||
|
||||
```
|
||||
meson setup build --cross-file=package/crossfiles/i686-linux32.meson
|
||||
meson setup .. --cross-file=../package/crossfiles/i686-linux32.meson
|
||||
```
|
||||
|
||||
## Build documentation
|
||||
|
||||
1. Install [doxygen](https://www.doxygen.nl/) and [graphviz](https://www.graphviz.org/)
|
||||
2. Run `meson setup build -Denable_docs=true` to create the build directory
|
||||
3. Run `ninja -C build doc/html` to build the docs
|
||||
1. Make sure [doxygen](https://www.doxygen.nl/) and [graphviz](https://www.graphviz.org/) are installed.
|
||||
2. Run `meson setup .. -Denable_docs=true` to configure meson to generate docs from the build directory.
|
||||
3. Run `ninja doc/html` to build the docs
|
||||
|
||||
The result can be found in `build/doc/html/`. An online version built from master can be found [here](https://videolan.videolan.me/dav1d/).
|
||||
|
||||
@@ -121,6 +124,13 @@ The result can be found in `build/doc/html/`. An online version built from maste
|
||||
2. During meson configuration, specify `-Dtestdata_tests=true`
|
||||
3. Run `meson test -v` after compiling
|
||||
|
||||
## Decoder conformance tests (optional but encouraged)
|
||||
|
||||
1. Download the argon conformance bitstreams from https://streams.videolan.org/argon/
|
||||
2. Extract into dav1d directory by running `tar -xvf argon.tar.zst`
|
||||
3. Execute tests with `tests/dav1d_argon.bash -d build/tools/dav1d -a argon`
|
||||
4. Expected outcome is `2763 files successfully verified in XXmYYs (dav1d 1.x.y-zz-gHHHHHHH filmgrain=1 cpumask=-1)`
|
||||
|
||||
# Support
|
||||
|
||||
This project is partially funded by the *Alliance for Open Media*/**AOM** and is supported by TwoOrioles and VideoLabs.
|
||||
|
||||
+37
-25
@@ -120,6 +120,7 @@ static void dp_settings_print_usage(const char *const app,
|
||||
" --highquality: enable high quality rendering\n"
|
||||
" --zerocopy/-z: enable zero copy upload path\n"
|
||||
" --gpugrain/-g: enable GPU grain synthesis\n"
|
||||
" --fullscreen/-f: enable full screen mode\n"
|
||||
" --version/-v: print version and exit\n"
|
||||
" --renderer/-r: select renderer backend (default: auto)\n");
|
||||
exit(1);
|
||||
@@ -144,7 +145,7 @@ static void dp_rd_ctx_parse_args(Dav1dPlayRenderContext *rd_ctx,
|
||||
Dav1dSettings *lib_settings = &rd_ctx->lib_settings;
|
||||
|
||||
// Short options
|
||||
static const char short_opts[] = "i:vuzgr:";
|
||||
static const char short_opts[] = "i:vuzgfr:";
|
||||
|
||||
enum {
|
||||
ARG_THREADS = 256,
|
||||
@@ -162,6 +163,7 @@ static void dp_rd_ctx_parse_args(Dav1dPlayRenderContext *rd_ctx,
|
||||
{ "highquality", 0, NULL, ARG_HIGH_QUALITY },
|
||||
{ "zerocopy", 0, NULL, 'z' },
|
||||
{ "gpugrain", 0, NULL, 'g' },
|
||||
{ "fullscreen", 0, NULL, 'f'},
|
||||
{ "renderer", 0, NULL, 'r'},
|
||||
{ NULL, 0, NULL, 0 },
|
||||
};
|
||||
@@ -186,6 +188,9 @@ static void dp_rd_ctx_parse_args(Dav1dPlayRenderContext *rd_ctx,
|
||||
case 'g':
|
||||
settings->gpugrain = true;
|
||||
break;
|
||||
case 'f':
|
||||
settings->fullscreen = true;
|
||||
break;
|
||||
case 'r':
|
||||
settings->renderer_name = optarg;
|
||||
break;
|
||||
@@ -240,35 +245,37 @@ static Dav1dPlayRenderContext *dp_rd_ctx_create(int argc, char **argv)
|
||||
return NULL;
|
||||
}
|
||||
|
||||
// Parse and validate arguments
|
||||
dav1d_default_settings(&rd_ctx->lib_settings);
|
||||
memset(&rd_ctx->settings, 0, sizeof(rd_ctx->settings));
|
||||
dp_rd_ctx_parse_args(rd_ctx, argc, argv);
|
||||
|
||||
// Init SDL2 library
|
||||
if (SDL_Init(SDL_INIT_VIDEO | SDL_INIT_TIMER) < 0) {
|
||||
fprintf(stderr, "SDL_Init failed: %s\n", SDL_GetError());
|
||||
goto fail;
|
||||
}
|
||||
|
||||
// Register a custom event to notify our SDL main thread
|
||||
// about new frames
|
||||
rd_ctx->event_types = SDL_RegisterEvents(3);
|
||||
if (rd_ctx->event_types == UINT32_MAX) {
|
||||
fprintf(stderr, "Failure to create custom SDL event types!\n");
|
||||
free(rd_ctx);
|
||||
return NULL;
|
||||
goto fail;
|
||||
}
|
||||
|
||||
rd_ctx->fifo = dp_fifo_create(5);
|
||||
if (rd_ctx->fifo == NULL) {
|
||||
fprintf(stderr, "Failed to create FIFO for output pictures!\n");
|
||||
free(rd_ctx);
|
||||
return NULL;
|
||||
goto fail;
|
||||
}
|
||||
|
||||
rd_ctx->lock = SDL_CreateMutex();
|
||||
if (rd_ctx->lock == NULL) {
|
||||
fprintf(stderr, "SDL_CreateMutex failed: %s\n", SDL_GetError());
|
||||
dp_fifo_destroy(rd_ctx->fifo);
|
||||
free(rd_ctx);
|
||||
return NULL;
|
||||
goto fail;
|
||||
}
|
||||
|
||||
// Parse and validate arguments
|
||||
dav1d_default_settings(&rd_ctx->lib_settings);
|
||||
memset(&rd_ctx->settings, 0, sizeof(rd_ctx->settings));
|
||||
dp_rd_ctx_parse_args(rd_ctx, argc, argv);
|
||||
|
||||
// Select renderer
|
||||
renderer_info = dp_get_renderer(rd_ctx->settings.renderer_name);
|
||||
|
||||
@@ -279,15 +286,21 @@ static Dav1dPlayRenderContext *dp_rd_ctx_create(int argc, char **argv)
|
||||
printf("Using %s renderer\n", renderer_info->name);
|
||||
}
|
||||
|
||||
rd_ctx->rd_priv = (renderer_info) ? renderer_info->create_renderer() : NULL;
|
||||
rd_ctx->rd_priv = (renderer_info) ? renderer_info->create_renderer(&rd_ctx->settings) : NULL;
|
||||
if (rd_ctx->rd_priv == NULL) {
|
||||
SDL_DestroyMutex(rd_ctx->lock);
|
||||
dp_fifo_destroy(rd_ctx->fifo);
|
||||
free(rd_ctx);
|
||||
return NULL;
|
||||
goto fail;
|
||||
}
|
||||
|
||||
return rd_ctx;
|
||||
|
||||
fail:
|
||||
if (rd_ctx->lock)
|
||||
SDL_DestroyMutex(rd_ctx->lock);
|
||||
if (rd_ctx->fifo)
|
||||
dp_fifo_destroy(rd_ctx->fifo);
|
||||
free(rd_ctx);
|
||||
SDL_Quit();
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -662,10 +675,6 @@ int main(int argc, char **argv)
|
||||
return 1;
|
||||
}
|
||||
|
||||
// Init SDL2 library
|
||||
if (SDL_Init(SDL_INIT_VIDEO | SDL_INIT_TIMER) < 0)
|
||||
return 10;
|
||||
|
||||
// Create render context
|
||||
Dav1dPlayRenderContext *rd_ctx = dp_rd_ctx_create(argc, argv);
|
||||
if (rd_ctx == NULL) {
|
||||
@@ -711,9 +720,7 @@ int main(int argc, char **argv)
|
||||
if (e->type == SDL_QUIT) {
|
||||
dp_rd_ctx_request_shutdown(rd_ctx);
|
||||
dp_fifo_flush(rd_ctx->fifo, destroy_pic);
|
||||
SDL_FlushEvent(rd_ctx->event_types + DAV1D_EVENT_NEW_FRAME);
|
||||
SDL_FlushEvent(rd_ctx->event_types + DAV1D_EVENT_SEEK_FRAME);
|
||||
num_frame_events = 0;
|
||||
goto out;
|
||||
} else if (e->type == SDL_WINDOWEVENT) {
|
||||
if (e->window.event == SDL_WINDOWEVENT_SIZE_CHANGED) {
|
||||
// TODO: Handle window resizes
|
||||
@@ -724,6 +731,10 @@ int main(int argc, char **argv)
|
||||
SDL_KeyboardEvent *kbde = (SDL_KeyboardEvent *)e;
|
||||
if (kbde->keysym.sym == SDLK_SPACE) {
|
||||
dp_rd_ctx_toggle_pause(rd_ctx);
|
||||
} else if (kbde->keysym.sym == SDLK_ESCAPE) {
|
||||
dp_rd_ctx_request_shutdown(rd_ctx);
|
||||
dp_fifo_flush(rd_ctx->fifo, destroy_pic);
|
||||
goto out;
|
||||
} else if (kbde->keysym.sym == SDLK_LEFT ||
|
||||
kbde->keysym.sym == SDLK_RIGHT)
|
||||
{
|
||||
@@ -776,5 +787,6 @@ out:;
|
||||
int decoder_ret = 0;
|
||||
SDL_WaitThread(decoder_thread, &decoder_ret);
|
||||
dp_rd_ctx_destroy(rd_ctx);
|
||||
SDL_Quit();
|
||||
return decoder_ret;
|
||||
}
|
||||
|
||||
+21
-7
@@ -30,22 +30,32 @@
|
||||
#include "dav1d/dav1d.h"
|
||||
|
||||
#include <SDL.h>
|
||||
#ifdef HAVE_PLACEBO
|
||||
#if HAVE_PLACEBO
|
||||
# include <libplacebo/config.h>
|
||||
#endif
|
||||
|
||||
// Check libplacebo Vulkan rendering
|
||||
#if defined(HAVE_VULKAN) && defined(SDL_VIDEO_VULKAN)
|
||||
#if HAVE_VULKAN && defined(SDL_VIDEO_VULKAN)
|
||||
# if defined(PL_HAVE_VULKAN) && PL_HAVE_VULKAN
|
||||
# define HAVE_RENDERER_PLACEBO
|
||||
# define HAVE_PLACEBO_VULKAN
|
||||
# define HAVE_RENDERER_PLACEBO 1
|
||||
# define HAVE_PLACEBO_VULKAN 1
|
||||
# endif
|
||||
#endif
|
||||
|
||||
// Check libplacebo OpenGL rendering
|
||||
#if defined(PL_HAVE_OPENGL) && PL_HAVE_OPENGL
|
||||
# define HAVE_RENDERER_PLACEBO
|
||||
# define HAVE_PLACEBO_OPENGL
|
||||
# define HAVE_RENDERER_PLACEBO 1
|
||||
# define HAVE_PLACEBO_OPENGL 1
|
||||
#endif
|
||||
|
||||
#ifndef HAVE_RENDERER_PLACEBO
|
||||
#define HAVE_RENDERER_PLACEBO 0
|
||||
#endif
|
||||
#ifndef HAVE_PLACEBO_VULKAN
|
||||
#define HAVE_PLACEBO_VULKAN 0
|
||||
#endif
|
||||
#ifndef HAVE_PLACEBO_OPENGL
|
||||
#define HAVE_PLACEBO_OPENGL 0
|
||||
#endif
|
||||
|
||||
/**
|
||||
@@ -61,6 +71,7 @@ typedef struct {
|
||||
int untimed;
|
||||
int zerocopy;
|
||||
int gpugrain;
|
||||
int fullscreen;
|
||||
} Dav1dPlaySettings;
|
||||
|
||||
#define WINDOW_WIDTH 910
|
||||
@@ -82,7 +93,7 @@ typedef struct rdr_info
|
||||
// Cookie passed to the renderer implementation callbacks
|
||||
void *cookie;
|
||||
// Callback to create the renderer
|
||||
void* (*create_renderer)(void);
|
||||
void* (*create_renderer)(const Dav1dPlaySettings *settings);
|
||||
// Callback to destroy the renderer
|
||||
void (*destroy_renderer)(void *cookie);
|
||||
// Callback to the render function that renders a prevously sent frame
|
||||
@@ -129,6 +140,9 @@ static inline SDL_Window *dp_create_sdl_window(int window_flags)
|
||||
|
||||
win = SDL_CreateWindow("Dav1dPlay", SDL_WINDOWPOS_CENTERED, SDL_WINDOWPOS_CENTERED,
|
||||
WINDOW_WIDTH, WINDOW_HEIGHT, window_flags);
|
||||
if (!win)
|
||||
return NULL;
|
||||
|
||||
SDL_SetWindowResizable(win, SDL_TRUE);
|
||||
|
||||
return win;
|
||||
|
||||
@@ -26,17 +26,17 @@
|
||||
|
||||
#include "dp_renderer.h"
|
||||
|
||||
#ifdef HAVE_RENDERER_PLACEBO
|
||||
#if HAVE_RENDERER_PLACEBO
|
||||
#include <assert.h>
|
||||
|
||||
#include <libplacebo/renderer.h>
|
||||
#include <libplacebo/utils/dav1d.h>
|
||||
|
||||
#ifdef HAVE_PLACEBO_VULKAN
|
||||
#if HAVE_PLACEBO_VULKAN
|
||||
# include <libplacebo/vulkan.h>
|
||||
# include <SDL_vulkan.h>
|
||||
#endif
|
||||
#ifdef HAVE_PLACEBO_OPENGL
|
||||
#if HAVE_PLACEBO_OPENGL
|
||||
# include <libplacebo/opengl.h>
|
||||
# include <SDL_opengl.h>
|
||||
#endif
|
||||
@@ -53,7 +53,7 @@ typedef struct renderer_priv_ctx
|
||||
pl_log log;
|
||||
// Placebo renderer
|
||||
pl_renderer renderer;
|
||||
#ifdef HAVE_PLACEBO_VULKAN
|
||||
#if HAVE_PLACEBO_VULKAN
|
||||
// Placebo Vulkan handle
|
||||
pl_vulkan vk;
|
||||
// Placebo Vulkan instance
|
||||
@@ -61,9 +61,11 @@ typedef struct renderer_priv_ctx
|
||||
// Vulkan surface
|
||||
VkSurfaceKHR surf;
|
||||
#endif
|
||||
#ifdef HAVE_PLACEBO_OPENGL
|
||||
#if HAVE_PLACEBO_OPENGL
|
||||
// Placebo OpenGL handle
|
||||
pl_opengl gl;
|
||||
// SDL OpenGL context
|
||||
SDL_GLContext gl_context;
|
||||
#endif
|
||||
// Placebo GPU
|
||||
pl_gpu gpu;
|
||||
@@ -77,19 +79,27 @@ typedef struct renderer_priv_ctx
|
||||
} Dav1dPlayRendererPrivateContext;
|
||||
|
||||
static Dav1dPlayRendererPrivateContext*
|
||||
placebo_renderer_create_common(int window_flags)
|
||||
placebo_renderer_create_common(const Dav1dPlaySettings *settings, int window_flags)
|
||||
{
|
||||
if (settings->fullscreen)
|
||||
window_flags |= SDL_WINDOW_FULLSCREEN_DESKTOP;
|
||||
|
||||
// Create Window
|
||||
SDL_Window *sdlwin = dp_create_sdl_window(window_flags | SDL_WINDOW_RESIZABLE);
|
||||
if (sdlwin == NULL)
|
||||
if (sdlwin == NULL) {
|
||||
fprintf(stderr, "Creating SDL window failed: %s\n", SDL_GetError());
|
||||
return NULL;
|
||||
}
|
||||
|
||||
SDL_ShowCursor(0);
|
||||
|
||||
// Alloc
|
||||
Dav1dPlayRendererPrivateContext *const rd_priv_ctx =
|
||||
calloc(1, sizeof(Dav1dPlayRendererPrivateContext));
|
||||
if (rd_priv_ctx == NULL)
|
||||
if (rd_priv_ctx == NULL) {
|
||||
fprintf(stderr, "Out of memory!\n");
|
||||
return NULL;
|
||||
|
||||
}
|
||||
rd_priv_ctx->win = sdlwin;
|
||||
|
||||
// Init libplacebo
|
||||
@@ -102,6 +112,7 @@ static Dav1dPlayRendererPrivateContext*
|
||||
#endif
|
||||
));
|
||||
if (rd_priv_ctx->log == NULL) {
|
||||
fprintf(stderr, "pl_log_create failed!\n");
|
||||
free(rd_priv_ctx);
|
||||
return NULL;
|
||||
}
|
||||
@@ -118,24 +129,32 @@ static Dav1dPlayRendererPrivateContext*
|
||||
return rd_priv_ctx;
|
||||
}
|
||||
|
||||
#ifdef HAVE_PLACEBO_OPENGL
|
||||
static void *placebo_renderer_create_gl(void)
|
||||
#if HAVE_PLACEBO_OPENGL
|
||||
static void *placebo_renderer_create_gl(const Dav1dPlaySettings *settings)
|
||||
{
|
||||
SDL_Window *sdlwin = NULL;
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_FLAGS, SDL_GL_CONTEXT_DEBUG_FLAG);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_MAJOR_VERSION, 3);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_MINOR_VERSION, 0);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_PROFILE_MASK, SDL_GL_CONTEXT_PROFILE_CORE);
|
||||
|
||||
// Common init
|
||||
Dav1dPlayRendererPrivateContext *rd_priv_ctx =
|
||||
placebo_renderer_create_common(SDL_WINDOW_OPENGL);
|
||||
placebo_renderer_create_common(settings, SDL_WINDOW_OPENGL);
|
||||
|
||||
if (rd_priv_ctx == NULL)
|
||||
return NULL;
|
||||
sdlwin = rd_priv_ctx->win;
|
||||
|
||||
SDL_GLContext glcontext = SDL_GL_CreateContext(sdlwin);
|
||||
SDL_GL_MakeCurrent(sdlwin, glcontext);
|
||||
rd_priv_ctx->gl_context = SDL_GL_CreateContext(sdlwin);
|
||||
if (!rd_priv_ctx->gl_context) {
|
||||
fprintf(stderr, "Failed creating opengl context: %s\n", SDL_GetError());
|
||||
exit(2);
|
||||
}
|
||||
SDL_GL_MakeCurrent(sdlwin, rd_priv_ctx->gl_context);
|
||||
|
||||
rd_priv_ctx->gl = pl_opengl_create(rd_priv_ctx->log, pl_opengl_params(
|
||||
.allow_software = true,
|
||||
#ifndef NDEBUG
|
||||
.debug = true,
|
||||
#endif
|
||||
@@ -173,14 +192,14 @@ static void *placebo_renderer_create_gl(void)
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef HAVE_PLACEBO_VULKAN
|
||||
static void *placebo_renderer_create_vk(void)
|
||||
#if HAVE_PLACEBO_VULKAN
|
||||
static void *placebo_renderer_create_vk(const Dav1dPlaySettings *settings)
|
||||
{
|
||||
SDL_Window *sdlwin = NULL;
|
||||
|
||||
// Common init
|
||||
Dav1dPlayRendererPrivateContext *rd_priv_ctx =
|
||||
placebo_renderer_create_common(SDL_WINDOW_VULKAN);
|
||||
placebo_renderer_create_common(settings, SDL_WINDOW_VULKAN);
|
||||
|
||||
if (rd_priv_ctx == NULL)
|
||||
return NULL;
|
||||
@@ -270,16 +289,18 @@ static void placebo_renderer_destroy(void *cookie)
|
||||
for (int i = 0; i < 3; i++)
|
||||
pl_tex_destroy(rd_priv_ctx->gpu, &(rd_priv_ctx->plane_tex[i]));
|
||||
|
||||
#ifdef HAVE_PLACEBO_VULKAN
|
||||
#if HAVE_PLACEBO_VULKAN
|
||||
if (rd_priv_ctx->vk) {
|
||||
pl_vulkan_destroy(&(rd_priv_ctx->vk));
|
||||
vkDestroySurfaceKHR(rd_priv_ctx->vk_inst->instance, rd_priv_ctx->surf, NULL);
|
||||
pl_vk_inst_destroy(&(rd_priv_ctx->vk_inst));
|
||||
}
|
||||
#endif
|
||||
#ifdef HAVE_PLACEBO_OPENGL
|
||||
#if HAVE_PLACEBO_OPENGL
|
||||
if (rd_priv_ctx->gl)
|
||||
pl_opengl_destroy(&(rd_priv_ctx->gl));
|
||||
if (rd_priv_ctx->gl_context)
|
||||
SDL_GL_DeleteContext(rd_priv_ctx->gl_context);
|
||||
#endif
|
||||
|
||||
SDL_DestroyWindow(rd_priv_ctx->win);
|
||||
@@ -382,7 +403,7 @@ static void placebo_release_pic(Dav1dPicture *pic, void *cookie)
|
||||
SDL_UnlockMutex(rd_priv_ctx->lock);
|
||||
}
|
||||
|
||||
#ifdef HAVE_PLACEBO_VULKAN
|
||||
#if HAVE_PLACEBO_VULKAN
|
||||
const Dav1dPlayRenderInfo rdr_placebo_vk = {
|
||||
.name = "placebo-vk",
|
||||
.create_renderer = placebo_renderer_create_vk,
|
||||
@@ -397,7 +418,7 @@ const Dav1dPlayRenderInfo rdr_placebo_vk = {
|
||||
const Dav1dPlayRenderInfo rdr_placebo_vk = { NULL };
|
||||
#endif
|
||||
|
||||
#ifdef HAVE_PLACEBO_OPENGL
|
||||
#if HAVE_PLACEBO_OPENGL
|
||||
const Dav1dPlayRenderInfo rdr_placebo_gl = {
|
||||
.name = "placebo-gl",
|
||||
.create_renderer = placebo_renderer_create_gl,
|
||||
|
||||
@@ -43,15 +43,23 @@ typedef struct renderer_priv_ctx
|
||||
SDL_Texture *tex;
|
||||
} Dav1dPlayRendererPrivateContext;
|
||||
|
||||
static void *sdl_renderer_create(void)
|
||||
static void *sdl_renderer_create(const Dav1dPlaySettings *settings)
|
||||
{
|
||||
SDL_Window *win = dp_create_sdl_window(0);
|
||||
if (win == NULL)
|
||||
int window_flags = 0;
|
||||
if (settings->fullscreen)
|
||||
window_flags |= SDL_WINDOW_FULLSCREEN_DESKTOP;
|
||||
|
||||
SDL_Window *win = dp_create_sdl_window(window_flags);
|
||||
if (win == NULL) {
|
||||
fprintf(stderr, "Creating SDL window failed: %s\n", SDL_GetError());
|
||||
return NULL;
|
||||
}
|
||||
SDL_ShowCursor(0);
|
||||
|
||||
// Alloc
|
||||
Dav1dPlayRendererPrivateContext *rd_priv_ctx = malloc(sizeof(Dav1dPlayRendererPrivateContext));
|
||||
if (rd_priv_ctx == NULL) {
|
||||
fprintf(stderr, "Out of memory!\n");
|
||||
return NULL;
|
||||
}
|
||||
rd_priv_ctx->win = win;
|
||||
@@ -79,7 +87,9 @@ static void sdl_renderer_destroy(void *cookie)
|
||||
Dav1dPlayRendererPrivateContext *rd_priv_ctx = cookie;
|
||||
assert(rd_priv_ctx != NULL);
|
||||
|
||||
SDL_DestroyTexture(rd_priv_ctx->tex);
|
||||
SDL_DestroyRenderer(rd_priv_ctx->renderer);
|
||||
SDL_DestroyWindow(rd_priv_ctx->win);
|
||||
SDL_DestroyMutex(rd_priv_ctx->lock);
|
||||
free(rd_priv_ctx);
|
||||
}
|
||||
@@ -142,6 +152,7 @@ static int sdl_update_texture(void *cookie, Dav1dPicture *dav1d_pic,
|
||||
if (texture == NULL) {
|
||||
texture = SDL_CreateTexture(rd_priv_ctx->renderer, SDL_PIXELFORMAT_IYUV,
|
||||
SDL_TEXTUREACCESS_STREAMING, width, height);
|
||||
SDL_RenderSetLogicalSize(rd_priv_ctx->renderer, width, height);
|
||||
}
|
||||
|
||||
SDL_UpdateYUVTexture(texture, NULL,
|
||||
|
||||
@@ -40,7 +40,7 @@ dav1dplay_sources = files(
|
||||
'dp_renderer_sdl.c',
|
||||
)
|
||||
|
||||
sdl2_dependency = dependency('sdl2', version: '>= 2.0.1', required: true)
|
||||
sdl2_dependency = dependency('sdl2', version: '>= 2.0.1', required: true, include_type: 'system')
|
||||
|
||||
if sdl2_dependency.found()
|
||||
dav1dplay_deps = [sdl2_dependency, libm_dependency]
|
||||
@@ -48,19 +48,23 @@ if sdl2_dependency.found()
|
||||
|
||||
placebo_dependency = dependency('libplacebo', version: '>= 4.160.0', required: false)
|
||||
|
||||
if placebo_dependency.found()
|
||||
have_vulkan = false
|
||||
have_placebo = placebo_dependency.found()
|
||||
if have_placebo
|
||||
dav1dplay_deps += placebo_dependency
|
||||
dav1dplay_cflags += '-DHAVE_PLACEBO'
|
||||
|
||||
# If libplacebo is found, we might be able to use Vulkan
|
||||
# with it, in which case we need the Vulkan library too.
|
||||
vulkan_dependency = dependency('vulkan', required: false)
|
||||
if vulkan_dependency.found()
|
||||
dav1dplay_deps += vulkan_dependency
|
||||
dav1dplay_cflags += '-DHAVE_VULKAN'
|
||||
have_vulkan = true
|
||||
endif
|
||||
endif
|
||||
|
||||
dav1dplay_cflags += '-DHAVE_PLACEBO=' + (have_placebo ? '1' : '0')
|
||||
dav1dplay_cflags += '-DHAVE_VULKAN=' + (have_vulkan ? '1' : '0')
|
||||
|
||||
dav1dplay = executable('dav1dplay',
|
||||
dav1dplay_sources,
|
||||
rev_target,
|
||||
|
||||
@@ -123,6 +123,12 @@
|
||||
#define EXTERN extern
|
||||
#endif
|
||||
|
||||
#if ARCH_X86_64 && __has_attribute(model)
|
||||
#define ATTR_MCMODEL_SMALL __attribute__((model("small")))
|
||||
#else
|
||||
#define ATTR_MCMODEL_SMALL
|
||||
#endif
|
||||
|
||||
#ifdef __clang__
|
||||
#define NO_SANITIZE(x) __attribute__((no_sanitize(x)))
|
||||
#else
|
||||
@@ -189,9 +195,13 @@ static inline int clzll(const unsigned long long mask) {
|
||||
#ifndef static_assert
|
||||
#define CHECK_OFFSET(type, field, name) \
|
||||
struct check_##type##_##field { int x[(name == offsetof(type, field)) ? 1 : -1]; }
|
||||
#define CHECK_SIZE(type, size) \
|
||||
struct check_##type##_size { int x[(size == sizeof(type)) ? 1 : -1]; }
|
||||
#else
|
||||
#define CHECK_OFFSET(type, field, name) \
|
||||
static_assert(name == offsetof(type, field), #field)
|
||||
#define CHECK_SIZE(type, size) \
|
||||
static_assert(size == sizeof(type), #type)
|
||||
#endif
|
||||
|
||||
#ifdef _MSC_VER
|
||||
|
||||
@@ -65,11 +65,11 @@ static inline int apply_sign64(const int v, const int64_t s) {
|
||||
}
|
||||
|
||||
static inline int ulog2(const unsigned v) {
|
||||
return 31 - clz(v);
|
||||
return 31 ^ clz(v);
|
||||
}
|
||||
|
||||
static inline int u64log2(const uint64_t v) {
|
||||
return 63 - clzll(v);
|
||||
return 63 ^ clzll(v);
|
||||
}
|
||||
|
||||
static inline unsigned inv_recenter(const unsigned r, const unsigned v) {
|
||||
|
||||
@@ -12,9 +12,6 @@
|
||||
|
||||
#define __GETOPT_H__
|
||||
|
||||
/* All the headers include this file. */
|
||||
#include <crtdefs.h>
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
@@ -43,6 +43,8 @@ extern "C" {
|
||||
#else
|
||||
#define DAV1D_API
|
||||
#endif
|
||||
#elif defined __OS2__
|
||||
#define DAV1D_API __declspec(dllexport)
|
||||
#else
|
||||
#if __GNUC__ >= 4
|
||||
#define DAV1D_API __attribute__ ((visibility ("default")))
|
||||
|
||||
@@ -187,14 +187,10 @@ typedef struct Dav1dContentLightLevel {
|
||||
} Dav1dContentLightLevel;
|
||||
|
||||
typedef struct Dav1dMasteringDisplay {
|
||||
///< 0.16 fixed point
|
||||
uint16_t primaries[3][2];
|
||||
///< 0.16 fixed point
|
||||
uint16_t white_point[2];
|
||||
///< 24.8 fixed point
|
||||
uint32_t max_luminance;
|
||||
///< 18.14 fixed point
|
||||
uint32_t min_luminance;
|
||||
uint16_t primaries[3][2]; ///< 0.16 fixed point
|
||||
uint16_t white_point[2]; ///< 0.16 fixed point
|
||||
uint32_t max_luminance; ///< 24.8 fixed point
|
||||
uint32_t min_luminance; ///< 18.14 fixed point
|
||||
} Dav1dMasteringDisplay;
|
||||
|
||||
typedef struct Dav1dITUTT35 {
|
||||
|
||||
@@ -22,24 +22,15 @@
|
||||
# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
|
||||
# SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
# installed version.h header generation
|
||||
version_h_data = configuration_data()
|
||||
version_h_data.set('DAV1D_API_VERSION_MAJOR', dav1d_api_version_major)
|
||||
version_h_data.set('DAV1D_API_VERSION_MINOR', dav1d_api_version_minor)
|
||||
version_h_data.set('DAV1D_API_VERSION_PATCH', dav1d_api_version_revision)
|
||||
version_h_target = configure_file(input: 'version.h.in',
|
||||
output: 'version.h',
|
||||
configuration: version_h_data)
|
||||
|
||||
dav1d_api_headers = [
|
||||
'common.h',
|
||||
'data.h',
|
||||
'dav1d.h',
|
||||
'headers.h',
|
||||
'picture.h',
|
||||
'version.h',
|
||||
]
|
||||
|
||||
# install headers
|
||||
install_headers(dav1d_api_headers,
|
||||
version_h_target,
|
||||
subdir : 'dav1d')
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* Copyright © 2019, VideoLAN and dav1d authors
|
||||
* Copyright © 2019-2024, VideoLAN and dav1d authors
|
||||
* All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
@@ -31,9 +31,9 @@
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#define DAV1D_API_VERSION_MAJOR @DAV1D_API_VERSION_MAJOR@
|
||||
#define DAV1D_API_VERSION_MINOR @DAV1D_API_VERSION_MINOR@
|
||||
#define DAV1D_API_VERSION_PATCH @DAV1D_API_VERSION_PATCH@
|
||||
#define DAV1D_API_VERSION_MAJOR 7
|
||||
#define DAV1D_API_VERSION_MINOR 0
|
||||
#define DAV1D_API_VERSION_PATCH 0
|
||||
|
||||
/**
|
||||
* Extract version components from the value returned by
|
||||
+147
-105
@@ -1,4 +1,4 @@
|
||||
# Copyright © 2018-2022, VideoLAN and dav1d authors
|
||||
# Copyright © 2018-2026, VideoLAN and dav1d authors
|
||||
# All rights reserved.
|
||||
#
|
||||
# Redistribution and use in source and binary forms, with or without
|
||||
@@ -23,18 +23,12 @@
|
||||
# SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
project('dav1d', ['c'],
|
||||
version: '1.4.2',
|
||||
version: '1.5.4',
|
||||
default_options: ['c_std=c99',
|
||||
'warning_level=2',
|
||||
'buildtype=release',
|
||||
'b_ndebug=if-release'],
|
||||
meson_version: '>= 0.49.0')
|
||||
|
||||
dav1d_soname_version = '7.0.0'
|
||||
dav1d_api_version_array = dav1d_soname_version.split('.')
|
||||
dav1d_api_version_major = dav1d_api_version_array[0]
|
||||
dav1d_api_version_minor = dav1d_api_version_array[1]
|
||||
dav1d_api_version_revision = dav1d_api_version_array[2]
|
||||
meson_version: '>= 0.54.0')
|
||||
|
||||
dav1d_src_root = meson.current_source_dir()
|
||||
cc = meson.get_compiler('c')
|
||||
@@ -48,7 +42,18 @@ cdata_asm = configuration_data()
|
||||
# Include directories
|
||||
dav1d_inc_dirs = include_directories(['.', 'include/dav1d', 'include'])
|
||||
|
||||
|
||||
dav1d_api_version_major = cc.get_define('DAV1D_API_VERSION_MAJOR',
|
||||
prefix: '#include "dav1d/version.h"',
|
||||
include_directories: dav1d_inc_dirs).strip()
|
||||
dav1d_api_version_minor = cc.get_define('DAV1D_API_VERSION_MINOR',
|
||||
prefix: '#include "dav1d/version.h"',
|
||||
include_directories: dav1d_inc_dirs).strip()
|
||||
dav1d_api_version_revision = cc.get_define('DAV1D_API_VERSION_PATCH',
|
||||
prefix: '#include "dav1d/version.h"',
|
||||
include_directories: dav1d_inc_dirs).strip()
|
||||
dav1d_soname_version = '@0@.@1@.@2@'.format(dav1d_api_version_major,
|
||||
dav1d_api_version_minor,
|
||||
dav1d_api_version_revision)
|
||||
|
||||
#
|
||||
# Option handling
|
||||
@@ -81,8 +86,6 @@ cdata.set10('TRIM_DSP_FUNCTIONS', get_option('trim_dsp') == 'true' or
|
||||
# Logging option
|
||||
cdata.set10('CONFIG_LOG', get_option('logging'))
|
||||
|
||||
cdata.set10('CONFIG_MACOS_KPERF', get_option('macos_kperf'))
|
||||
|
||||
#
|
||||
# OS/Compiler checks and defines
|
||||
#
|
||||
@@ -98,6 +101,11 @@ if host_machine.system() in ['linux', 'gnu', 'emscripten']
|
||||
add_project_arguments('-D_GNU_SOURCE', language: 'c')
|
||||
endif
|
||||
|
||||
have_clock_gettime = false
|
||||
have_sigaction = false
|
||||
have_posix_memalign = false
|
||||
have_memalign = false
|
||||
have_aligned_alloc = false
|
||||
if host_machine.system() == 'windows'
|
||||
cdata.set('_WIN32_WINNT', '0x0601')
|
||||
cdata.set('UNICODE', 1) # Define to 1 for Unicode (Wide Chars) APIs
|
||||
@@ -138,27 +146,34 @@ if host_machine.system() == 'windows'
|
||||
rc_data.set('API_VERSION_MAJOR', dav1d_api_version_major)
|
||||
rc_data.set('API_VERSION_MINOR', dav1d_api_version_minor)
|
||||
rc_data.set('API_VERSION_REVISION', dav1d_api_version_revision)
|
||||
rc_data.set('COPYRIGHT_YEARS', '2018-2024')
|
||||
rc_data.set('COPYRIGHT_YEARS', '2018-2026')
|
||||
else
|
||||
thread_dependency = dependency('threads')
|
||||
thread_compat_dep = []
|
||||
|
||||
rt_dependency = []
|
||||
if cc.has_function('clock_gettime', prefix : '#include <time.h>', args : test_args)
|
||||
cdata.set('HAVE_CLOCK_GETTIME', 1)
|
||||
have_clock_gettime = true
|
||||
elif host_machine.system() not in ['darwin', 'ios', 'tvos']
|
||||
rt_dependency = cc.find_library('rt', required: false)
|
||||
if not cc.has_function('clock_gettime', prefix : '#include <time.h>', args : test_args, dependencies : rt_dependency)
|
||||
error('clock_gettime not found')
|
||||
endif
|
||||
cdata.set('HAVE_CLOCK_GETTIME', 1)
|
||||
have_clock_gettime = true
|
||||
endif
|
||||
|
||||
if cc.has_function('posix_memalign', prefix : '#include <stdlib.h>', args : test_args)
|
||||
cdata.set('HAVE_POSIX_MEMALIGN', 1)
|
||||
endif
|
||||
have_sigaction = cc.has_function('sigaction', prefix : '#include <signal.h>', args : test_args)
|
||||
have_posix_memalign = cc.has_function('posix_memalign', prefix : '#include <stdlib.h>', args : test_args)
|
||||
have_memalign = cc.has_function('memalign', prefix : '#include <malloc.h>', args : test_args)
|
||||
have_aligned_alloc = cc.has_function('aligned_alloc', prefix : '#include <stdlib.h>', args : test_args)
|
||||
endif
|
||||
|
||||
cdata.set10('HAVE_CLOCK_GETTIME', have_clock_gettime)
|
||||
cdata.set10('HAVE_SIGACTION', have_sigaction)
|
||||
cdata.set10('HAVE_POSIX_MEMALIGN', have_posix_memalign)
|
||||
cdata.set10('HAVE_MEMALIGN', have_memalign)
|
||||
cdata.set10('HAVE_ALIGNED_ALLOC', have_aligned_alloc)
|
||||
|
||||
# check for fseeko on android. It is not always available if _FILE_OFFSET_BITS is defined to 64
|
||||
have_fseeko = true
|
||||
if host_machine.system() == 'android'
|
||||
@@ -175,12 +190,12 @@ if host_machine.system() == 'android'
|
||||
endif
|
||||
|
||||
libdl_dependency = []
|
||||
have_dlsym = false
|
||||
if host_machine.system() == 'linux'
|
||||
libdl_dependency = cc.find_library('dl', required : false)
|
||||
if cc.has_function('dlsym', prefix : '#include <dlfcn.h>', args : test_args, dependencies : libdl_dependency)
|
||||
cdata.set('HAVE_DLSYM', 1)
|
||||
endif
|
||||
have_dlsym = cc.has_function('dlsym', prefix : '#include <dlfcn.h>', args : test_args, dependencies : libdl_dependency)
|
||||
endif
|
||||
cdata.set10('HAVE_DLSYM', have_dlsym)
|
||||
|
||||
libm_dependency = cc.find_library('m', required: false)
|
||||
|
||||
@@ -209,19 +224,13 @@ if host_machine.cpu_family().startswith('wasm')
|
||||
stdatomic_dependencies += thread_dependency.partial_dependency(compile_args: true)
|
||||
endif
|
||||
|
||||
if cc.check_header('unistd.h')
|
||||
cdata.set('HAVE_UNISTD_H', 1)
|
||||
endif
|
||||
|
||||
if cc.check_header('io.h')
|
||||
cdata.set('HAVE_IO_H', 1)
|
||||
endif
|
||||
|
||||
if cc.check_header('pthread_np.h')
|
||||
cdata.set('HAVE_PTHREAD_NP_H', 1)
|
||||
test_args += '-DHAVE_PTHREAD_NP_H'
|
||||
endif
|
||||
cdata.set10('HAVE_SYS_TYPES_H', cc.check_header('sys/types.h'))
|
||||
cdata.set10('HAVE_UNISTD_H', cc.check_header('unistd.h'))
|
||||
cdata.set10('HAVE_IO_H', cc.check_header('io.h'))
|
||||
|
||||
have_pthread_np = cc.check_header('pthread_np.h')
|
||||
cdata.set10('HAVE_PTHREAD_NP_H', have_pthread_np)
|
||||
test_args += '-DHAVE_PTHREAD_NP_H=' + (have_pthread_np ? '1' : '0')
|
||||
|
||||
# Function checks
|
||||
|
||||
@@ -234,35 +243,32 @@ else
|
||||
getopt_dependency = []
|
||||
endif
|
||||
|
||||
have_getauxval = false
|
||||
have_elf_aux_info = false
|
||||
if (host_machine.cpu_family() == 'aarch64' or
|
||||
host_machine.cpu_family().startswith('arm') or
|
||||
host_machine.cpu_family().startswith('loongarch') or
|
||||
host_machine.cpu() == 'ppc64le' or
|
||||
host_machine.cpu_family().startswith('riscv'))
|
||||
if cc.has_function('getauxval', prefix : '#include <sys/auxv.h>', args : test_args)
|
||||
cdata.set('HAVE_GETAUXVAL', 1)
|
||||
endif
|
||||
if cc.has_function('elf_aux_info', prefix : '#include <sys/auxv.h>', args : test_args)
|
||||
cdata.set('HAVE_ELF_AUX_INFO', 1)
|
||||
endif
|
||||
have_getauxval = cc.has_function('getauxval', prefix : '#include <sys/auxv.h>', args : test_args)
|
||||
have_elf_aux_info = cc.has_function('elf_aux_info', prefix : '#include <sys/auxv.h>', args : test_args)
|
||||
endif
|
||||
|
||||
cdata.set10('HAVE_GETAUXVAL', have_getauxval)
|
||||
cdata.set10('HAVE_ELF_AUX_INFO', have_elf_aux_info)
|
||||
|
||||
pthread_np_prefix = '''
|
||||
#include <pthread.h>
|
||||
#ifdef HAVE_PTHREAD_NP_H
|
||||
#if HAVE_PTHREAD_NP_H
|
||||
#include <pthread_np.h>
|
||||
#endif
|
||||
'''
|
||||
if cc.has_function('pthread_getaffinity_np', prefix : pthread_np_prefix, args : test_args, dependencies : thread_dependency)
|
||||
cdata.set('HAVE_PTHREAD_GETAFFINITY_NP', 1)
|
||||
endif
|
||||
if cc.has_function('pthread_setaffinity_np', prefix : pthread_np_prefix, args : test_args, dependencies : thread_dependency)
|
||||
cdata.set('HAVE_PTHREAD_SETAFFINITY_NP', 1)
|
||||
endif
|
||||
cdata.set10('HAVE_PTHREAD_GETAFFINITY_NP', cc.has_function('pthread_getaffinity_np', prefix : pthread_np_prefix, args : test_args, dependencies : thread_dependency))
|
||||
cdata.set10('HAVE_PTHREAD_SETAFFINITY_NP', cc.has_function('pthread_setaffinity_np', prefix : pthread_np_prefix, args : test_args, dependencies : thread_dependency))
|
||||
cdata.set10('HAVE_PTHREAD_SETNAME_NP', cc.has_function('pthread_setname_np', prefix : pthread_np_prefix, args : test_args, dependencies : thread_dependency))
|
||||
cdata.set10('HAVE_PTHREAD_SET_NAME_NP', cc.has_function('pthread_set_name_np', prefix : pthread_np_prefix, args : test_args, dependencies : thread_dependency))
|
||||
|
||||
if cc.compiles('int x = _Generic(0, default: 0);', name: '_Generic', args: test_args)
|
||||
cdata.set('HAVE_C11_GENERIC', 1)
|
||||
endif
|
||||
cdata.set10('HAVE_C11_GENERIC', cc.compiles('int x = _Generic(0, default: 0);', name: '_Generic', args: test_args))
|
||||
|
||||
# Compiler flag tests
|
||||
|
||||
@@ -296,7 +302,8 @@ else
|
||||
optional_arguments += [
|
||||
'-wd4028', # parameter different from declaration
|
||||
'-wd4090', # broken with arrays of pointers
|
||||
'-wd4996' # use of POSIX functions
|
||||
'-wd4996', # use of POSIX functions
|
||||
'-wd5287', # operands are different enum types
|
||||
]
|
||||
endif
|
||||
|
||||
@@ -341,8 +348,49 @@ if host_machine.cpu_family().startswith('x86')
|
||||
cdata_asm.set('STACK_ALIGNMENT', stack_alignment)
|
||||
endif
|
||||
|
||||
#
|
||||
# ASM specific stuff
|
||||
#
|
||||
|
||||
use_gaspp = false
|
||||
if (is_asm_enabled and
|
||||
(host_machine.cpu_family() == 'aarch64' or
|
||||
host_machine.cpu_family().startswith('arm')) and
|
||||
cc.get_argument_syntax() == 'msvc' and
|
||||
(cc.get_id() != 'clang-cl' or meson.version().version_compare('<0.58.0')))
|
||||
gaspp = find_program('gas-preprocessor.pl')
|
||||
use_gaspp = true
|
||||
gaspp_args = [
|
||||
'-as-type', 'armasm',
|
||||
'-arch', host_machine.cpu_family(),
|
||||
'--',
|
||||
host_machine.cpu_family() == 'aarch64' ? 'armasm64' : 'armasm',
|
||||
'-nologo',
|
||||
'-I@0@'.format(dav1d_src_root),
|
||||
'-I@0@/'.format(meson.current_build_dir()),
|
||||
]
|
||||
gaspp_gen = generator(gaspp,
|
||||
output: '@BASENAME@.obj',
|
||||
arguments: gaspp_args + [
|
||||
'@INPUT@',
|
||||
'-c',
|
||||
'-o', '@OUTPUT@'
|
||||
])
|
||||
endif
|
||||
|
||||
cdata.set10('ARCH_AARCH64', host_machine.cpu_family() == 'aarch64' or host_machine.cpu() == 'arm64')
|
||||
cdata.set10('ARCH_ARM', host_machine.cpu_family().startswith('arm') and host_machine.cpu() != 'arm64')
|
||||
|
||||
have_as_func = false
|
||||
have_as_arch = false
|
||||
aarch64_extensions = {
|
||||
'dotprod': 'udot v0.4s, v0.16b, v0.16b',
|
||||
'i8mm': 'usdot v0.4s, v0.16b, v0.16b',
|
||||
'sve': 'whilelt p0.s, x0, x1',
|
||||
'sve2': 'sqrdmulh z0.s, z0.s, z0.s',
|
||||
}
|
||||
supported_aarch64_archexts = []
|
||||
supported_aarch64_instructions = []
|
||||
if (is_asm_enabled and
|
||||
(host_machine.cpu_family() == 'aarch64' or
|
||||
host_machine.cpu_family().startswith('arm')))
|
||||
@@ -353,7 +401,6 @@ if (is_asm_enabled and
|
||||
);
|
||||
'''
|
||||
have_as_func = cc.compiles(as_func_code)
|
||||
cdata.set10('HAVE_AS_FUNC', have_as_func)
|
||||
|
||||
# fedora package build infrastructure uses a gcc specs file to enable
|
||||
# '-fPIE' by default. The chosen way only adds '-fPIE' to the C compiler
|
||||
@@ -374,7 +421,6 @@ if (is_asm_enabled and
|
||||
|
||||
if host_machine.cpu_family() == 'aarch64'
|
||||
have_as_arch = cc.compiles('''__asm__ (".arch armv8-a");''')
|
||||
cdata.set10('HAVE_AS_ARCH_DIRECTIVE', have_as_arch)
|
||||
as_arch_str = ''
|
||||
if have_as_arch
|
||||
as_arch_level = 'armv8-a'
|
||||
@@ -403,36 +449,53 @@ if (is_asm_enabled and
|
||||
cdata.set('AS_ARCH_LEVEL', as_arch_level)
|
||||
as_arch_str = '".arch ' + as_arch_level + '\\n"'
|
||||
endif
|
||||
extensions = {
|
||||
'dotprod': 'udot v0.4s, v0.16b, v0.16b',
|
||||
'i8mm': 'usdot v0.4s, v0.16b, v0.16b',
|
||||
'sve': 'whilelt p0.s, x0, x1',
|
||||
'sve2': 'sqrdmulh z0.s, z0.s, z0.s',
|
||||
}
|
||||
foreach name, instr : extensions
|
||||
# Test for support for the various extensions. First test if
|
||||
# the assembler supports the .arch_extension directive for
|
||||
# enabling/disabling the extension, then separately check whether
|
||||
# the instructions themselves are supported. Even if .arch_extension
|
||||
# isn't supported, we may be able to assemble the instructions
|
||||
# if the .arch level includes support for them.
|
||||
code = '__asm__ (' + as_arch_str
|
||||
code += '".arch_extension ' + name + '\\n"'
|
||||
code += ');'
|
||||
supports_archext = cc.compiles(code)
|
||||
cdata.set10('HAVE_AS_ARCHEXT_' + name.to_upper() + '_DIRECTIVE', supports_archext)
|
||||
code = '__asm__ (' + as_arch_str
|
||||
if supports_archext
|
||||
if use_gaspp
|
||||
python3 = import('python').find_installation()
|
||||
endif
|
||||
foreach name, instr : aarch64_extensions
|
||||
if use_gaspp
|
||||
f = configure_file(
|
||||
command: [python3, '-c', 'import sys; print(sys.argv[1])', '@0@'.format(instr)],
|
||||
output: 'test-@0@.S'.format(name),
|
||||
capture: true)
|
||||
r = run_command(gaspp, gaspp_args, f, '-c', '-o', meson.current_build_dir() / 'test-' + name + '.obj', check: false)
|
||||
message('Checking for gaspp/armasm64 ' + name.to_upper() + ': ' + (r.returncode() == 0 ? 'YES' : 'NO'))
|
||||
if r.returncode() == 0
|
||||
supported_aarch64_instructions += name
|
||||
endif
|
||||
else
|
||||
# Test for support for the various extensions. First test if
|
||||
# the assembler supports the .arch_extension directive for
|
||||
# enabling/disabling the extension, then separately check whether
|
||||
# the instructions themselves are supported. Even if .arch_extension
|
||||
# isn't supported, we may be able to assemble the instructions
|
||||
# if the .arch level includes support for them.
|
||||
code = '__asm__ (' + as_arch_str
|
||||
code += '".arch_extension ' + name + '\\n"'
|
||||
code += ');'
|
||||
supports_archext = cc.compiles(code)
|
||||
code = '__asm__ (' + as_arch_str
|
||||
if supports_archext
|
||||
supported_aarch64_archexts += name
|
||||
code += '".arch_extension ' + name + '\\n"'
|
||||
endif
|
||||
code += '"' + instr + '\\n"'
|
||||
code += ');'
|
||||
if cc.compiles(code, name: name.to_upper())
|
||||
supported_aarch64_instructions += name
|
||||
endif
|
||||
endif
|
||||
code += '"' + instr + '\\n"'
|
||||
code += ');'
|
||||
supports_instr = cc.compiles(code, name: name.to_upper())
|
||||
cdata.set10('HAVE_' + name.to_upper(), supports_instr)
|
||||
endforeach
|
||||
endif
|
||||
endif
|
||||
|
||||
cdata.set10('HAVE_AS_FUNC', have_as_func)
|
||||
cdata.set10('HAVE_AS_ARCH_DIRECTIVE', have_as_arch)
|
||||
foreach name, _ : aarch64_extensions
|
||||
cdata.set10('HAVE_AS_ARCHEXT_' + name.to_upper() + '_DIRECTIVE', name in supported_aarch64_archexts)
|
||||
cdata.set10('HAVE_' + name.to_upper(), name in supported_aarch64_instructions)
|
||||
endforeach
|
||||
|
||||
cdata.set10('ARCH_X86', host_machine.cpu_family().startswith('x86'))
|
||||
cdata.set10('ARCH_X86_64', host_machine.cpu_family() == 'x86_64')
|
||||
cdata.set10('ARCH_X86_32', host_machine.cpu_family() == 'x86')
|
||||
@@ -460,15 +523,12 @@ cdata.set10('ARCH_LOONGARCH64', host_machine.cpu_family() == 'loongarch64')
|
||||
# meson's cc.symbols_have_underscore_prefix() is unfortunately unrelieably
|
||||
# when additional flags like '-fprofile-instr-generate' are passed via CFLAGS
|
||||
# see following meson issue https://github.com/mesonbuild/meson/issues/5482
|
||||
if (host_machine.system() in ['darwin', 'ios', 'tvos'] or
|
||||
if (host_machine.system() in ['darwin', 'ios', 'tvos', 'os/2'] or
|
||||
(host_machine.system() == 'windows' and host_machine.cpu_family() == 'x86'))
|
||||
cdata.set10('PREFIX', true)
|
||||
cdata_asm.set10('PREFIX', true)
|
||||
endif
|
||||
|
||||
#
|
||||
# ASM specific stuff
|
||||
#
|
||||
if is_asm_enabled and host_machine.cpu_family().startswith('x86')
|
||||
|
||||
# NASM compiler support
|
||||
@@ -499,7 +559,13 @@ if is_asm_enabled and host_machine.cpu_family().startswith('x86')
|
||||
else
|
||||
nasm_format = 'elf'
|
||||
endif
|
||||
if host_machine.cpu_family() == 'x86_64'
|
||||
if host_machine.system() == 'os/2'
|
||||
if get_option('os2_emxomf')
|
||||
nasm_format = 'obj2'
|
||||
else
|
||||
nasm_format = 'aout'
|
||||
endif
|
||||
elif host_machine.cpu_family() == 'x86_64'
|
||||
nasm_format += '64'
|
||||
else
|
||||
nasm_format += '32'
|
||||
@@ -519,30 +585,6 @@ if is_asm_enabled and host_machine.cpu_family().startswith('x86')
|
||||
])
|
||||
endif
|
||||
|
||||
use_gaspp = false
|
||||
if (is_asm_enabled and
|
||||
(host_machine.cpu_family() == 'aarch64' or
|
||||
host_machine.cpu_family().startswith('arm')) and
|
||||
cc.get_argument_syntax() == 'msvc' and
|
||||
(cc.get_id() != 'clang-cl' or meson.version().version_compare('<0.58.0')))
|
||||
gaspp = find_program('gas-preprocessor.pl')
|
||||
use_gaspp = true
|
||||
gaspp_gen = generator(gaspp,
|
||||
output: '@BASENAME@.obj',
|
||||
arguments: [
|
||||
'-as-type', 'armasm',
|
||||
'-arch', host_machine.cpu_family(),
|
||||
'--',
|
||||
host_machine.cpu_family() == 'aarch64' ? 'armasm64' : 'armasm',
|
||||
'-nologo',
|
||||
'-I@0@'.format(dav1d_src_root),
|
||||
'-I@0@/'.format(meson.current_build_dir()),
|
||||
'@INPUT@',
|
||||
'-c',
|
||||
'-o', '@OUTPUT@'
|
||||
])
|
||||
endif
|
||||
|
||||
if is_asm_enabled and host_machine.cpu_family().startswith('riscv')
|
||||
as_option_code = '''__asm__ (
|
||||
".option arch, +v\n"
|
||||
|
||||
@@ -68,8 +68,3 @@ option('trim_dsp',
|
||||
choices: ['true', 'false', 'if-release'],
|
||||
value: 'if-release',
|
||||
description: 'Eliminate redundant DSP functions where possible')
|
||||
|
||||
option('macos_kperf',
|
||||
type: 'boolean',
|
||||
value: false,
|
||||
description: 'Use the private macOS kperf API for benchmarking')
|
||||
|
||||
@@ -3,7 +3,7 @@ c = 'clang'
|
||||
cpp = 'clang++'
|
||||
ar = 'aarch64-linux-gnu-ar'
|
||||
strip = 'aarch64-linux-gnu-strip'
|
||||
exe_wrapper = 'qemu-aarch64'
|
||||
exe_wrapper = ['qemu-aarch64', '-L', '/usr/aarch64-linux-gnu/']
|
||||
|
||||
[properties]
|
||||
c_args = '-target aarch64-linux-gnu'
|
||||
|
||||
@@ -3,7 +3,7 @@ c = 'aarch64-linux-gnu-gcc'
|
||||
cpp = 'aarch64-linux-gnu-g++'
|
||||
ar = 'aarch64-linux-gnu-ar'
|
||||
strip = 'aarch64-linux-gnu-strip'
|
||||
exe_wrapper = 'qemu-aarch64'
|
||||
exe_wrapper = ['qemu-aarch64', '-L', '/usr/aarch64-linux-gnu/']
|
||||
|
||||
[host_machine]
|
||||
system = 'linux'
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
[binaries]
|
||||
c = ['clang', '-arch', 'arm64', '-isysroot', '/Applications/Xcode.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS.sdk']
|
||||
cpp = ['clang++', '-arch', 'arm64', '-isysroot', '/Applications/Xcode.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS.sdk']
|
||||
objc = ['clang', '-arch', 'arm64', '-isysroot', '/Applications/Xcode.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS.sdk']
|
||||
objcpp = ['clang++', '-arch', 'arm64', '-isysroot', '/Applications/Xcode.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS.sdk']
|
||||
ar = 'ar'
|
||||
strip = 'strip'
|
||||
|
||||
[built-in options]
|
||||
c_args = ['-miphoneos-version-min=11.0']
|
||||
cpp_args = ['-miphoneos-version-min=11.0']
|
||||
c_link_args = ['-miphoneos-version-min=11.0']
|
||||
cpp_link_args = ['-miphoneos-version-min=11.0']
|
||||
objc_args = ['-miphoneos-version-min=11.0']
|
||||
objcpp_args = ['-miphoneos-version-min=11.0']
|
||||
|
||||
[properties]
|
||||
root = '/Applications/Xcode.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer'
|
||||
needs_exe_wrapper = true
|
||||
|
||||
[host_machine]
|
||||
system = 'darwin'
|
||||
subsystem = 'ios'
|
||||
kernel = 'xnu'
|
||||
cpu_family = 'aarch64'
|
||||
cpu = 'aarch64'
|
||||
endian = 'little'
|
||||
@@ -1,10 +1,10 @@
|
||||
[binaries]
|
||||
c = 'loongarch64-unknown-linux-gnu-gcc'
|
||||
cpp = 'loongarch64-unknown-linux-gnu-c++'
|
||||
ar = 'loongarch64-unknown-linux-gnu-ar'
|
||||
strip = 'loongarch64-unknown-linux-gnu-strip'
|
||||
c = 'loongarch64-linux-gnu-gcc'
|
||||
cpp = 'loongarch64-linux-gnu-c++'
|
||||
ar = 'loongarch64-linux-gnu-ar'
|
||||
strip = 'loongarch64-linux-gnu-strip'
|
||||
pkgconfig = 'pkg-config'
|
||||
exe_wrapper = 'qemu-loongarch64'
|
||||
exe_wrapper = ['qemu-loongarch64', '-L', '/usr/loongarch64-linux-gnu/']
|
||||
|
||||
[host_machine]
|
||||
system = 'linux'
|
||||
|
||||
@@ -3,7 +3,7 @@ c = 'clang'
|
||||
cpp = 'clang++'
|
||||
ar = 'riscv64-linux-gnu-ar'
|
||||
strip = 'riscv64-linux-gnu-strip'
|
||||
exe_wrapper = 'qemu-riscv64'
|
||||
exe_wrapper = ['qemu-riscv64', '-L', '/usr/riscv64-linux-gnu/']
|
||||
|
||||
[properties]
|
||||
c_args = '-target riscv64-linux-gnu'
|
||||
|
||||
@@ -3,7 +3,7 @@ c = 'riscv64-linux-gnu-gcc'
|
||||
cpp = 'riscv64-linux-gnu-g++'
|
||||
ar = 'riscv64-linux-gnu-ar'
|
||||
strip = 'riscv64-linux-gnu-strip'
|
||||
exe_wrapper = 'qemu-riscv64'
|
||||
exe_wrapper = ['qemu-riscv64', '-L', '/usr/riscv64-linux-gnu/']
|
||||
|
||||
[host_machine]
|
||||
system = 'linux'
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
[binaries]
|
||||
c = ['clang', '-arch', 'x86_64', '-isysroot', '/Applications/Xcode.app/Contents/Developer/Platforms/iPhoneSimulator.platform/Developer/SDKs/iPhoneSimulator.sdk']
|
||||
cpp = ['clang++', '-arch', 'x86_64', '-isysroot', '/Applications/Xcode.app/Contents/Developer/Platforms/iPhoneSimulator.platform/Developer/SDKs/iPhoneSimulator.sdk']
|
||||
objc = ['clang', '-arch', 'x86_64', '-isysroot', '/Applications/Xcode.app/Contents/Developer/Platforms/iPhoneSimulator.platform/Developer/SDKs/iPhoneSimulator.sdk']
|
||||
objcpp = ['clang++', '-arch', 'x86_64', '-isysroot', '/Applications/Xcode.app/Contents/Developer/Platforms/iPhoneSimulator.platform/Developer/SDKs/iPhoneSimulator.sdk']
|
||||
ar = 'ar'
|
||||
strip = 'strip'
|
||||
|
||||
[built-in options]
|
||||
c_args = ['-miphoneos-version-min=11.0']
|
||||
cpp_args = ['-miphoneos-version-min=11.0']
|
||||
c_link_args = ['-miphoneos-version-min=11.0']
|
||||
cpp_link_args = ['-miphoneos-version-min=11.0']
|
||||
objc_args = ['-miphoneos-version-min=11.0']
|
||||
objcpp_args = ['-miphoneos-version-min=11.0']
|
||||
|
||||
[properties]
|
||||
root = '/Applications/Xcode.app/Contents/Developer/Platforms/iPhoneSimulator.platform/Developer'
|
||||
needs_exe_wrapper = true
|
||||
|
||||
[host_machine]
|
||||
system = 'darwin'
|
||||
subsystem = 'ios-simulator'
|
||||
kernel = 'xnu'
|
||||
cpu_family = 'x86_64'
|
||||
cpu = 'x86_64'
|
||||
endian = 'little'
|
||||
+2
-2
@@ -540,7 +540,7 @@ L(ipred_dc_left_w64):
|
||||
vst1.8 {d0, d1, d2, d3}, [r12, :128]!
|
||||
vst1.8 {d0, d1, d2, d3}, [r0, :128], r1
|
||||
vst1.8 {d0, d1, d2, d3}, [r12, :128], r1
|
||||
subs r4, r4, #4
|
||||
subs r4, r4, #4
|
||||
vst1.8 {d0, d1, d2, d3}, [r0, :128]!
|
||||
vst1.8 {d0, d1, d2, d3}, [r12, :128]!
|
||||
vst1.8 {d0, d1, d2, d3}, [r0, :128], r1
|
||||
@@ -692,7 +692,7 @@ L(ipred_dc_w16):
|
||||
2:
|
||||
vst1.8 {d0, d1}, [r0, :128], r1
|
||||
vst1.8 {d0, d1}, [r12, :128], r1
|
||||
subs r4, r4, #4
|
||||
subs r4, r4, #4
|
||||
vst1.8 {d0, d1}, [r0, :128], r1
|
||||
vst1.8 {d0, d1}, [r12, :128], r1
|
||||
bgt 2b
|
||||
|
||||
+470
-532
File diff suppressed because it is too large
Load Diff
+523
-567
File diff suppressed because it is too large
Load Diff
@@ -28,341 +28,119 @@
|
||||
#include "src/arm/asm.S"
|
||||
#include "util.S"
|
||||
|
||||
#define SUM_STRIDE (384+16)
|
||||
|
||||
// void dav1d_sgr_box3_v_neon(int32_t *sumsq, int16_t *sum,
|
||||
// const int w, const int h,
|
||||
// const enum LrEdgeFlags edges);
|
||||
function sgr_box3_v_neon, export=1
|
||||
// void dav1d_sgr_box3_row_v_neon(int32_t **sumsq, int16_t **sum,
|
||||
// int32_t *sumsq_out, int16_t *sum_out,
|
||||
// const int w);
|
||||
function sgr_box3_row_v_neon, export=1
|
||||
push {r4-r9,lr}
|
||||
ldr r4, [sp, #28]
|
||||
add r12, r3, #2 // Number of output rows to move back
|
||||
mov lr, r3 // Number of input rows to move back
|
||||
add r2, r2, #2 // Actual summed width
|
||||
mov r7, #(4*SUM_STRIDE) // sumsq stride
|
||||
mov r8, #(2*SUM_STRIDE) // sum stride
|
||||
sub r0, r0, #(4*SUM_STRIDE) // sumsq -= stride
|
||||
sub r1, r1, #(2*SUM_STRIDE) // sum -= stride
|
||||
|
||||
tst r4, #4 // LR_HAVE_TOP
|
||||
beq 0f
|
||||
// If have top, read from row -2.
|
||||
sub r5, r0, #(4*SUM_STRIDE)
|
||||
sub r6, r1, #(2*SUM_STRIDE)
|
||||
add lr, lr, #2
|
||||
b 1f
|
||||
0:
|
||||
// !LR_HAVE_TOP
|
||||
// If we don't have top, read from row 0 even if
|
||||
// we start writing to row -1.
|
||||
add r5, r0, #(4*SUM_STRIDE)
|
||||
add r6, r1, #(2*SUM_STRIDE)
|
||||
1:
|
||||
|
||||
tst r4, #8 // LR_HAVE_BOTTOM
|
||||
beq 1f
|
||||
// LR_HAVE_BOTTOM
|
||||
add r3, r3, #2 // Sum all h+2 lines with the main loop
|
||||
add lr, lr, #2
|
||||
1:
|
||||
mov r9, r3 // Backup of h for next loops
|
||||
ldrd r6, r7, [r0]
|
||||
ldr r0, [r0, #8]
|
||||
add r4, r4, #2
|
||||
ldrd r8, r9, [r1]
|
||||
ldr r1, [r1, #8]
|
||||
|
||||
1:
|
||||
// Start of horizontal loop; start one vertical filter slice.
|
||||
// Start loading rows into q8-q13 and q0-q2 taking top
|
||||
// padding into consideration.
|
||||
tst r4, #4 // LR_HAVE_TOP
|
||||
vld1.32 {q8, q9}, [r5, :128], r7
|
||||
vld1.16 {q0}, [r6, :128], r8
|
||||
beq 2f
|
||||
// LR_HAVE_TOP
|
||||
vld1.32 {q10, q11}, [r5, :128], r7
|
||||
vld1.16 {q1}, [r6, :128], r8
|
||||
vld1.32 {q12, q13}, [r5, :128], r7
|
||||
vld1.16 {q2}, [r6, :128], r8
|
||||
b 3f
|
||||
2: // !LR_HAVE_TOP
|
||||
vmov q10, q8
|
||||
vmov q11, q9
|
||||
vmov q1, q0
|
||||
vmov q12, q8
|
||||
vmov q13, q9
|
||||
vmov q2, q0
|
||||
vld1.32 {q8, q9}, [r6]!
|
||||
vld1.32 {q10, q11}, [r7]!
|
||||
vld1.16 {q14}, [r8]!
|
||||
vld1.16 {q15}, [r9]!
|
||||
subs r4, r4, #8
|
||||
|
||||
3:
|
||||
subs r3, r3, #1
|
||||
.macro add3
|
||||
vadd.i32 q8, q8, q10
|
||||
vadd.i32 q9, q9, q11
|
||||
|
||||
vld1.32 {q12, q13}, [r0]!
|
||||
|
||||
vadd.i16 q14, q14, q15
|
||||
|
||||
vld1.16 {q15}, [r1]!
|
||||
vadd.i32 q8, q8, q12
|
||||
vadd.i32 q9, q9, q13
|
||||
vadd.i16 q14, q14, q15
|
||||
|
||||
vst1.32 {q8, q9}, [r2]!
|
||||
vst1.16 {q14}, [r3]!
|
||||
|
||||
bgt 1b
|
||||
pop {r4-r9,pc}
|
||||
endfunc
|
||||
|
||||
// void dav1d_sgr_box5_row_v_neon(int32_t **sumsq, int16_t **sum,
|
||||
// int32_t *sumsq_out, int16_t *sum_out,
|
||||
// const int w);
|
||||
function sgr_box5_row_v_neon, export=1
|
||||
push {r4-r11,lr}
|
||||
ldr lr, [sp, #36]
|
||||
|
||||
ldrd r4, r5, [r0]
|
||||
ldrd r6, r7, [r0, #8]
|
||||
ldr r0, [r0, #16]
|
||||
add lr, lr, #2
|
||||
ldrd r8, r9, [r1]
|
||||
ldrd r10, r11, [r1, #8]
|
||||
ldr r1, [r1, #16]
|
||||
|
||||
1:
|
||||
vld1.32 {q8, q9}, [r4]!
|
||||
vld1.32 {q10, q11}, [r5]!
|
||||
vld1.32 {q12, q13}, [r6]!
|
||||
vld1.32 {q14, q15}, [r7]!
|
||||
vld1.16 {q0}, [r8]!
|
||||
vld1.16 {q1}, [r9]!
|
||||
vld1.16 {q2}, [r10]!
|
||||
vld1.16 {q3}, [r11]!
|
||||
subs lr, lr, #8
|
||||
|
||||
vadd.i32 q8, q8, q10
|
||||
vadd.i32 q9, q9, q11
|
||||
vadd.i32 q12, q12, q14
|
||||
vadd.i32 q13, q13, q15
|
||||
|
||||
vld1.32 {q14, q15}, [r0]!
|
||||
|
||||
vadd.i16 q0, q0, q1
|
||||
vadd.i16 q2, q2, q3
|
||||
|
||||
vld1.16 {q3}, [r1]!
|
||||
vadd.i32 q8, q8, q12
|
||||
vadd.i32 q9, q9, q13
|
||||
vadd.i16 q0, q0, q2
|
||||
vst1.32 {q8, q9}, [r0, :128], r7
|
||||
vst1.16 {q0}, [r1, :128], r8
|
||||
.endm
|
||||
add3
|
||||
vmov q8, q10
|
||||
vmov q9, q11
|
||||
vmov q0, q1
|
||||
vmov q10, q12
|
||||
vmov q11, q13
|
||||
vmov q1, q2
|
||||
ble 4f
|
||||
vld1.32 {q12, q13}, [r5, :128], r7
|
||||
vld1.16 {q2}, [r6, :128], r8
|
||||
b 3b
|
||||
|
||||
4:
|
||||
tst r4, #8 // LR_HAVE_BOTTOM
|
||||
bne 5f
|
||||
// !LR_HAVE_BOTTOM
|
||||
// Produce two more rows, extending the already loaded rows.
|
||||
add3
|
||||
vmov q8, q10
|
||||
vmov q9, q11
|
||||
vmov q0, q1
|
||||
add3
|
||||
|
||||
5: // End of one vertical slice.
|
||||
subs r2, r2, #8
|
||||
ble 0f
|
||||
// Move pointers back up to the top and loop horizontally.
|
||||
// Input pointers
|
||||
mls r5, r7, lr, r5
|
||||
mls r6, r8, lr, r6
|
||||
// Output pointers
|
||||
mls r0, r7, r12, r0
|
||||
mls r1, r8, r12, r1
|
||||
add r0, r0, #32
|
||||
add r1, r1, #16
|
||||
add r5, r5, #32
|
||||
add r6, r6, #16
|
||||
mov r3, r9
|
||||
b 1b
|
||||
|
||||
0:
|
||||
pop {r4-r9,pc}
|
||||
.purgem add3
|
||||
endfunc
|
||||
|
||||
// void dav1d_sgr_box5_v_neon(int32_t *sumsq, int16_t *sum,
|
||||
// const int w, const int h,
|
||||
// const enum LrEdgeFlags edges);
|
||||
function sgr_box5_v_neon, export=1
|
||||
push {r4-r9,lr}
|
||||
vpush {q5-q7}
|
||||
ldr r4, [sp, #76]
|
||||
add r12, r3, #2 // Number of output rows to move back
|
||||
mov lr, r3 // Number of input rows to move back
|
||||
add r2, r2, #8 // Actual summed width
|
||||
mov r7, #(4*SUM_STRIDE) // sumsq stride
|
||||
mov r8, #(2*SUM_STRIDE) // sum stride
|
||||
sub r0, r0, #(4*SUM_STRIDE) // sumsq -= stride
|
||||
sub r1, r1, #(2*SUM_STRIDE) // sum -= stride
|
||||
|
||||
tst r4, #4 // LR_HAVE_TOP
|
||||
beq 0f
|
||||
// If have top, read from row -2.
|
||||
sub r5, r0, #(4*SUM_STRIDE)
|
||||
sub r6, r1, #(2*SUM_STRIDE)
|
||||
add lr, lr, #2
|
||||
b 1f
|
||||
0:
|
||||
// !LR_HAVE_TOP
|
||||
// If we don't have top, read from row 0 even if
|
||||
// we start writing to row -1.
|
||||
add r5, r0, #(4*SUM_STRIDE)
|
||||
add r6, r1, #(2*SUM_STRIDE)
|
||||
1:
|
||||
|
||||
tst r4, #8 // LR_HAVE_BOTTOM
|
||||
beq 0f
|
||||
// LR_HAVE_BOTTOM
|
||||
add r3, r3, #2 // Handle h+2 lines with the main loop
|
||||
add lr, lr, #2
|
||||
b 1f
|
||||
0:
|
||||
// !LR_HAVE_BOTTOM
|
||||
sub r3, r3, #1 // Handle h-1 lines with the main loop
|
||||
1:
|
||||
mov r9, r3 // Backup of h for next loops
|
||||
|
||||
1:
|
||||
// Start of horizontal loop; start one vertical filter slice.
|
||||
// Start loading rows into q6-q15 and q0-q3,q5 taking top
|
||||
// padding into consideration.
|
||||
tst r4, #4 // LR_HAVE_TOP
|
||||
vld1.32 {q6, q7}, [r5, :128], r7
|
||||
vld1.16 {q0}, [r6, :128], r8
|
||||
beq 2f
|
||||
// LR_HAVE_TOP
|
||||
vld1.32 {q10, q11}, [r5, :128], r7
|
||||
vld1.16 {q2}, [r6, :128], r8
|
||||
vmov q8, q6
|
||||
vmov q9, q7
|
||||
vmov q1, q0
|
||||
vld1.32 {q12, q13}, [r5, :128], r7
|
||||
vld1.16 {q3}, [r6, :128], r8
|
||||
b 3f
|
||||
2: // !LR_HAVE_TOP
|
||||
vmov q8, q6
|
||||
vmov q9, q7
|
||||
vmov q1, q0
|
||||
vmov q10, q6
|
||||
vmov q11, q7
|
||||
vmov q2, q0
|
||||
vmov q12, q6
|
||||
vmov q13, q7
|
||||
vmov q3, q0
|
||||
|
||||
3:
|
||||
cmp r3, #0
|
||||
beq 4f
|
||||
vld1.32 {q14, q15}, [r5, :128], r7
|
||||
vld1.16 {q5}, [r6, :128], r8
|
||||
|
||||
3:
|
||||
// Start of vertical loop
|
||||
subs r3, r3, #2
|
||||
.macro add5
|
||||
vadd.i32 q6, q6, q8
|
||||
vadd.i32 q7, q7, q9
|
||||
vadd.i16 q0, q0, q1
|
||||
vadd.i32 q6, q6, q10
|
||||
vadd.i32 q7, q7, q11
|
||||
vadd.i16 q0, q0, q2
|
||||
vadd.i32 q6, q6, q12
|
||||
vadd.i32 q7, q7, q13
|
||||
vadd.i32 q8, q8, q14
|
||||
vadd.i32 q9, q9, q15
|
||||
vadd.i16 q0, q0, q3
|
||||
vadd.i32 q6, q6, q14
|
||||
vadd.i32 q7, q7, q15
|
||||
vadd.i16 q0, q0, q5
|
||||
vst1.32 {q6, q7}, [r0, :128], r7
|
||||
vst1.16 {q0}, [r1, :128], r8
|
||||
.endm
|
||||
add5
|
||||
.macro shift2
|
||||
vmov q6, q10
|
||||
vmov q7, q11
|
||||
vmov q0, q2
|
||||
vmov q8, q12
|
||||
vmov q9, q13
|
||||
vmov q1, q3
|
||||
vmov q10, q14
|
||||
vmov q11, q15
|
||||
vmov q2, q5
|
||||
.endm
|
||||
shift2
|
||||
add r0, r0, r7
|
||||
add r1, r1, r8
|
||||
ble 5f
|
||||
vld1.32 {q12, q13}, [r5, :128], r7
|
||||
vld1.16 {q3}, [r6, :128], r8
|
||||
vld1.32 {q14, q15}, [r5, :128], r7
|
||||
vld1.16 {q5}, [r6, :128], r8
|
||||
b 3b
|
||||
|
||||
4:
|
||||
// h == 1, !LR_HAVE_BOTTOM.
|
||||
// Pad the last row with the only content row, and add.
|
||||
vmov q14, q12
|
||||
vmov q15, q13
|
||||
vmov q5, q3
|
||||
add5
|
||||
shift2
|
||||
add r0, r0, r7
|
||||
add r1, r1, r8
|
||||
add5
|
||||
b 6f
|
||||
vst1.32 {q8, q9}, [r2]!
|
||||
vst1.16 {q0}, [r3]!
|
||||
|
||||
5:
|
||||
tst r4, #8 // LR_HAVE_BOTTOM
|
||||
bne 6f
|
||||
// !LR_HAVE_BOTTOM
|
||||
cmp r3, #0
|
||||
bne 5f
|
||||
// The intended three edge rows left; output the one at h-2 and
|
||||
// the past edge one at h.
|
||||
vld1.32 {q12, q13}, [r5, :128], r7
|
||||
vld1.16 {q3}, [r6, :128], r8
|
||||
// Pad the past-edge row from the last content row.
|
||||
vmov q14, q12
|
||||
vmov q15, q13
|
||||
vmov q5, q3
|
||||
add5
|
||||
shift2
|
||||
add r0, r0, r7
|
||||
add r1, r1, r8
|
||||
// The last two rows are already padded properly here.
|
||||
add5
|
||||
b 6f
|
||||
|
||||
5:
|
||||
// r3 == -1, two rows left, output one.
|
||||
// Pad the last two rows from the mid one.
|
||||
vmov q12, q10
|
||||
vmov q13, q11
|
||||
vmov q3, q2
|
||||
vmov q14, q10
|
||||
vmov q15, q11
|
||||
vmov q5, q2
|
||||
add5
|
||||
add r0, r0, r7
|
||||
add r1, r1, r8
|
||||
b 6f
|
||||
|
||||
6: // End of one vertical slice.
|
||||
subs r2, r2, #8
|
||||
ble 0f
|
||||
// Move pointers back up to the top and loop horizontally.
|
||||
// Input pointers
|
||||
mls r5, r7, lr, r5
|
||||
mls r6, r8, lr, r6
|
||||
// Output pointers
|
||||
mls r0, r7, r12, r0
|
||||
mls r1, r8, r12, r1
|
||||
add r0, r0, #32
|
||||
add r1, r1, #16
|
||||
add r5, r5, #32
|
||||
add r6, r6, #16
|
||||
mov r3, r9
|
||||
b 1b
|
||||
|
||||
0:
|
||||
vpop {q5-q7}
|
||||
pop {r4-r9,pc}
|
||||
.purgem add5
|
||||
bgt 1b
|
||||
pop {r4-r11,pc}
|
||||
endfunc
|
||||
|
||||
// void dav1d_sgr_calc_ab1_neon(int32_t *a, int16_t *b,
|
||||
// const int w, const int h, const int strength,
|
||||
// const int bitdepth_max);
|
||||
// void dav1d_sgr_calc_ab2_neon(int32_t *a, int16_t *b,
|
||||
// const int w, const int h, const int strength,
|
||||
// const int bitdepth_max);
|
||||
function sgr_calc_ab1_neon, export=1
|
||||
// void dav1d_sgr_calc_row_ab1_neon(int32_t *a, int16_t *b,
|
||||
// const int w, const int strength,
|
||||
// const int bitdepth_max);
|
||||
// void dav1d_sgr_calc_row_ab2_neon(int32_t *a, int16_t *b,
|
||||
// const int w, const int strength,
|
||||
// const int bitdepth_max);
|
||||
function sgr_calc_row_ab1_neon, export=1
|
||||
push {r4-r7,lr}
|
||||
vpush {q4-q7}
|
||||
ldrd r4, r5, [sp, #84]
|
||||
add r3, r3, #2 // h += 2
|
||||
clz r6, r5
|
||||
ldr r4, [sp, #84]
|
||||
clz r6, r4
|
||||
vmov.i32 q15, #9 // n
|
||||
movw r5, #455
|
||||
mov lr, #SUM_STRIDE
|
||||
b sgr_calc_ab_neon
|
||||
endfunc
|
||||
|
||||
function sgr_calc_ab2_neon, export=1
|
||||
function sgr_calc_row_ab2_neon, export=1
|
||||
push {r4-r7,lr}
|
||||
vpush {q4-q7}
|
||||
ldrd r4, r5, [sp, #84]
|
||||
add r3, r3, #3 // h += 3
|
||||
clz r6, r5
|
||||
asr r3, r3, #1 // h /= 2
|
||||
ldr r4, [sp, #84]
|
||||
clz r6, r4
|
||||
vmov.i32 q15, #25 // n
|
||||
mov r5, #164
|
||||
mov lr, #(2*SUM_STRIDE)
|
||||
endfunc
|
||||
|
||||
function sgr_calc_ab_neon
|
||||
@@ -379,20 +157,14 @@ function sgr_calc_ab_neon
|
||||
vmov.i8 d14, #254 // idx of last 1
|
||||
vmov.i8 d15, #32 // elements consumed in first vtbl
|
||||
add r2, r2, #2 // w += 2
|
||||
add r12, r2, #7
|
||||
bic r12, r12, #7 // aligned w
|
||||
sub r12, lr, r12 // increment between rows
|
||||
vdup.32 q12, r4
|
||||
sub r0, r0, #(4*(SUM_STRIDE))
|
||||
sub r1, r1, #(2*(SUM_STRIDE))
|
||||
mov r4, r2 // backup of w
|
||||
vdup.32 q12, r3
|
||||
vsub.i8 q8, q8, q11
|
||||
vsub.i8 q9, q9, q11
|
||||
vsub.i8 q10, q10, q11
|
||||
vdup.32 q13, r7 // -2*bitdepth_min_8
|
||||
1:
|
||||
vld1.32 {q0, q1}, [r0, :128] // a
|
||||
vld1.16 {q2}, [r1, :128] // b
|
||||
vdup.32 q13, r7 // -2*bitdepth_min_8
|
||||
vdup.16 q14, r6 // -bitdepth_min_8
|
||||
subs r2, r2, #8
|
||||
vrshl.s32 q0, q0, q13
|
||||
@@ -426,7 +198,6 @@ function sgr_calc_ab_neon
|
||||
vadd.i8 d1, d1, d2
|
||||
vmovl.u8 q0, d1 // x
|
||||
|
||||
vmov.i16 q13, #256
|
||||
vdup.32 q14, r5 // one_by_x
|
||||
|
||||
vmull.u16 q1, d0, d4 // x * BB[i]
|
||||
@@ -435,19 +206,11 @@ function sgr_calc_ab_neon
|
||||
vmul.i32 q2, q2, q14 // x * BB[i] * sgr_one_by_x
|
||||
vrshr.s32 q1, q1, #12 // AA[i]
|
||||
vrshr.s32 q2, q2, #12 // AA[i]
|
||||
vsub.i16 q0, q13, q0 // 256 - x
|
||||
|
||||
vst1.32 {q1, q2}, [r0, :128]!
|
||||
vst1.16 {q0}, [r1, :128]!
|
||||
bgt 1b
|
||||
|
||||
subs r3, r3, #1
|
||||
ble 0f
|
||||
add r0, r0, r12, lsl #2
|
||||
add r1, r1, r12, lsl #1
|
||||
mov r2, r4
|
||||
b 1b
|
||||
0:
|
||||
vpop {q4-q7}
|
||||
pop {r4-r7,pc}
|
||||
endfunc
|
||||
|
||||
+133
-323
@@ -30,44 +30,30 @@
|
||||
#define FILTER_OUT_STRIDE 384
|
||||
|
||||
.macro sgr_funcs bpc
|
||||
// void dav1d_sgr_finish_filter1_Xbpc_neon(int16_t *tmp,
|
||||
// const pixel *src, const ptrdiff_t stride,
|
||||
// const int32_t *a, const int16_t *b,
|
||||
// const int w, const int h);
|
||||
function sgr_finish_filter1_\bpc\()bpc_neon, export=1
|
||||
// void dav1d_sgr_finish_filter_row1_Xbpc_neon(int16_t *tmp,
|
||||
// const pixel *src,
|
||||
// const int32_t **a, const int16_t **b,
|
||||
// const int w);
|
||||
function sgr_finish_filter_row1_\bpc\()bpc_neon, export=1
|
||||
push {r4-r11,lr}
|
||||
vpush {q4-q7}
|
||||
ldrd r4, r5, [sp, #100]
|
||||
ldr r6, [sp, #108]
|
||||
sub r7, r3, #(4*SUM_STRIDE)
|
||||
add r8, r3, #(4*SUM_STRIDE)
|
||||
sub r9, r4, #(2*SUM_STRIDE)
|
||||
add r10, r4, #(2*SUM_STRIDE)
|
||||
mov r11, #SUM_STRIDE
|
||||
mov r12, #FILTER_OUT_STRIDE
|
||||
add lr, r5, #3
|
||||
bic lr, lr, #3 // Aligned width
|
||||
.if \bpc == 8
|
||||
sub r2, r2, lr
|
||||
.else
|
||||
sub r2, r2, lr, lsl #1
|
||||
.endif
|
||||
sub r12, r12, lr
|
||||
sub r11, r11, lr
|
||||
sub r11, r11, #4 // We read 4 extra elements from both a and b
|
||||
mov lr, r5
|
||||
ldr r4, [sp, #100]
|
||||
ldrd r6, r7, [r2]
|
||||
ldr r2, [r2, #8]
|
||||
ldrd r8, r9, [r3]
|
||||
ldr r3, [r3, #8]
|
||||
vmov.i16 q14, #3
|
||||
vmov.i32 q15, #3
|
||||
1:
|
||||
vld1.16 {q0}, [r9, :128]!
|
||||
vld1.16 {q1}, [r4, :128]!
|
||||
vld1.16 {q2}, [r10, :128]!
|
||||
vld1.32 {q8, q9}, [r7, :128]!
|
||||
vld1.32 {q10, q11}, [r3, :128]!
|
||||
vld1.32 {q12, q13}, [r8, :128]!
|
||||
vld1.16 {q0}, [r8, :128]!
|
||||
vld1.16 {q1}, [r9, :128]!
|
||||
vld1.16 {q2}, [r3, :128]!
|
||||
vld1.32 {q8, q9}, [r6, :128]!
|
||||
vld1.32 {q10, q11}, [r7, :128]!
|
||||
vld1.32 {q12, q13}, [r2, :128]!
|
||||
|
||||
2:
|
||||
subs r5, r5, #4
|
||||
subs r4, r4, #4
|
||||
vext.8 d6, d0, d1, #2 // -stride
|
||||
vext.8 d7, d2, d3, #2 // 0
|
||||
vext.8 d8, d4, d5, #2 // +stride
|
||||
@@ -108,7 +94,7 @@ function sgr_finish_filter1_\bpc\()bpc_neon, export=1
|
||||
vmovl.u8 q12, d24 // src
|
||||
.endif
|
||||
vmov d0, d1
|
||||
vmlal.u16 q3, d2, d24 // b + a * src
|
||||
vmlsl.u16 q3, d2, d24 // b - a * src
|
||||
vmov d2, d3
|
||||
vrshrn.i32 d6, q3, #9
|
||||
vmov d4, d5
|
||||
@@ -118,67 +104,42 @@ function sgr_finish_filter1_\bpc\()bpc_neon, export=1
|
||||
vmov q8, q9
|
||||
vmov q10, q11
|
||||
vmov q12, q13
|
||||
vld1.16 {d1}, [r9, :64]!
|
||||
vld1.16 {d3}, [r4, :64]!
|
||||
vld1.16 {d5}, [r10, :64]!
|
||||
vld1.32 {q9}, [r7, :128]!
|
||||
vld1.32 {q11}, [r3, :128]!
|
||||
vld1.32 {q13}, [r8, :128]!
|
||||
vld1.16 {d1}, [r8, :64]!
|
||||
vld1.16 {d3}, [r9, :64]!
|
||||
vld1.16 {d5}, [r3, :64]!
|
||||
vld1.32 {q9}, [r6, :128]!
|
||||
vld1.32 {q11}, [r7, :128]!
|
||||
vld1.32 {q13}, [r2, :128]!
|
||||
b 2b
|
||||
|
||||
3:
|
||||
subs r6, r6, #1
|
||||
ble 0f
|
||||
mov r5, lr
|
||||
add r0, r0, r12, lsl #1
|
||||
add r1, r1, r2
|
||||
add r3, r3, r11, lsl #2
|
||||
add r7, r7, r11, lsl #2
|
||||
add r8, r8, r11, lsl #2
|
||||
add r4, r4, r11, lsl #1
|
||||
add r9, r9, r11, lsl #1
|
||||
add r10, r10, r11, lsl #1
|
||||
b 1b
|
||||
0:
|
||||
vpop {q4-q7}
|
||||
pop {r4-r11,pc}
|
||||
endfunc
|
||||
|
||||
// void dav1d_sgr_finish_filter2_Xbpc_neon(int16_t *tmp,
|
||||
// const pixel *src, const ptrdiff_t stride,
|
||||
// const int32_t *a, const int16_t *b,
|
||||
// const int w, const int h);
|
||||
function sgr_finish_filter2_\bpc\()bpc_neon, export=1
|
||||
// void dav1d_sgr_finish_filter2_2rows_Xbpc_neon(int16_t *tmp,
|
||||
// const pixel *src, const ptrdiff_t stride,
|
||||
// const int32_t **a, const int16_t **b,
|
||||
// const int w, const int h);
|
||||
function sgr_finish_filter2_2rows_\bpc\()bpc_neon, export=1
|
||||
push {r4-r11,lr}
|
||||
vpush {q4-q7}
|
||||
ldrd r4, r5, [sp, #100]
|
||||
ldr r6, [sp, #108]
|
||||
add r7, r3, #(4*(SUM_STRIDE))
|
||||
sub r3, r3, #(4*(SUM_STRIDE))
|
||||
add r8, r4, #(2*(SUM_STRIDE))
|
||||
sub r4, r4, #(2*(SUM_STRIDE))
|
||||
mov r9, #(2*SUM_STRIDE)
|
||||
mov r10, #FILTER_OUT_STRIDE
|
||||
add r11, r5, #7
|
||||
bic r11, r11, #7 // Aligned width
|
||||
.if \bpc == 8
|
||||
sub r2, r2, r11
|
||||
.else
|
||||
sub r2, r2, r11, lsl #1
|
||||
.endif
|
||||
sub r10, r10, r11
|
||||
sub r9, r9, r11
|
||||
sub r9, r9, #4 // We read 4 extra elements from a
|
||||
sub r12, r9, #4 // We read 8 extra elements from b
|
||||
ldrd r8, r9, [r3]
|
||||
ldrd r10, r11, [r4]
|
||||
mov r7, #2*FILTER_OUT_STRIDE
|
||||
add r2, r1, r2
|
||||
add r7, r7, r0
|
||||
mov lr, r5
|
||||
|
||||
1:
|
||||
vld1.16 {q0, q1}, [r4, :128]!
|
||||
vld1.16 {q2, q3}, [r8, :128]!
|
||||
vld1.32 {q8, q9}, [r3, :128]!
|
||||
vld1.32 {q11, q12}, [r7, :128]!
|
||||
vld1.32 {q10}, [r3, :128]!
|
||||
vld1.32 {q13}, [r7, :128]!
|
||||
vld1.16 {q0, q1}, [r10, :128]!
|
||||
vld1.16 {q2, q3}, [r11, :128]!
|
||||
vld1.32 {q8, q9}, [r8, :128]!
|
||||
vld1.32 {q11, q12}, [r9, :128]!
|
||||
vld1.32 {q10}, [r8, :128]!
|
||||
vld1.32 {q13}, [r9, :128]!
|
||||
|
||||
2:
|
||||
vmov.i16 q14, #5
|
||||
@@ -229,8 +190,8 @@ function sgr_finish_filter2_\bpc\()bpc_neon, export=1
|
||||
.if \bpc == 8
|
||||
vmovl.u8 q2, d4
|
||||
.endif
|
||||
vmlal.u16 q4, d0, d4 // b + a * src
|
||||
vmlal.u16 q5, d1, d5 // b + a * src
|
||||
vmlsl.u16 q4, d0, d4 // b - a * src
|
||||
vmlsl.u16 q5, d1, d5 // b - a * src
|
||||
vmov q0, q1
|
||||
vrshrn.i32 d8, q4, #9
|
||||
vrshrn.i32 d9, q5, #9
|
||||
@@ -240,26 +201,24 @@ function sgr_finish_filter2_\bpc\()bpc_neon, export=1
|
||||
ble 3f
|
||||
vmov q8, q10
|
||||
vmov q11, q13
|
||||
vld1.16 {q1}, [r4, :128]!
|
||||
vld1.16 {q3}, [r8, :128]!
|
||||
vld1.32 {q9, q10}, [r3, :128]!
|
||||
vld1.32 {q12, q13}, [r7, :128]!
|
||||
vld1.16 {q1}, [r10, :128]!
|
||||
vld1.16 {q3}, [r11, :128]!
|
||||
vld1.32 {q9, q10}, [r8, :128]!
|
||||
vld1.32 {q12, q13}, [r9, :128]!
|
||||
b 2b
|
||||
|
||||
3:
|
||||
subs r6, r6, #1
|
||||
ble 0f
|
||||
mov r5, lr
|
||||
add r0, r0, r10, lsl #1
|
||||
add r1, r1, r2
|
||||
add r3, r3, r9, lsl #2
|
||||
add r7, r7, r9, lsl #2
|
||||
add r4, r4, r12, lsl #1
|
||||
add r8, r8, r12, lsl #1
|
||||
ldrd r8, r9, [r3]
|
||||
ldrd r10, r11, [r4]
|
||||
mov r0, r7
|
||||
mov r1, r2
|
||||
|
||||
vld1.32 {q8, q9}, [r3, :128]!
|
||||
vld1.16 {q0, q1}, [r4, :128]!
|
||||
vld1.32 {q10}, [r3, :128]!
|
||||
vld1.32 {q8, q9}, [r9, :128]!
|
||||
vld1.16 {q0, q1}, [r11, :128]!
|
||||
vld1.32 {q10}, [r9, :128]!
|
||||
|
||||
vmov.i16 q12, #5
|
||||
vmov.i16 q13, #6
|
||||
@@ -291,8 +250,8 @@ function sgr_finish_filter2_\bpc\()bpc_neon, export=1
|
||||
vmul.i32 q5, q5, q15 // * 6
|
||||
vmla.i32 q5, q9, q14 // * 5 -> b
|
||||
|
||||
vmlal.u16 q4, d4, d22 // b + a * src
|
||||
vmlal.u16 q5, d5, d23
|
||||
vmlsl.u16 q4, d4, d22 // b - a * src
|
||||
vmlsl.u16 q5, d5, d23
|
||||
vmov q0, q1
|
||||
vrshrn.i32 d8, q4, #8
|
||||
vrshrn.i32 d9, q5, #8
|
||||
@@ -300,301 +259,152 @@ function sgr_finish_filter2_\bpc\()bpc_neon, export=1
|
||||
vst1.16 {q4}, [r0, :128]!
|
||||
|
||||
ble 5f
|
||||
vld1.16 {q1}, [r4, :128]!
|
||||
vld1.32 {q9, q10}, [r3, :128]!
|
||||
vld1.16 {q1}, [r11, :128]!
|
||||
vld1.32 {q9, q10}, [r9, :128]!
|
||||
b 4b
|
||||
|
||||
5:
|
||||
subs r6, r6, #1
|
||||
ble 0f
|
||||
mov r5, lr
|
||||
sub r3, r3, r11, lsl #2 // Rewind r3/r4 to where they started
|
||||
sub r4, r4, r11, lsl #1
|
||||
add r0, r0, r10, lsl #1
|
||||
add r1, r1, r2
|
||||
sub r3, r3, #16
|
||||
sub r4, r4, #16
|
||||
b 1b
|
||||
0:
|
||||
vpop {q4-q7}
|
||||
pop {r4-r11,pc}
|
||||
endfunc
|
||||
|
||||
// void dav1d_sgr_weighted1_Xbpc_neon(pixel *dst, const ptrdiff_t dst_stride,
|
||||
// const pixel *src, const ptrdiff_t src_stride,
|
||||
// const int16_t *t1, const int w, const int h,
|
||||
// const int wt, const int bitdepth_max);
|
||||
function sgr_weighted1_\bpc\()bpc_neon, export=1
|
||||
push {r4-r9,lr}
|
||||
ldrd r4, r5, [sp, #28]
|
||||
ldrd r6, r7, [sp, #36]
|
||||
// void dav1d_sgr_weighted_row1_Xbpc_neon(pixel *dst,
|
||||
// const int16_t *t1, const int w,
|
||||
// const int w1, const int bitdepth_max);
|
||||
function sgr_weighted_row1_\bpc\()bpc_neon, export=1
|
||||
push {lr}
|
||||
.if \bpc == 16
|
||||
ldr r8, [sp, #44]
|
||||
ldr lr, [sp, #4]
|
||||
.endif
|
||||
vdup.16 d31, r7
|
||||
cmp r6, #2
|
||||
vdup.16 d31, r3
|
||||
.if \bpc == 16
|
||||
vdup.16 q14, r8
|
||||
vmov.i16 q13, #0
|
||||
vdup.16 q14, lr
|
||||
.endif
|
||||
add r9, r0, r1
|
||||
add r12, r2, r3
|
||||
add lr, r4, #2*FILTER_OUT_STRIDE
|
||||
mov r7, #(4*FILTER_OUT_STRIDE)
|
||||
lsl r1, r1, #1
|
||||
lsl r3, r3, #1
|
||||
add r8, r5, #7
|
||||
bic r8, r8, #7 // Aligned width
|
||||
.if \bpc == 8
|
||||
sub r1, r1, r8
|
||||
sub r3, r3, r8
|
||||
.else
|
||||
sub r1, r1, r8, lsl #1
|
||||
sub r3, r3, r8, lsl #1
|
||||
.endif
|
||||
sub r7, r7, r8, lsl #1
|
||||
mov r8, r5
|
||||
blt 2f
|
||||
|
||||
1:
|
||||
.if \bpc == 8
|
||||
vld1.8 {d0}, [r2, :64]!
|
||||
vld1.8 {d16}, [r12, :64]!
|
||||
vld1.8 {d0}, [r0, :64]
|
||||
.else
|
||||
vld1.16 {q0}, [r2, :128]!
|
||||
vld1.16 {q8}, [r12, :128]!
|
||||
vld1.16 {q0}, [r0, :128]
|
||||
.endif
|
||||
vld1.16 {q1}, [r4, :128]!
|
||||
vld1.16 {q9}, [lr, :128]!
|
||||
subs r5, r5, #8
|
||||
.if \bpc == 8
|
||||
vshll.u8 q0, d0, #4 // u
|
||||
vshll.u8 q8, d16, #4 // u
|
||||
.else
|
||||
vshl.i16 q0, q0, #4 // u
|
||||
vshl.i16 q8, q8, #4 // u
|
||||
.endif
|
||||
vsub.i16 q1, q1, q0 // t1 - u
|
||||
vsub.i16 q9, q9, q8 // t1 - u
|
||||
vshll.u16 q2, d0, #7 // u << 7
|
||||
vshll.u16 q3, d1, #7 // u << 7
|
||||
vshll.u16 q10, d16, #7 // u << 7
|
||||
vshll.u16 q11, d17, #7 // u << 7
|
||||
vmlal.s16 q2, d2, d31 // v
|
||||
vmlal.s16 q3, d3, d31 // v
|
||||
vmlal.s16 q10, d18, d31 // v
|
||||
vmlal.s16 q11, d19, d31 // v
|
||||
.if \bpc == 8
|
||||
vld1.16 {q1}, [r1, :128]!
|
||||
subs r2, r2, #8
|
||||
vmull.s16 q2, d2, d31 // v
|
||||
vmull.s16 q3, d3, d31 // v
|
||||
vrshrn.i32 d4, q2, #11
|
||||
vrshrn.i32 d5, q3, #11
|
||||
vrshrn.i32 d20, q10, #11
|
||||
vrshrn.i32 d21, q11, #11
|
||||
vqmovun.s16 d4, q2
|
||||
vqmovun.s16 d20, q10
|
||||
vst1.8 {d4}, [r0, :64]!
|
||||
vst1.8 {d20}, [r9, :64]!
|
||||
.else
|
||||
vqrshrun.s32 d4, q2, #11
|
||||
vqrshrun.s32 d5, q3, #11
|
||||
vqrshrun.s32 d20, q10, #11
|
||||
vqrshrun.s32 d21, q11, #11
|
||||
vmin.u16 q2, q2, q14
|
||||
vmin.u16 q10, q10, q14
|
||||
vst1.16 {q2}, [r0, :128]!
|
||||
vst1.16 {q10}, [r9, :128]!
|
||||
.endif
|
||||
bgt 1b
|
||||
|
||||
sub r6, r6, #2
|
||||
cmp r6, #1
|
||||
blt 0f
|
||||
mov r5, r8
|
||||
add r0, r0, r1
|
||||
add r9, r9, r1
|
||||
add r2, r2, r3
|
||||
add r12, r12, r3
|
||||
add r4, r4, r7
|
||||
add lr, lr, r7
|
||||
beq 2f
|
||||
b 1b
|
||||
|
||||
2:
|
||||
.if \bpc == 8
|
||||
vld1.8 {d0}, [r2, :64]!
|
||||
.else
|
||||
vld1.16 {q0}, [r2, :128]!
|
||||
.endif
|
||||
vld1.16 {q1}, [r4, :128]!
|
||||
subs r5, r5, #8
|
||||
.if \bpc == 8
|
||||
vshll.u8 q0, d0, #4 // u
|
||||
.else
|
||||
vshl.i16 q0, q0, #4 // u
|
||||
.endif
|
||||
vsub.i16 q1, q1, q0 // t1 - u
|
||||
vshll.u16 q2, d0, #7 // u << 7
|
||||
vshll.u16 q3, d1, #7 // u << 7
|
||||
vmlal.s16 q2, d2, d31 // v
|
||||
vmlal.s16 q3, d3, d31 // v
|
||||
.if \bpc == 8
|
||||
vrshrn.i32 d4, q2, #11
|
||||
vrshrn.i32 d5, q3, #11
|
||||
vaddw.u8 q2, q2, d0
|
||||
vqmovun.s16 d2, q2
|
||||
vst1.8 {d2}, [r0, :64]!
|
||||
.else
|
||||
vqrshrun.s32 d4, q2, #11
|
||||
vqrshrun.s32 d5, q3, #11
|
||||
vadd.i16 q2, q2, q0
|
||||
vmax.s16 q2, q2, q13
|
||||
vmin.u16 q2, q2, q14
|
||||
vst1.16 {q2}, [r0, :128]!
|
||||
.endif
|
||||
bgt 2b
|
||||
bgt 1b
|
||||
0:
|
||||
pop {r4-r9,pc}
|
||||
pop {pc}
|
||||
endfunc
|
||||
|
||||
// void dav1d_sgr_weighted2_Xbpc_neon(pixel *dst, const ptrdiff_t stride,
|
||||
// const pixel *src, const ptrdiff_t src_stride,
|
||||
// const int16_t *t1, const int16_t *t2,
|
||||
// const int w, const int h,
|
||||
// const int16_t wt[2], const int bitdepth_max);
|
||||
function sgr_weighted2_\bpc\()bpc_neon, export=1
|
||||
push {r4-r11,lr}
|
||||
ldrd r4, r5, [sp, #36]
|
||||
ldrd r6, r7, [sp, #44]
|
||||
push {r4-r8,lr}
|
||||
ldrd r4, r5, [sp, #24]
|
||||
.if \bpc == 8
|
||||
ldr r8, [sp, #52]
|
||||
ldr r6, [sp, #32]
|
||||
.else
|
||||
ldrd r8, r9, [sp, #52]
|
||||
ldrd r6, r7, [sp, #32]
|
||||
.endif
|
||||
cmp r7, #2
|
||||
add r10, r0, r1
|
||||
add r11, r2, r3
|
||||
add r12, r4, #2*FILTER_OUT_STRIDE
|
||||
add lr, r5, #2*FILTER_OUT_STRIDE
|
||||
vld2.16 {d30[], d31[]}, [r8] // wt[0], wt[1]
|
||||
cmp r5, #2
|
||||
add r8, r0, r1
|
||||
add r12, r2, #2*FILTER_OUT_STRIDE
|
||||
add lr, r3, #2*FILTER_OUT_STRIDE
|
||||
vld2.16 {d30[], d31[]}, [r6] // wt[0], wt[1]
|
||||
.if \bpc == 16
|
||||
vdup.16 q14, r9
|
||||
vdup.16 q14, r7
|
||||
.endif
|
||||
mov r8, #4*FILTER_OUT_STRIDE
|
||||
lsl r1, r1, #1
|
||||
lsl r3, r3, #1
|
||||
add r9, r6, #7
|
||||
bic r9, r9, #7 // Aligned width
|
||||
.if \bpc == 8
|
||||
sub r1, r1, r9
|
||||
sub r3, r3, r9
|
||||
.else
|
||||
sub r1, r1, r9, lsl #1
|
||||
sub r3, r3, r9, lsl #1
|
||||
.endif
|
||||
sub r8, r8, r9, lsl #1
|
||||
mov r9, r6
|
||||
blt 2f
|
||||
1:
|
||||
.if \bpc == 8
|
||||
vld1.8 {d0}, [r2, :64]!
|
||||
vld1.8 {d16}, [r11, :64]!
|
||||
vld1.8 {d0}, [r0, :64]
|
||||
vld1.8 {d16}, [r8, :64]
|
||||
.else
|
||||
vld1.16 {q0}, [r2, :128]!
|
||||
vld1.16 {q8}, [r11, :128]!
|
||||
vld1.16 {q0}, [r0, :128]
|
||||
vld1.16 {q8}, [r8, :128]
|
||||
.endif
|
||||
vld1.16 {q1}, [r4, :128]!
|
||||
vld1.16 {q1}, [r2, :128]!
|
||||
vld1.16 {q9}, [r12, :128]!
|
||||
vld1.16 {q2}, [r5, :128]!
|
||||
vld1.16 {q2}, [r3, :128]!
|
||||
vld1.16 {q10}, [lr, :128]!
|
||||
subs r6, r6, #8
|
||||
.if \bpc == 8
|
||||
vshll.u8 q0, d0, #4 // u
|
||||
vshll.u8 q8, d16, #4 // u
|
||||
.else
|
||||
vshl.i16 q0, q0, #4 // u
|
||||
vshl.i16 q8, q8, #4 // u
|
||||
.endif
|
||||
vsub.i16 q1, q1, q0 // t1 - u
|
||||
vsub.i16 q2, q2, q0 // t2 - u
|
||||
vsub.i16 q9, q9, q8 // t1 - u
|
||||
vsub.i16 q10, q10, q8 // t2 - u
|
||||
vshll.u16 q3, d0, #7 // u << 7
|
||||
vshll.u16 q0, d1, #7 // u << 7
|
||||
vshll.u16 q11, d16, #7 // u << 7
|
||||
vshll.u16 q8, d17, #7 // u << 7
|
||||
vmlal.s16 q3, d2, d30 // wt[0] * (t1 - u)
|
||||
vmlal.s16 q3, d4, d31 // wt[1] * (t2 - u)
|
||||
vmlal.s16 q0, d3, d30 // wt[0] * (t1 - u)
|
||||
vmlal.s16 q0, d5, d31 // wt[1] * (t2 - u)
|
||||
vmlal.s16 q11, d18, d30 // wt[0] * (t1 - u)
|
||||
vmlal.s16 q11, d20, d31 // wt[1] * (t2 - u)
|
||||
vmlal.s16 q8, d19, d30 // wt[0] * (t1 - u)
|
||||
vmlal.s16 q8, d21, d31 // wt[1] * (t2 - u)
|
||||
.if \bpc == 8
|
||||
subs r4, r4, #8
|
||||
vmull.s16 q3, d2, d30 // wt[0] * t1
|
||||
vmlal.s16 q3, d4, d31 // wt[1] * t2
|
||||
vmull.s16 q12, d3, d30 // wt[0] * t1
|
||||
vmlal.s16 q12, d5, d31 // wt[1] * t2
|
||||
vmull.s16 q11, d18, d30 // wt[0] * t1
|
||||
vmlal.s16 q11, d20, d31 // wt[1] * t2
|
||||
vmull.s16 q13, d19, d30 // wt[0] * t1
|
||||
vmlal.s16 q13, d21, d31 // wt[1] * t2
|
||||
vrshrn.i32 d6, q3, #11
|
||||
vrshrn.i32 d7, q0, #11
|
||||
vrshrn.i32 d7, q12, #11
|
||||
vrshrn.i32 d22, q11, #11
|
||||
vrshrn.i32 d23, q8, #11
|
||||
vrshrn.i32 d23, q13, #11
|
||||
.if \bpc == 8
|
||||
vaddw.u8 q3, q3, d0
|
||||
vaddw.u8 q11, q11, d16
|
||||
vqmovun.s16 d6, q3
|
||||
vqmovun.s16 d22, q11
|
||||
vst1.8 {d6}, [r0, :64]!
|
||||
vst1.8 {d22}, [r10, :64]!
|
||||
vst1.8 {d22}, [r8, :64]!
|
||||
.else
|
||||
vqrshrun.s32 d6, q3, #11
|
||||
vqrshrun.s32 d7, q0, #11
|
||||
vqrshrun.s32 d22, q11, #11
|
||||
vqrshrun.s32 d23, q8, #11
|
||||
vmov.i16 q13, #0
|
||||
vadd.i16 q3, q3, q0
|
||||
vadd.i16 q11, q11, q8
|
||||
vmax.s16 q3, q3, q13
|
||||
vmax.s16 q11, q11, q13
|
||||
vmin.u16 q3, q3, q14
|
||||
vmin.u16 q11, q11, q14
|
||||
vst1.16 {q3}, [r0, :128]!
|
||||
vst1.16 {q11}, [r10, :128]!
|
||||
vst1.16 {q11}, [r8, :128]!
|
||||
.endif
|
||||
bgt 1b
|
||||
|
||||
subs r7, r7, #2
|
||||
cmp r7, #1
|
||||
blt 0f
|
||||
mov r6, r9
|
||||
add r0, r0, r1
|
||||
add r10, r10, r1
|
||||
add r2, r2, r3
|
||||
add r11, r11, r3
|
||||
add r4, r4, r8
|
||||
add r12, r12, r8
|
||||
add r5, r5, r8
|
||||
add lr, lr, r8
|
||||
beq 2f
|
||||
b 1b
|
||||
b 0f
|
||||
|
||||
2:
|
||||
.if \bpc == 8
|
||||
vld1.8 {d0}, [r2, :64]!
|
||||
vld1.8 {d0}, [r0, :64]
|
||||
.else
|
||||
vld1.16 {q0}, [r2, :128]!
|
||||
vld1.16 {q0}, [r0, :128]
|
||||
.endif
|
||||
vld1.16 {q1}, [r4, :128]!
|
||||
vld1.16 {q2}, [r5, :128]!
|
||||
subs r6, r6, #8
|
||||
.if \bpc == 8
|
||||
vshll.u8 q0, d0, #4 // u
|
||||
.else
|
||||
vshl.i16 q0, q0, #4 // u
|
||||
.endif
|
||||
vsub.i16 q1, q1, q0 // t1 - u
|
||||
vsub.i16 q2, q2, q0 // t2 - u
|
||||
vshll.u16 q3, d0, #7 // u << 7
|
||||
vshll.u16 q0, d1, #7 // u << 7
|
||||
vmlal.s16 q3, d2, d30 // wt[0] * (t1 - u)
|
||||
vmlal.s16 q3, d4, d31 // wt[1] * (t2 - u)
|
||||
vmlal.s16 q0, d3, d30 // wt[0] * (t1 - u)
|
||||
vmlal.s16 q0, d5, d31 // wt[1] * (t2 - u)
|
||||
.if \bpc == 8
|
||||
vld1.16 {q1}, [r2, :128]!
|
||||
vld1.16 {q2}, [r3, :128]!
|
||||
subs r4, r4, #8
|
||||
vmull.s16 q3, d2, d30 // wt[0] * t1
|
||||
vmlal.s16 q3, d4, d31 // wt[1] * t2
|
||||
vmull.s16 q11, d3, d30 // wt[0] * t1
|
||||
vmlal.s16 q11, d5, d31 // wt[1] * t2
|
||||
vrshrn.i32 d6, q3, #11
|
||||
vrshrn.i32 d7, q0, #11
|
||||
vrshrn.i32 d7, q11, #11
|
||||
.if \bpc == 8
|
||||
vaddw.u8 q3, q3, d0
|
||||
vqmovun.s16 d6, q3
|
||||
vst1.8 {d6}, [r0, :64]!
|
||||
.else
|
||||
vqrshrun.s32 d6, q3, #11
|
||||
vqrshrun.s32 d7, q0, #11
|
||||
vmov.i16 q13, #0
|
||||
vadd.i16 q3, q3, q0
|
||||
vmax.s16 q3, q3, q13
|
||||
vmin.u16 q3, q3, q14
|
||||
vst1.16 {q3}, [r0, :128]!
|
||||
.endif
|
||||
bgt 1b
|
||||
bgt 2b
|
||||
0:
|
||||
pop {r4-r11,pc}
|
||||
pop {r4-r8,pc}
|
||||
endfunc
|
||||
.endm
|
||||
|
||||
+47
-42
@@ -109,7 +109,7 @@ L(\type\()_tbl):
|
||||
vst1.32 {d17[0]}, [r0, :32], r1
|
||||
vst1.32 {d17[1]}, [r6, :32], r1
|
||||
beq 0f
|
||||
\type d18, d19, q0, q1, q2, q3
|
||||
\type d18, d19, q0, q1, q2, q3
|
||||
cmp r5, #8
|
||||
vst1.32 {d18[0]}, [r0, :32], r1
|
||||
vst1.32 {d18[1]}, [r6, :32], r1
|
||||
@@ -119,7 +119,7 @@ L(\type\()_tbl):
|
||||
\type d16, d17, q0, q1, q2, q3
|
||||
vst1.32 {d16[0]}, [r0, :32], r1
|
||||
vst1.32 {d16[1]}, [r6, :32], r1
|
||||
\type d18, d19, q0, q1, q2, q3
|
||||
\type d18, d19, q0, q1, q2, q3
|
||||
vst1.32 {d17[0]}, [r0, :32], r1
|
||||
vst1.32 {d17[1]}, [r6, :32], r1
|
||||
vst1.32 {d18[0]}, [r0, :32], r1
|
||||
@@ -288,7 +288,7 @@ L(w_mask_\type\()_tbl):
|
||||
vadd.s16 d21, d22, d23
|
||||
vpadd.s16 d20, d20, d21 // (128 - m) + (128 - n) (column wise addition)
|
||||
vsub.s16 d20, d30, d20 // (256 - sign) - ((128 - m) + (128 - n))
|
||||
vrshrn.u16 d20, q10, #2 // ((256 - sign) - ((128 - m) + (128 - n)) + 2) >> 2
|
||||
vrshrn.u16 d20, q10, #2 // ((256 - sign) - ((128 - m) + (128 - n)) + 2) >> 2
|
||||
vst1.32 {d20[0]}, [r6, :32]!
|
||||
.endif
|
||||
vst1.32 {d24[0]}, [r0, :32], r1
|
||||
@@ -611,7 +611,7 @@ L(blend_h_tbl):
|
||||
vld2.u8 {d2[], d3[]}, [r5, :16]!
|
||||
vld1.u8 {d1}, [r2, :64]!
|
||||
subs r4, r4, #2
|
||||
vext.u8 d2, d2, d3, #4
|
||||
vext.u8 d2, d2, d3, #4
|
||||
vld1.32 {d0[]}, [r0, :32]
|
||||
vsub.i8 d6, d22, d2
|
||||
vld1.32 {d0[1]}, [r12, :32]
|
||||
@@ -623,7 +623,7 @@ L(blend_h_tbl):
|
||||
bgt 4b
|
||||
pop {r4-r5,pc}
|
||||
80:
|
||||
vmov.i8 q8, #64
|
||||
vmov.i8 q8, #64
|
||||
add r12, r0, r1
|
||||
lsl r1, r1, #1
|
||||
8:
|
||||
@@ -933,7 +933,7 @@ L(put_tbl):
|
||||
endfunc
|
||||
|
||||
|
||||
// This has got the same signature as the put_8tap functions,
|
||||
// This has got the same signature as the prep_8tap functions,
|
||||
// assumes that the caller has loaded the h argument into r4,
|
||||
// and assumes that r8 is set to (clz(w)-24), and r7 to w*2.
|
||||
function prep_neon
|
||||
@@ -948,21 +948,26 @@ L(prep_tbl):
|
||||
.word 640f - L(prep_tbl) + CONFIG_THUMB
|
||||
.word 320f - L(prep_tbl) + CONFIG_THUMB
|
||||
.word 160f - L(prep_tbl) + CONFIG_THUMB
|
||||
.word 8f - L(prep_tbl) + CONFIG_THUMB
|
||||
.word 4f - L(prep_tbl) + CONFIG_THUMB
|
||||
.word 80f - L(prep_tbl) + CONFIG_THUMB
|
||||
.word 40f - L(prep_tbl) + CONFIG_THUMB
|
||||
|
||||
40:
|
||||
add r9, r1, r2
|
||||
lsl r2, r2, #1
|
||||
4:
|
||||
vld1.32 {d0[]}, [r1], r2
|
||||
vld1.32 {d2[]}, [r1], r2
|
||||
vld1.32 {d0[]}, [r1], r2
|
||||
vld1.32 {d0[1]}, [r9], r2
|
||||
subs r4, r4, #2
|
||||
vshll.u8 q0, d0, #4
|
||||
vshll.u8 q1, d2, #4
|
||||
vst1.16 {d1, d2}, [r0, :64]!
|
||||
vst1.16 {d0, d1}, [r0, :64]!
|
||||
bgt 4b
|
||||
pop {r4-r11,pc}
|
||||
80:
|
||||
add r9, r1, r2
|
||||
lsl r2, r2, #1
|
||||
8:
|
||||
vld1.8 {d0}, [r1], r2
|
||||
vld1.8 {d2}, [r1], r2
|
||||
vld1.8 {d2}, [r9], r2
|
||||
subs r4, r4, #2
|
||||
vshll.u8 q0, d0, #4
|
||||
vshll.u8 q1, d2, #4
|
||||
@@ -1671,7 +1676,7 @@ L(\type\()_8tap_v_tbl):
|
||||
.endif
|
||||
|
||||
40:
|
||||
bgt 480f
|
||||
bgt 480f
|
||||
|
||||
// 4x2, 4x4 v
|
||||
cmp \h, #2
|
||||
@@ -2493,8 +2498,8 @@ L(\type\()_bilin_h_tbl):
|
||||
2:
|
||||
vld1.32 {d4[]}, [\src], \s_strd
|
||||
vld1.32 {d6[]}, [\sr2], \s_strd
|
||||
vext.8 d5, d4, d4, #1
|
||||
vext.8 d7, d6, d6, #1
|
||||
vext.8 d5, d4, d4, #1
|
||||
vext.8 d7, d6, d6, #1
|
||||
vtrn.16 q2, q3
|
||||
subs \h, \h, #2
|
||||
vmull.u8 q3, d4, d0
|
||||
@@ -2514,8 +2519,8 @@ L(\type\()_bilin_h_tbl):
|
||||
4:
|
||||
vld1.8 {d4}, [\src], \s_strd
|
||||
vld1.8 {d6}, [\sr2], \s_strd
|
||||
vext.8 d5, d4, d4, #1
|
||||
vext.8 d7, d6, d6, #1
|
||||
vext.8 d5, d4, d4, #1
|
||||
vext.8 d7, d6, d6, #1
|
||||
vtrn.32 q2, q3
|
||||
subs \h, \h, #2
|
||||
vmull.u8 q3, d4, d0
|
||||
@@ -2547,8 +2552,8 @@ L(\type\()_bilin_h_tbl):
|
||||
vmlal.u8 q8, d18, d1
|
||||
vmlal.u8 q10, d22, d1
|
||||
.ifc \type, put
|
||||
vqrshrn.u16 d16, q8, #4
|
||||
vqrshrn.u16 d18, q10, #4
|
||||
vqrshrn.u16 d16, q8, #4
|
||||
vqrshrn.u16 d18, q10, #4
|
||||
vst1.8 {d16}, [\dst, :64], \d_strd
|
||||
vst1.8 {d18}, [\ds2, :64], \d_strd
|
||||
.else
|
||||
@@ -2707,7 +2712,7 @@ L(\type\()_bilin_v_tbl):
|
||||
vst1.16 {d5}, [\ds2, :64], \d_strd
|
||||
.endif
|
||||
ble 0f
|
||||
vmov d16, d18
|
||||
vmov d16, d18
|
||||
b 4b
|
||||
0:
|
||||
pop {r4-r11,pc}
|
||||
@@ -2876,8 +2881,8 @@ L(\type\()_bilin_hv_tbl):
|
||||
|
||||
vmov d17, d18
|
||||
|
||||
vmul.u16 q10, q8, q2
|
||||
vmla.u16 q10, q9, q3
|
||||
vmul.u16 q10, q8, q2
|
||||
vmla.u16 q10, q9, q3
|
||||
subs \h, \h, #2
|
||||
.ifc \type, put
|
||||
vqrshrn.u16 d20, q10, #8
|
||||
@@ -3049,8 +3054,8 @@ function warp_affine_8x8\t\()_8bpc_neon, export=1
|
||||
ldr r6, [sp, #108]
|
||||
ldrd r8, r9, [r4]
|
||||
sxth r7, r8
|
||||
asr r8, r8, #16
|
||||
asr r4, r9, #16
|
||||
asr r8, r8, #16
|
||||
asr r4, r9, #16
|
||||
sxth r9, r9
|
||||
mov r10, #8
|
||||
sub r2, r2, r3, lsl #1
|
||||
@@ -3102,26 +3107,26 @@ function warp_affine_8x8\t\()_8bpc_neon, export=1
|
||||
|
||||
// This ordering of vmull/vmlal is highly beneficial for
|
||||
// Cortex A8/A9/A53 here, but harmful for Cortex A7.
|
||||
vmull.s16 q0, d16, d2
|
||||
vmlal.s16 q0, d18, d4
|
||||
vmlal.s16 q0, d20, d6
|
||||
vmlal.s16 q0, d22, d8
|
||||
vmlal.s16 q0, d24, d10
|
||||
vmlal.s16 q0, d26, d12
|
||||
vmull.s16 q1, d17, d3
|
||||
vmlal.s16 q1, d19, d5
|
||||
vmlal.s16 q1, d21, d7
|
||||
vmlal.s16 q1, d23, d9
|
||||
vmlal.s16 q1, d25, d11
|
||||
vmlal.s16 q1, d27, d13
|
||||
vmull.s16 q0, d16, d2
|
||||
vmlal.s16 q0, d18, d4
|
||||
vmlal.s16 q0, d20, d6
|
||||
vmlal.s16 q0, d22, d8
|
||||
vmlal.s16 q0, d24, d10
|
||||
vmlal.s16 q0, d26, d12
|
||||
vmull.s16 q1, d17, d3
|
||||
vmlal.s16 q1, d19, d5
|
||||
vmlal.s16 q1, d21, d7
|
||||
vmlal.s16 q1, d23, d9
|
||||
vmlal.s16 q1, d25, d11
|
||||
vmlal.s16 q1, d27, d13
|
||||
|
||||
vmovl.s8 q2, d14
|
||||
vmovl.s8 q3, d15
|
||||
|
||||
vmlal.s16 q0, d28, d4
|
||||
vmlal.s16 q0, d30, d6
|
||||
vmlal.s16 q1, d29, d5
|
||||
vmlal.s16 q1, d31, d7
|
||||
vmlal.s16 q0, d28, d4
|
||||
vmlal.s16 q0, d30, d6
|
||||
vmlal.s16 q1, d29, d5
|
||||
vmlal.s16 q1, d31, d7
|
||||
|
||||
.ifb \t
|
||||
vmov.i16 q7, #128
|
||||
@@ -3313,7 +3318,7 @@ function emu_edge_8bpc_neon, export=1
|
||||
subs r3, r3, #1
|
||||
vst1.8 {q0, q1}, [r6, :128], r7
|
||||
bgt 2b
|
||||
mls r6, r7, r10, r6 // dst -= bottom_ext * stride
|
||||
mls r6, r7, r10, r6 // dst -= bottom_ext * stride
|
||||
subs r4, r4, #32 // bw -= 32
|
||||
add r6, r6, #32 // dst += 32
|
||||
bgt 1b
|
||||
|
||||
+1
-1
@@ -366,7 +366,7 @@ function msac_decode_hi_tok_neon, export=1
|
||||
add r5, r0, #DIF + 2
|
||||
vld1.16 {q8}, [r4, :128]
|
||||
mov r2, #-24
|
||||
vand d20, d0, d30 // cdf & 0xffc0
|
||||
vand d20, d0, d30 // cdf & 0xffc0
|
||||
ldr r10, [r0, #ALLOW_UPDATE_CDF]
|
||||
vld1.16 {d2[]}, [r5, :16] // dif >> (EC_WIN_SIZE - 16)
|
||||
sub sp, sp, #48
|
||||
|
||||
+5
-5
@@ -133,10 +133,10 @@ function save_tmvs_neon, export=1
|
||||
and r9, r7, #30 // (y & 15) * 2
|
||||
ldr r9, [r2, r9, lsl #2] // b = rr[(y & 15) * 2]
|
||||
add r9, r9, #12 // &b[... + 1]
|
||||
mla r10, r4, r11, r9 // end_cand_b = &b[col_end8*2 + 1]
|
||||
mla r9, r6, r11, r9 // cand_b = &b[x*2 + 1]
|
||||
mla r10, r4, r11, r9 // end_cand_b = &b[col_end8*2 + 1]
|
||||
mla r9, r6, r11, r9 // cand_b = &b[x*2 + 1]
|
||||
|
||||
mla r3, r6, r3, r0 // &rp[x]
|
||||
mla r3, r6, r3, r0 // &rp[x]
|
||||
|
||||
push {r2,r4,r6}
|
||||
|
||||
@@ -175,8 +175,8 @@ function save_tmvs_neon, export=1
|
||||
vmov.u16 r6, d2[1]
|
||||
ldr r11, [r11, #4] // Fetch jump table entry
|
||||
ldr r2, [r2, #4]
|
||||
add r4, r12, r4, lsl #4
|
||||
add r6, r12, r6, lsl #4
|
||||
add r4, r12, r4, lsl #4
|
||||
add r6, r12, r6, lsl #4
|
||||
vld1.8 {d2, d3}, [r4] // Load permutation table base on case
|
||||
vld1.8 {d4, d5}, [r6]
|
||||
add r11, r8, r11 // Find jump table target
|
||||
|
||||
+22
-4
@@ -31,18 +31,36 @@
|
||||
|
||||
#include "config.h"
|
||||
#include "src/arm/asm.S"
|
||||
#include "src/arm/arm-arch.h"
|
||||
|
||||
.macro v4bx rd
|
||||
#if __ARM_ARCH >= 5 || defined(__ARM_ARCH_4T__)
|
||||
bx \rd
|
||||
#else
|
||||
mov pc, \rd
|
||||
#endif
|
||||
.endm
|
||||
|
||||
.macro v4blx rd
|
||||
#if __ARM_ARCH >= 5
|
||||
blx \rd
|
||||
#else
|
||||
mov lr, pc
|
||||
v4bx \rd
|
||||
#endif
|
||||
.endm
|
||||
|
||||
.macro movrel_local rd, val, offset=0
|
||||
#if defined(PIC)
|
||||
#if (__ARM_ARCH >= 7 || defined(__ARM_ARCH_6T2__)) && !defined(PIC)
|
||||
movw \rd, #:lower16:\val+\offset
|
||||
movt \rd, #:upper16:\val+\offset
|
||||
#else
|
||||
ldr \rd, 90001f
|
||||
b 90002f
|
||||
90001:
|
||||
.word \val + \offset - (90002f + 8 - 4 * CONFIG_THUMB)
|
||||
90002:
|
||||
add \rd, \rd, pc
|
||||
#else
|
||||
movw \rd, #:lower16:\val+\offset
|
||||
movt \rd, #:upper16:\val+\offset
|
||||
#endif
|
||||
.endm
|
||||
|
||||
|
||||
+55
-50
@@ -884,12 +884,12 @@ function generate_grain_\type\()_8bpc_neon, export=1
|
||||
.else
|
||||
add x4, x1, #FGD_AR_COEFFS_UV
|
||||
.endif
|
||||
adr x16, L(gen_grain_\type\()_tbl)
|
||||
movrel x16, gen_grain_\type\()_tbl
|
||||
ldr w17, [x1, #FGD_AR_COEFF_LAG]
|
||||
add w9, w9, #4
|
||||
ldrh w17, [x16, w17, uxtw #1]
|
||||
ldrsw x17, [x16, w17, uxtw #2]
|
||||
dup v31.8h, w9 // 4 + data->grain_scale_shift
|
||||
sub x16, x16, w17, uxtw
|
||||
add x16, x16, x17
|
||||
neg v31.8h, v31.8h
|
||||
|
||||
.ifc \type, uv_444
|
||||
@@ -1075,13 +1075,14 @@ L(generate_grain_\type\()_lag3):
|
||||
ldp x30, x19, [sp], #96
|
||||
AARCH64_VALIDATE_LINK_REGISTER
|
||||
ret
|
||||
|
||||
L(gen_grain_\type\()_tbl):
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag0)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag1)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag2)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag3)
|
||||
endfunc
|
||||
|
||||
jumptable gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag0) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag1) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag2) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag3) - gen_grain_\type\()_tbl
|
||||
endjumptable
|
||||
.endm
|
||||
|
||||
gen_grain_82 y
|
||||
@@ -1118,12 +1119,12 @@ function generate_grain_\type\()_8bpc_neon, export=1
|
||||
ldr w2, [x1, #FGD_SEED]
|
||||
ldr w9, [x1, #FGD_GRAIN_SCALE_SHIFT]
|
||||
add x4, x1, #FGD_AR_COEFFS_UV
|
||||
adr x16, L(gen_grain_\type\()_tbl)
|
||||
movrel x16, gen_grain_\type\()_tbl
|
||||
ldr w17, [x1, #FGD_AR_COEFF_LAG]
|
||||
add w9, w9, #4
|
||||
ldrh w17, [x16, w17, uxtw #1]
|
||||
ldrsw x17, [x16, w17, uxtw #2]
|
||||
dup v31.8h, w9 // 4 + data->grain_scale_shift
|
||||
sub x16, x16, w17, uxtw
|
||||
add x16, x16, x17
|
||||
neg v31.8h, v31.8h
|
||||
|
||||
cmp w13, #0
|
||||
@@ -1272,13 +1273,14 @@ L(generate_grain_\type\()_lag3):
|
||||
ldp x30, x19, [sp], #96
|
||||
AARCH64_VALIDATE_LINK_REGISTER
|
||||
ret
|
||||
|
||||
L(gen_grain_\type\()_tbl):
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag0)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag1)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag2)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag3)
|
||||
endfunc
|
||||
|
||||
jumptable gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag0) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag1) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag2) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag3) - gen_grain_\type\()_tbl
|
||||
endjumptable
|
||||
.endm
|
||||
|
||||
gen_grain_44 uv_420
|
||||
@@ -1407,18 +1409,18 @@ function fgy_32x32_8bpc_neon, export=1
|
||||
add_offset x5, w6, x10, x5, x9
|
||||
|
||||
ldr w11, [sp, #24] // type
|
||||
adr x13, L(fgy_loop_tbl)
|
||||
movrel x13, fgy_loop_tbl
|
||||
|
||||
add x4, x12, #32 // grain_lut += FG_BLOCK_SIZE * bx
|
||||
add x6, x14, x9, lsl #5 // grain_lut += grain_stride * FG_BLOCK_SIZE * by
|
||||
|
||||
tst w11, #1
|
||||
ldrh w11, [x13, w11, uxtw #1]
|
||||
ldrsw x11, [x13, w11, uxtw #2]
|
||||
|
||||
add x8, x16, x9, lsl #5 // grain_lut += grain_stride * FG_BLOCK_SIZE * by
|
||||
add x8, x8, #32 // grain_lut += FG_BLOCK_SIZE * bx
|
||||
|
||||
sub x11, x13, w11, uxtw
|
||||
add x11, x13, x11
|
||||
|
||||
b.eq 1f
|
||||
// y overlap
|
||||
@@ -1555,14 +1557,15 @@ L(loop_\ox\oy):
|
||||
fgy 0, 1
|
||||
fgy 1, 0
|
||||
fgy 1, 1
|
||||
|
||||
L(fgy_loop_tbl):
|
||||
.hword L(fgy_loop_tbl) - L(loop_00)
|
||||
.hword L(fgy_loop_tbl) - L(loop_01)
|
||||
.hword L(fgy_loop_tbl) - L(loop_10)
|
||||
.hword L(fgy_loop_tbl) - L(loop_11)
|
||||
endfunc
|
||||
|
||||
jumptable fgy_loop_tbl
|
||||
.word L(loop_00) - fgy_loop_tbl
|
||||
.word L(loop_01) - fgy_loop_tbl
|
||||
.word L(loop_10) - fgy_loop_tbl
|
||||
.word L(loop_11) - fgy_loop_tbl
|
||||
endjumptable
|
||||
|
||||
// void dav1d_fguv_32x32_420_8bpc_neon(pixel *const dst,
|
||||
// const pixel *const src,
|
||||
// const ptrdiff_t stride,
|
||||
@@ -1646,11 +1649,11 @@ function fguv_32x32_\layout\()_8bpc_neon, export=1
|
||||
ldr w13, [sp, #64] // type
|
||||
|
||||
movrel x16, overlap_coeffs_\sx
|
||||
adr x14, L(fguv_loop_sx\sx\()_tbl)
|
||||
movrel x14, fguv_loop_sx\sx\()_tbl
|
||||
|
||||
ld1 {v27.8b, v28.8b}, [x16] // overlap_coeffs
|
||||
tst w13, #1
|
||||
ldrh w13, [x14, w13, uxtw #1]
|
||||
ldrsw x13, [x14, w13, uxtw #2]
|
||||
|
||||
b.eq 1f
|
||||
// y overlap
|
||||
@@ -1658,7 +1661,7 @@ function fguv_32x32_\layout\()_8bpc_neon, export=1
|
||||
mov w9, #(2 >> \sy)
|
||||
|
||||
1:
|
||||
sub x13, x14, w13, uxtw
|
||||
add x13, x14, x13
|
||||
|
||||
.if \sy
|
||||
movi v25.16b, #23
|
||||
@@ -1848,18 +1851,19 @@ L(fguv_loop_sx0_csfl\csfl\()_\ox\oy):
|
||||
ldr x30, [sp], #32
|
||||
AARCH64_VALIDATE_LINK_REGISTER
|
||||
ret
|
||||
|
||||
L(fguv_loop_sx0_tbl):
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl0_00)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl0_01)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl0_10)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl0_11)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl1_00)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl1_01)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl1_10)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl1_11)
|
||||
endfunc
|
||||
|
||||
jumptable fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl0_00) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl0_01) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl0_10) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl0_11) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl1_00) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl1_01) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl1_10) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl1_11) - fguv_loop_sx0_tbl
|
||||
endjumptable
|
||||
|
||||
function fguv_loop_sx1_neon
|
||||
.macro fguv_loop_sx1 csfl, ox, oy
|
||||
L(fguv_loop_sx1_csfl\csfl\()_\ox\oy):
|
||||
@@ -1997,14 +2001,15 @@ L(fguv_loop_sx1_csfl\csfl\()_\ox\oy):
|
||||
ldr x30, [sp], #32
|
||||
AARCH64_VALIDATE_LINK_REGISTER
|
||||
ret
|
||||
|
||||
L(fguv_loop_sx1_tbl):
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl0_00)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl0_01)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl0_10)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl0_11)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl1_00)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl1_01)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl1_10)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl1_11)
|
||||
endfunc
|
||||
|
||||
jumptable fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl0_00) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl0_01) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl0_10) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl0_11) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl1_00) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl1_01) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl1_10) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl1_11) - fguv_loop_sx1_tbl
|
||||
endjumptable
|
||||
|
||||
+57
-52
@@ -695,7 +695,7 @@ function gen_grain_uv_420_lag0_4_neon
|
||||
str x30, [sp, #-16]!
|
||||
ld1 {v16.4h, v17.4h}, [x19]
|
||||
ld1 {v18.4h, v19.4h}, [x12]
|
||||
add x19, x19, #32
|
||||
add x19, x19, #32
|
||||
addp v16.4h, v16.4h, v17.4h
|
||||
addp v17.4h, v18.4h, v19.4h
|
||||
add v16.4h, v16.4h, v17.4h
|
||||
@@ -708,7 +708,7 @@ function gen_grain_uv_422_lag0_4_neon
|
||||
AARCH64_SIGN_LINK_REGISTER
|
||||
str x30, [sp, #-16]!
|
||||
ld1 {v16.4h, v17.4h}, [x19]
|
||||
add x19, x19, #32
|
||||
add x19, x19, #32
|
||||
addp v16.4h, v16.4h, v17.4h
|
||||
srshr v4.4h, v16.4h, #1
|
||||
get_grain_4 v0
|
||||
@@ -740,12 +740,12 @@ function generate_grain_\type\()_16bpc_neon, export=1
|
||||
add x4, x1, #FGD_AR_COEFFS_UV
|
||||
.endif
|
||||
add w9, w9, w15 // grain_scale_shift - bitdepth_min_8
|
||||
adr x16, L(gen_grain_\type\()_tbl)
|
||||
movrel x16, gen_grain_\type\()_tbl
|
||||
ldr w17, [x1, #FGD_AR_COEFF_LAG]
|
||||
add w9, w9, #4
|
||||
ldrh w17, [x16, w17, uxtw #1]
|
||||
ldrsw x17, [x16, w17, uxtw #2]
|
||||
dup v31.8h, w9 // 4 - bitdepth_min_8 + data->grain_scale_shift
|
||||
sub x16, x16, w17, uxtw
|
||||
add x16, x16, x17
|
||||
neg v31.8h, v31.8h
|
||||
|
||||
.ifc \type, uv_444
|
||||
@@ -945,13 +945,14 @@ L(generate_grain_\type\()_lag3):
|
||||
ldp x30, x19, [sp], #96
|
||||
AARCH64_VALIDATE_LINK_REGISTER
|
||||
ret
|
||||
|
||||
L(gen_grain_\type\()_tbl):
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag0)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag1)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag2)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag3)
|
||||
endfunc
|
||||
|
||||
jumptable gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag0) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag1) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag2) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag3) - gen_grain_\type\()_tbl
|
||||
endjumptable
|
||||
.endm
|
||||
|
||||
gen_grain_82 y
|
||||
@@ -991,12 +992,12 @@ function generate_grain_\type\()_16bpc_neon, export=1
|
||||
ldr w9, [x1, #FGD_GRAIN_SCALE_SHIFT]
|
||||
add x4, x1, #FGD_AR_COEFFS_UV
|
||||
add w9, w9, w15 // grain_scale_shift - bitdepth_min_8
|
||||
adr x16, L(gen_grain_\type\()_tbl)
|
||||
movrel x16, gen_grain_\type\()_tbl
|
||||
ldr w17, [x1, #FGD_AR_COEFF_LAG]
|
||||
add w9, w9, #4
|
||||
ldrh w17, [x16, w17, uxtw #1]
|
||||
ldrsw x17, [x16, w17, uxtw #2]
|
||||
dup v31.8h, w9 // 4 - bitdepth_min_8 + data->grain_scale_shift
|
||||
sub x16, x16, w17, uxtw
|
||||
add x16, x16, x17
|
||||
neg v31.8h, v31.8h
|
||||
|
||||
cmp w13, #0
|
||||
@@ -1155,13 +1156,14 @@ L(generate_grain_\type\()_lag3):
|
||||
ldp x30, x19, [sp], #96
|
||||
AARCH64_VALIDATE_LINK_REGISTER
|
||||
ret
|
||||
|
||||
L(gen_grain_\type\()_tbl):
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag0)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag1)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag2)
|
||||
.hword L(gen_grain_\type\()_tbl) - L(generate_grain_\type\()_lag3)
|
||||
endfunc
|
||||
|
||||
jumptable gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag0) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag1) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag2) - gen_grain_\type\()_tbl
|
||||
.word L(generate_grain_\type\()_lag3) - gen_grain_\type\()_tbl
|
||||
endjumptable
|
||||
.endm
|
||||
|
||||
gen_grain_44 uv_420
|
||||
@@ -1306,18 +1308,18 @@ function fgy_32x32_16bpc_neon, export=1
|
||||
add_offset x5, w6, x10, x5, x9
|
||||
|
||||
ldr w11, [sp, #88] // type
|
||||
adr x13, L(fgy_loop_tbl)
|
||||
movrel x13, fgy_loop_tbl
|
||||
|
||||
add x4, x12, #32*2 // grain_lut += FG_BLOCK_SIZE * bx
|
||||
add x6, x14, x9, lsl #5 // grain_lut += grain_stride * FG_BLOCK_SIZE * by
|
||||
|
||||
tst w11, #1
|
||||
ldrh w11, [x13, w11, uxtw #1]
|
||||
ldrsw x11, [x13, w11, uxtw #2]
|
||||
|
||||
add x8, x16, x9, lsl #5 // grain_lut += grain_stride * FG_BLOCK_SIZE * by
|
||||
add x8, x8, #32*2 // grain_lut += FG_BLOCK_SIZE * bx
|
||||
|
||||
sub x11, x13, w11, uxtw
|
||||
add x11, x13, x11
|
||||
|
||||
b.eq 1f
|
||||
// y overlap
|
||||
@@ -1480,14 +1482,15 @@ L(loop_\ox\oy):
|
||||
fgy 0, 1
|
||||
fgy 1, 0
|
||||
fgy 1, 1
|
||||
|
||||
L(fgy_loop_tbl):
|
||||
.hword L(fgy_loop_tbl) - L(loop_00)
|
||||
.hword L(fgy_loop_tbl) - L(loop_01)
|
||||
.hword L(fgy_loop_tbl) - L(loop_10)
|
||||
.hword L(fgy_loop_tbl) - L(loop_11)
|
||||
endfunc
|
||||
|
||||
jumptable fgy_loop_tbl
|
||||
.word L(loop_00) - fgy_loop_tbl
|
||||
.word L(loop_01) - fgy_loop_tbl
|
||||
.word L(loop_10) - fgy_loop_tbl
|
||||
.word L(loop_11) - fgy_loop_tbl
|
||||
endjumptable
|
||||
|
||||
// void dav1d_fguv_32x32_420_16bpc_neon(pixel *const dst,
|
||||
// const pixel *const src,
|
||||
// const ptrdiff_t stride,
|
||||
@@ -1589,11 +1592,11 @@ function fguv_32x32_\layout\()_16bpc_neon, export=1
|
||||
ldr w13, [sp, #112] // type
|
||||
|
||||
movrel x16, overlap_coeffs_\sx
|
||||
adr x14, L(fguv_loop_sx\sx\()_tbl)
|
||||
movrel x14, fguv_loop_sx\sx\()_tbl
|
||||
|
||||
ld1 {v27.4h, v28.4h}, [x16] // overlap_coeffs
|
||||
tst w13, #1
|
||||
ldrh w13, [x14, w13, uxtw #1]
|
||||
ldrsw x13, [x14, w13, uxtw #2]
|
||||
|
||||
b.eq 1f
|
||||
// y overlap
|
||||
@@ -1601,7 +1604,7 @@ function fguv_32x32_\layout\()_16bpc_neon, export=1
|
||||
mov w9, #(2 >> \sy)
|
||||
|
||||
1:
|
||||
sub x13, x14, w13, uxtw
|
||||
add x13, x14, x13
|
||||
|
||||
.if \sy
|
||||
movi v25.8h, #23
|
||||
@@ -1818,18 +1821,19 @@ L(fguv_loop_sx0_csfl\csfl\()_\ox\oy):
|
||||
ldr x30, [sp], #80
|
||||
AARCH64_VALIDATE_LINK_REGISTER
|
||||
ret
|
||||
|
||||
L(fguv_loop_sx0_tbl):
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl0_00)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl0_01)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl0_10)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl0_11)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl1_00)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl1_01)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl1_10)
|
||||
.hword L(fguv_loop_sx0_tbl) - L(fguv_loop_sx0_csfl1_11)
|
||||
endfunc
|
||||
|
||||
jumptable fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl0_00) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl0_01) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl0_10) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl0_11) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl1_00) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl1_01) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl1_10) - fguv_loop_sx0_tbl
|
||||
.word L(fguv_loop_sx0_csfl1_11) - fguv_loop_sx0_tbl
|
||||
endjumptable
|
||||
|
||||
function fguv_loop_sx1_neon
|
||||
.macro fguv_loop_sx1 csfl, ox, oy
|
||||
L(fguv_loop_sx1_csfl\csfl\()_\ox\oy):
|
||||
@@ -1984,14 +1988,15 @@ L(fguv_loop_sx1_csfl\csfl\()_\ox\oy):
|
||||
ldr x30, [sp], #80
|
||||
AARCH64_VALIDATE_LINK_REGISTER
|
||||
ret
|
||||
|
||||
L(fguv_loop_sx1_tbl):
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl0_00)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl0_01)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl0_10)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl0_11)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl1_00)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl1_01)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl1_10)
|
||||
.hword L(fguv_loop_sx1_tbl) - L(fguv_loop_sx1_csfl1_11)
|
||||
endfunc
|
||||
|
||||
jumptable fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl0_00) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl0_01) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl0_10) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl0_11) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl1_00) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl1_01) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl1_10) - fguv_loop_sx1_tbl
|
||||
.word L(fguv_loop_sx1_csfl1_11) - fguv_loop_sx1_tbl
|
||||
endjumptable
|
||||
|
||||
+631
-685
File diff suppressed because it is too large
Load Diff
+333
-293
File diff suppressed because it is too large
Load Diff
+2
-2
@@ -545,7 +545,7 @@ endfunc
|
||||
sqrshrn2 \o0\().8h, v17.4s, #12
|
||||
|
||||
.ifc \o2, v17
|
||||
mov v17.16b, v18.16b
|
||||
mov v17.16b, v18.16b
|
||||
.endif
|
||||
|
||||
sqrshrn \o1\().4h, v6.4s, #12
|
||||
@@ -1993,7 +1993,7 @@ function inv_dct32_odd_8h_x16_neon, export=1
|
||||
|
||||
smull_smlal v4, v5, v25, v27, v0.h[0], v0.h[0], .8h // -> t26a
|
||||
smull_smlsl v6, v7, v25, v27, v0.h[0], v0.h[0], .8h // -> t21a
|
||||
mov v27.16b, v22.16b // t27
|
||||
mov v27.16b, v22.16b // t27
|
||||
sqrshrn_sz v26, v4, v5, #12, .8h // t26a
|
||||
|
||||
smull_smlsl v24, v25, v21, v23, v0.h[0], v0.h[0], .8h // -> t22
|
||||
|
||||
@@ -305,7 +305,7 @@ function lpf_16_wd\wd\()_neon
|
||||
rshrn2 v0.16b, v9.8h, #3
|
||||
|
||||
add v8.8h, v8.8h, v2.8h
|
||||
add v9.8h , v9.8h, v3.8h
|
||||
add v9.8h, v9.8h, v3.8h
|
||||
|
||||
bit v21.16b, v10.16b, v14.16b
|
||||
bit v22.16b, v11.16b, v14.16b
|
||||
@@ -696,7 +696,7 @@ function lpf_v_8_16_neon
|
||||
lpf_16_wd8
|
||||
|
||||
sub x16, x0, x1, lsl #1
|
||||
sub x16, x16, x1
|
||||
sub x16, x16, x1
|
||||
st1 {v21.16b}, [x16], x1 // p2
|
||||
st1 {v24.16b}, [x0], x1 // q0
|
||||
st1 {v22.16b}, [x16], x1 // p1
|
||||
@@ -817,7 +817,7 @@ function lpf_v_16_16_neon
|
||||
lpf_16_wd16
|
||||
|
||||
sub x16, x0, x1, lsl #2
|
||||
sub x16, x16, x1, lsl #1
|
||||
sub x16, x16, x1, lsl #1
|
||||
st1 {v0.16b}, [x16], x1 // p5
|
||||
st1 {v6.16b}, [x0], x1 // q0
|
||||
st1 {v1.16b}, [x16], x1 // p4
|
||||
@@ -1051,7 +1051,7 @@ function lpf_\dir\()_sb_\type\()_8bpc_neon, export=1
|
||||
adds x16, x16, x17
|
||||
b.eq 7f // if (!L) continue;
|
||||
neg v5.16b, v5.16b // -sharp[0]
|
||||
movrel x16, word_1248
|
||||
movrel x16, word_1248
|
||||
ushr v12.16b, v1.16b, #4 // H
|
||||
ld1 {v16.4s}, [x16]
|
||||
sshl v3.16b, v1.16b, v5.16b // L >> sharp[0]
|
||||
|
||||
@@ -550,7 +550,7 @@ function lpf_v_8_8_neon
|
||||
lpf_8_wd8
|
||||
|
||||
sub x16, x0, x1, lsl #1
|
||||
sub x16, x16, x1
|
||||
sub x16, x16, x1
|
||||
st1 {v21.8h}, [x16], x1 // p2
|
||||
st1 {v24.8h}, [x0], x1 // q0
|
||||
st1 {v22.8h}, [x16], x1 // p1
|
||||
@@ -781,7 +781,7 @@ function lpf_\dir\()_sb_\type\()_16bpc_neon, export=1
|
||||
mov w8, w7 // bitdepth_max
|
||||
clz w9, w8
|
||||
mov w10, #24
|
||||
sub w9, w10, w9 // bitdepth_min_8
|
||||
sub w9, w10, w9 // bitdepth_min_8
|
||||
stp d8, d9, [sp, #-0x40]!
|
||||
stp d10, d11, [sp, #0x10]
|
||||
stp d12, d13, [sp, #0x20]
|
||||
@@ -836,7 +836,7 @@ function lpf_\dir\()_sb_\type\()_16bpc_neon, export=1
|
||||
cmp x16, #0
|
||||
b.eq 7f // if (!L) continue;
|
||||
neg v5.8b, v5.8b // -sharp[0]
|
||||
movrel x16, word_12
|
||||
movrel x16, word_12
|
||||
ushr v12.8b, v1.8b, #4 // H
|
||||
ld1 {v16.2s}, [x16]
|
||||
sshl v3.8b, v1.8b, v5.8b // L >> sharp[0]
|
||||
|
||||
@@ -1129,7 +1129,7 @@ function sgr_box3_row_h_16bpc_neon, export=1
|
||||
// again; it's not strictly needed in those cases (we pad enough here),
|
||||
// but keeping the code as simple as possible.
|
||||
|
||||
// Insert padding in v0.b[w] onwards
|
||||
// Insert padding in v0.h[w] onwards
|
||||
movrel x13, right_ext_mask
|
||||
sub x13, x13, w4, uxtw #1
|
||||
ld1 {v28.16b, v29.16b}, [x13]
|
||||
@@ -1224,7 +1224,7 @@ function sgr_box5_row_h_16bpc_neon, export=1
|
||||
// this ends up called again; it's not strictly needed in those
|
||||
// cases (we pad enough here), but keeping the code as simple as possible.
|
||||
|
||||
// Insert padding in v0.b[w+1] onwards; fuse the +1 into the
|
||||
// Insert padding in v0.h[w+1] onwards; fuse the +1 into the
|
||||
// buffer pointer.
|
||||
movrel x13, right_ext_mask, -1
|
||||
sub x13, x13, w4, uxtw #1
|
||||
|
||||
+170
-108
@@ -28,14 +28,77 @@
|
||||
#include "src/arm/asm.S"
|
||||
#include "util.S"
|
||||
|
||||
// Series of LUTs for efficiently computing sgr's 1 - x/(x+1) table.
|
||||
// In the comments, let RefTable denote the original, reference table.
|
||||
const x_by_x_tables
|
||||
// RangeMins
|
||||
//
|
||||
// Min(RefTable[i*8:i*8+8])
|
||||
// First two values are zeroed.
|
||||
//
|
||||
// Lookup using RangeMins[(x >> 3)]
|
||||
.byte 0, 0, 11, 8, 6, 5, 5, 4, 4, 3, 3, 3, 2, 2, 2, 2
|
||||
.byte 2, 2, 2, 2, 2, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0
|
||||
|
||||
// DiffMasks
|
||||
//
|
||||
// This contains a bit pattern, indicating at which index positions the value of RefTable changes. For each range
|
||||
// in the RangeMins table (covering 8 RefTable entries), we have one byte; each bit indicates whether the value of
|
||||
// RefTable changes at that particular index.
|
||||
// Using popcount, we can integrate the diff bit field. By shifting away bits in a byte, we can refine the range of
|
||||
// the integral. Finally, adding the integral to RangeMins[(x>>3)] reconstructs RefTable (for x > 15).
|
||||
//
|
||||
// Lookup using DiffMasks[(x >> 3)]
|
||||
.byte 0x00, 0x00, 0xD4, 0x44
|
||||
.byte 0x42, 0x04, 0x00, 0x00
|
||||
.byte 0x00, 0x80, 0x00, 0x00
|
||||
.byte 0x04, 0x00, 0x00, 0x00
|
||||
.byte 0x00, 0x00, 0x00, 0x00
|
||||
.byte 0x00, 0x40, 0x00, 0x00
|
||||
.byte 0x00, 0x00, 0x00, 0x00
|
||||
.byte 0x00, 0x00, 0x00, 0x02
|
||||
// Binary form:
|
||||
// 0b00000000, 0b00000000, 0b11010100, 0b01000100
|
||||
// 0b01000010, 0b00000100, 0b00000000, 0b00000000
|
||||
// 0b00000000, 0b10000000, 0b00000000, 0b00000000
|
||||
// 0b00000100, 0b00000000, 0b00000000, 0b00000000
|
||||
// 0b00000000, 0b00000000, 0b00000000, 0b00000000
|
||||
// 0b00000000, 0b01000000, 0b00000000, 0b00000000
|
||||
// 0b00000000, 0b00000000, 0b00000000, 0b00000000
|
||||
// 0b00000000, 0b00000000, 0b00000000, 0b00000010
|
||||
|
||||
// RefLo
|
||||
//
|
||||
// RefTable[0:16]
|
||||
// i.e. First 16 elements of the original table.
|
||||
// Add to the sum obtained in the rest of the other lut logic to include the first 16 bytes of RefTable.
|
||||
//
|
||||
// Lookup using RangeMins[x] (tbl will replace x > 15 with 0)
|
||||
.byte 255, 128, 85, 64, 51, 43, 37, 32, 28, 26, 23, 21, 20, 18, 17, 16
|
||||
|
||||
// Pseudo assembly
|
||||
//
|
||||
// hi_bits = x >> 3
|
||||
// tbl ref, {RefLo}, x
|
||||
// tbl diffs, {DiffMasks[0:16], DiffMasks[16:32]}, hi_bits
|
||||
// tbl min, {RangeMins[0:16], RangeMins[16:32]}, hi_bits
|
||||
// lo_bits = x & 0x7
|
||||
// diffs = diffs << lo_bits
|
||||
// ref = ref + min
|
||||
// integral = popcnt(diffs)
|
||||
// ref = ref + integral
|
||||
// return ref
|
||||
endconst
|
||||
|
||||
// void dav1d_sgr_box3_vert_neon(int32_t **sumsq, int16_t **sum,
|
||||
// int32_t *AA, int16_t *BB,
|
||||
// const int w, const int s,
|
||||
// const int bitdepth_max);
|
||||
function sgr_box3_vert_neon, export=1
|
||||
stp d8, d9, [sp, #-0x30]!
|
||||
stp d8, d9, [sp, #-0x40]!
|
||||
stp d10, d11, [sp, #0x10]
|
||||
stp d12, d13, [sp, #0x20]
|
||||
stp d14, d15, [sp, #0x30]
|
||||
|
||||
add w4, w4, #2
|
||||
clz w9, w6 // bitdepth_max
|
||||
@@ -49,93 +112,109 @@ function sgr_box3_vert_neon, export=1
|
||||
movi v31.4s, #9 // n
|
||||
|
||||
sub w9, w9, #24 // -bitdepth_min_8
|
||||
movrel x12, X(sgr_x_by_x)
|
||||
movrel x12, x_by_x_tables
|
||||
mov w13, #455 // one_by_x
|
||||
ld1 {v16.16b, v17.16b, v18.16b}, [x12]
|
||||
ld1 {v24.16b, v25.16b, v26.16b, v27.16b}, [x12] // RangeMins, DiffMasks
|
||||
movi v22.16b, #0x7
|
||||
ldr q23, [x12, #64] //RefLo
|
||||
dup v6.8h, w9 // -bitdepth_min_8
|
||||
movi v19.16b, #5
|
||||
movi v20.8b, #55 // idx of last 5
|
||||
movi v21.8b, #72 // idx of last 4
|
||||
movi v22.8b, #101 // idx of last 3
|
||||
movi v23.8b, #169 // idx of last 2
|
||||
movi v24.8b, #254 // idx of last 1
|
||||
saddl v7.4s, v6.4h, v6.4h // -2*bitdepth_min_8
|
||||
movi v29.8h, #1, lsl #8
|
||||
dup v30.4s, w13 // one_by_x
|
||||
|
||||
sub v16.16b, v16.16b, v19.16b
|
||||
sub v17.16b, v17.16b, v19.16b
|
||||
sub v18.16b, v18.16b, v19.16b
|
||||
|
||||
ld1 {v8.4s, v9.4s}, [x5], #32
|
||||
ld1 {v10.4s, v11.4s}, [x6], #32
|
||||
ld1 {v12.8h}, [x7], #16
|
||||
ld1 {v13.8h}, [x8], #16
|
||||
ld1 {v0.4s, v1.4s}, [x0], #32
|
||||
ld1 {v2.8h}, [x1], #16
|
||||
ld1 {v8.4s, v9.4s, v10.4s, v11.4s}, [x5], #64
|
||||
ld1 {v12.4s, v13.4s, v14.4s, v15.4s}, [x6], #64
|
||||
ld1 {v16.4s, v17.4s, v18.4s, v19.4s}, [x0], #64
|
||||
ld1 {v20.8h, v21.8h}, [x8], #32
|
||||
ld1 {v0.8h, v1.8h}, [x7], #32
|
||||
1:
|
||||
ld1 {v2.8h, v3.8h}, [x1], #32
|
||||
add v8.4s, v8.4s, v12.4s
|
||||
add v9.4s, v9.4s, v13.4s
|
||||
add v10.4s, v10.4s, v14.4s
|
||||
add v11.4s, v11.4s, v15.4s
|
||||
add v0.8h, v0.8h, v20.8h
|
||||
add v1.8h, v1.8h, v21.8h
|
||||
|
||||
add v8.4s, v8.4s, v10.4s
|
||||
add v9.4s, v9.4s, v11.4s
|
||||
add v16.4s, v16.4s, v8.4s
|
||||
add v17.4s, v17.4s, v9.4s
|
||||
add v18.4s, v18.4s, v10.4s
|
||||
add v19.4s, v19.4s, v11.4s
|
||||
add v4.8h, v2.8h, v0.8h
|
||||
add v5.8h, v3.8h, v1.8h
|
||||
|
||||
add v12.8h, v12.8h, v13.8h
|
||||
srshl v16.4s, v16.4s, v7.4s
|
||||
srshl v17.4s, v17.4s, v7.4s
|
||||
srshl v18.4s, v18.4s, v7.4s
|
||||
srshl v19.4s, v19.4s, v7.4s
|
||||
srshl v9.8h, v4.8h, v6.8h
|
||||
srshl v13.8h, v5.8h, v6.8h
|
||||
mul v16.4s, v16.4s, v31.4s // a * n
|
||||
mul v17.4s, v17.4s, v31.4s // a * n
|
||||
mul v18.4s, v18.4s, v31.4s // a * n
|
||||
mul v19.4s, v19.4s, v31.4s // a * n
|
||||
umull v8.4s, v9.4h, v9.4h // b * b
|
||||
umull2 v9.4s, v9.8h, v9.8h // b * b
|
||||
umull v12.4s, v13.4h, v13.4h // b * b
|
||||
umull2 v13.4s, v13.8h, v13.8h // b * b
|
||||
uqsub v16.4s, v16.4s, v8.4s // imax(a * n - b * b, 0)
|
||||
uqsub v17.4s, v17.4s, v9.4s // imax(a * n - b * b, 0)
|
||||
uqsub v18.4s, v18.4s, v12.4s // imax(a * n - b * b, 0)
|
||||
uqsub v19.4s, v19.4s, v13.4s // imax(a * n - b * b, 0)
|
||||
mul v16.4s, v16.4s, v28.4s // p * s
|
||||
mul v17.4s, v17.4s, v28.4s // p * s
|
||||
mul v18.4s, v18.4s, v28.4s // p * s
|
||||
mul v19.4s, v19.4s, v28.4s // p * s
|
||||
uqshrn v16.4h, v16.4s, #16
|
||||
uqshrn2 v16.8h, v17.4s, #16
|
||||
uqshrn v18.4h, v18.4s, #16
|
||||
uqshrn2 v18.8h, v19.4s, #16
|
||||
uqrshrn v1.8b, v16.8h, #4 // imin(z, 255)
|
||||
uqrshrn2 v1.16b, v18.8h, #4 // imin(z, 255)
|
||||
|
||||
subs w4, w4, #8
|
||||
add v0.4s, v0.4s, v8.4s
|
||||
add v1.4s, v1.4s, v9.4s
|
||||
add v2.8h, v2.8h, v12.8h
|
||||
ld1 {v16.4s, v17.4s}, [x0], #32
|
||||
subs w4, w4, #16
|
||||
|
||||
srshl v0.4s, v0.4s, v7.4s
|
||||
srshl v1.4s, v1.4s, v7.4s
|
||||
srshl v4.8h, v2.8h, v6.8h
|
||||
mul v0.4s, v0.4s, v31.4s // a * n
|
||||
mul v1.4s, v1.4s, v31.4s // a * n
|
||||
umull v3.4s, v4.4h, v4.4h // b * b
|
||||
umull2 v4.4s, v4.8h, v4.8h // b * b
|
||||
uqsub v0.4s, v0.4s, v3.4s // imax(a * n - b * b, 0)
|
||||
uqsub v1.4s, v1.4s, v4.4s // imax(a * n - b * b, 0)
|
||||
mul v0.4s, v0.4s, v28.4s // p * s
|
||||
mul v1.4s, v1.4s, v28.4s // p * s
|
||||
ld1 {v8.4s, v9.4s}, [x5], #32
|
||||
uqshrn v0.4h, v0.4s, #16
|
||||
uqshrn2 v0.8h, v1.4s, #16
|
||||
ld1 {v10.4s, v11.4s}, [x6], #32
|
||||
uqrshrn v0.8b, v0.8h, #4 // imin(z, 255)
|
||||
ushr v0.16b, v1.16b, #3
|
||||
ld1 {v8.4s, v9.4s}, [x5], #32
|
||||
tbl v2.16b, {v26.16b, v27.16b}, v0.16b // RangeMins
|
||||
tbl v0.16b, {v24.16b, v25.16b}, v0.16b // DiffMasks
|
||||
tbl v3.16b, {v23.16b}, v1.16b // RefLo
|
||||
and v1.16b, v1.16b, v22.16b
|
||||
ld1 {v12.4s, v13.4s}, [x6], #32
|
||||
ushl v1.16b, v2.16b, v1.16b
|
||||
ld1 {v20.8h, v21.8h}, [x8], #32
|
||||
add v3.16b, v3.16b, v0.16b
|
||||
cnt v1.16b, v1.16b
|
||||
ld1 {v18.4s, v19.4s}, [x0], #32
|
||||
add v3.16b, v3.16b, v1.16b
|
||||
ld1 {v10.4s, v11.4s}, [x5], #32
|
||||
uxtl v0.8h, v3.8b // x
|
||||
uxtl2 v1.8h, v3.16b // x
|
||||
|
||||
ld1 {v12.8h}, [x7], #16
|
||||
ld1 {v14.4s, v15.4s}, [x6], #32
|
||||
|
||||
cmhi v25.8b, v0.8b, v20.8b // = -1 if sgr_x_by_x[v0] < 5
|
||||
cmhi v26.8b, v0.8b, v21.8b // = -1 if sgr_x_by_x[v0] < 4
|
||||
tbl v1.8b, {v16.16b, v17.16b, v18.16b}, v0.8b
|
||||
cmhi v27.8b, v0.8b, v22.8b // = -1 if sgr_x_by_x[v0] < 3
|
||||
cmhi v4.8b, v0.8b, v23.8b // = -1 if sgr_x_by_x[v0] < 2
|
||||
add v25.8b, v25.8b, v26.8b
|
||||
cmhi v5.8b, v0.8b, v24.8b // = -1 if sgr_x_by_x[v0] < 1
|
||||
add v27.8b, v27.8b, v4.8b
|
||||
add v5.8b, v5.8b, v19.8b
|
||||
add v25.8b, v25.8b, v27.8b
|
||||
add v5.8b, v1.8b, v5.8b
|
||||
ld1 {v13.8h}, [x8], #16
|
||||
add v5.8b, v5.8b, v25.8b
|
||||
ld1 {v0.4s, v1.4s}, [x0], #32
|
||||
uxtl v5.8h, v5.8b // x
|
||||
umull v2.4s, v0.4h, v4.4h // x * BB[i]
|
||||
umull2 v3.4s, v0.8h, v4.8h // x * BB[i]
|
||||
umull v4.4s, v1.4h, v5.4h // x * BB[i]
|
||||
umull2 v5.4s, v1.8h, v5.8h // x * BB[i]
|
||||
mul v2.4s, v2.4s, v30.4s // x * BB[i] * sgr_one_by_x
|
||||
mul v3.4s, v3.4s, v30.4s // x * BB[i] * sgr_one_by_x
|
||||
mul v4.4s, v4.4s, v30.4s // x * BB[i] * sgr_one_by_x
|
||||
mul v5.4s, v5.4s, v30.4s // x * BB[i] * sgr_one_by_x
|
||||
st1 {v0.8h, v1.8h}, [x3], #32
|
||||
ld1 {v0.8h, v1.8h}, [x7], #32
|
||||
srshr v2.4s, v2.4s, #12 // AA[i]
|
||||
srshr v3.4s, v3.4s, #12 // AA[i]
|
||||
srshr v4.4s, v4.4s, #12 // AA[i]
|
||||
srshr v5.4s, v5.4s, #12 // AA[i]
|
||||
|
||||
umull v3.4s, v5.4h, v2.4h // x * BB[i]
|
||||
umull2 v4.4s, v5.8h, v2.8h // x * BB[i]
|
||||
mul v3.4s, v3.4s, v30.4s // x * BB[i] * sgr_one_by_x
|
||||
mul v4.4s, v4.4s, v30.4s // x * BB[i] * sgr_one_by_x
|
||||
srshr v3.4s, v3.4s, #12 // AA[i]
|
||||
srshr v4.4s, v4.4s, #12 // AA[i]
|
||||
sub v5.8h, v29.8h, v5.8h // 256 - x
|
||||
ld1 {v2.8h}, [x1], #16
|
||||
|
||||
st1 {v3.4s, v4.4s}, [x2], #32
|
||||
st1 {v5.8h}, [x3], #16
|
||||
st1 {v2.4s, v3.4s, v4.4s, v5.4s}, [x2], #64
|
||||
b.gt 1b
|
||||
|
||||
ldp d14, d15, [sp, #0x30]
|
||||
ldp d12, d13, [sp, #0x20]
|
||||
ldp d10, d11, [sp, #0x10]
|
||||
ldp d8, d9, [sp], 0x30
|
||||
ldp d8, d9, [sp], 0x40
|
||||
ret
|
||||
endfunc
|
||||
|
||||
@@ -144,10 +223,9 @@ endfunc
|
||||
// const int w, const int s,
|
||||
// const int bitdepth_max);
|
||||
function sgr_box5_vert_neon, export=1
|
||||
stp d8, d9, [sp, #-0x40]!
|
||||
stp d8, d9, [sp, #-0x30]!
|
||||
stp d10, d11, [sp, #0x10]
|
||||
stp d12, d13, [sp, #0x20]
|
||||
stp d14, d15, [sp, #0x30]
|
||||
|
||||
add w4, w4, #2
|
||||
clz w15, w6 // bitdepth_max
|
||||
@@ -163,24 +241,18 @@ function sgr_box5_vert_neon, export=1
|
||||
movi v31.4s, #25 // n
|
||||
|
||||
sub w15, w15, #24 // -bitdepth_min_8
|
||||
movrel x13, X(sgr_x_by_x)
|
||||
mov w14, #164 // one_by_x
|
||||
ld1 {v16.16b, v17.16b, v18.16b}, [x13]
|
||||
movrel x13, x_by_x_tables
|
||||
movi v30.4s, #164
|
||||
ld1 {v24.16b, v25.16b, v26.16b, v27.16b}, [x13] // RangeMins, DiffMasks
|
||||
dup v6.8h, w15 // -bitdepth_min_8
|
||||
movi v19.16b, #5
|
||||
movi v24.8b, #254 // idx of last 1
|
||||
movi v19.8b, #0x7
|
||||
ldr q18, [x13, #64] // RefLo
|
||||
saddl v7.4s, v6.4h, v6.4h // -2*bitdepth_min_8
|
||||
movi v29.8h, #1, lsl #8
|
||||
dup v30.4s, w14 // one_by_x
|
||||
|
||||
sub v16.16b, v16.16b, v19.16b
|
||||
sub v17.16b, v17.16b, v19.16b
|
||||
sub v18.16b, v18.16b, v19.16b
|
||||
|
||||
ld1 {v8.4s, v9.4s}, [x5], #32
|
||||
ld1 {v10.4s, v11.4s}, [x6], #32
|
||||
ld1 {v12.4s, v13.4s}, [x7], #32
|
||||
ld1 {v14.4s, v15.4s}, [x8], #32
|
||||
ld1 {v16.4s, v17.4s}, [x8], #32
|
||||
ld1 {v20.8h}, [x9], #16
|
||||
ld1 {v21.8h}, [x10], #16
|
||||
ld1 {v22.8h}, [x11], #16
|
||||
@@ -191,8 +263,8 @@ function sgr_box5_vert_neon, export=1
|
||||
1:
|
||||
add v8.4s, v8.4s, v10.4s
|
||||
add v9.4s, v9.4s, v11.4s
|
||||
add v12.4s, v12.4s, v14.4s
|
||||
add v13.4s, v13.4s, v15.4s
|
||||
add v12.4s, v12.4s, v16.4s
|
||||
add v13.4s, v13.4s, v17.4s
|
||||
|
||||
add v20.8h, v20.8h, v21.8h
|
||||
add v22.8h, v22.8h, v23.8h
|
||||
@@ -207,11 +279,6 @@ function sgr_box5_vert_neon, export=1
|
||||
|
||||
subs w4, w4, #8
|
||||
|
||||
movi v20.8b, #55 // idx of last 5
|
||||
movi v21.8b, #72 // idx of last 4
|
||||
movi v22.8b, #101 // idx of last 3
|
||||
movi v23.8b, #169 // idx of last 2
|
||||
|
||||
srshl v0.4s, v0.4s, v7.4s
|
||||
srshl v1.4s, v1.4s, v7.4s
|
||||
srshl v4.8h, v2.8h, v6.8h
|
||||
@@ -231,22 +298,19 @@ function sgr_box5_vert_neon, export=1
|
||||
|
||||
ld1 {v12.4s, v13.4s}, [x7], #32
|
||||
|
||||
cmhi v25.8b, v0.8b, v20.8b // = -1 if sgr_x_by_x[v0] < 5
|
||||
cmhi v26.8b, v0.8b, v21.8b // = -1 if sgr_x_by_x[v0] < 4
|
||||
tbl v1.8b, {v16.16b, v17.16b, v18.16b}, v0.8b
|
||||
cmhi v27.8b, v0.8b, v22.8b // = -1 if sgr_x_by_x[v0] < 3
|
||||
cmhi v4.8b, v0.8b, v23.8b // = -1 if sgr_x_by_x[v0] < 2
|
||||
ld1 {v14.4s, v15.4s}, [x8], #32
|
||||
add v25.8b, v25.8b, v26.8b
|
||||
cmhi v5.8b, v0.8b, v24.8b // = -1 if sgr_x_by_x[v0] < 1
|
||||
add v27.8b, v27.8b, v4.8b
|
||||
ushr v1.8b, v0.8b, #3
|
||||
ld1 {v16.4s, v17.4s}, [x8], #32
|
||||
tbl v5.8b, {v26.16b, v27.16b}, v1.8b // RangeMins
|
||||
tbl v1.8b, {v24.16b, v25.16b}, v1.8b // DiffMasks
|
||||
tbl v4.8b, {v18.16b}, v0.8b // RefLo
|
||||
and v0.8b, v0.8b, v19.8b
|
||||
ld1 {v20.8h}, [x9], #16
|
||||
add v5.8b, v5.8b, v19.8b
|
||||
add v25.8b, v25.8b, v27.8b
|
||||
ushl v5.8b, v5.8b, v0.8b
|
||||
add v4.8b, v4.8b, v1.8b
|
||||
ld1 {v21.8h}, [x10], #16
|
||||
add v5.8b, v1.8b, v5.8b
|
||||
cnt v5.8b, v5.8b
|
||||
ld1 {v22.8h}, [x11], #16
|
||||
add v5.8b, v5.8b, v25.8b
|
||||
add v5.8b, v4.8b, v5.8b
|
||||
ld1 {v23.8h}, [x12], #16
|
||||
uxtl v5.8h, v5.8b // x
|
||||
|
||||
@@ -257,16 +321,14 @@ function sgr_box5_vert_neon, export=1
|
||||
mul v4.4s, v4.4s, v30.4s // x * BB[i] * sgr_one_by_x
|
||||
srshr v3.4s, v3.4s, #12 // AA[i]
|
||||
srshr v4.4s, v4.4s, #12 // AA[i]
|
||||
sub v5.8h, v29.8h, v5.8h // 256 - x
|
||||
ld1 {v2.8h}, [x1], #16
|
||||
|
||||
st1 {v3.4s, v4.4s}, [x2], #32
|
||||
st1 {v5.8h}, [x3], #16
|
||||
b.gt 1b
|
||||
|
||||
ldp d14, d15, [sp, #0x30]
|
||||
ldp d12, d13, [sp, #0x20]
|
||||
ldp d10, d11, [sp, #0x10]
|
||||
ldp d8, d9, [sp], 0x40
|
||||
ldp d8, d9, [sp], 0x30
|
||||
ret
|
||||
endfunc
|
||||
|
||||
@@ -174,10 +174,10 @@ function sgr_finish_filter1_2rows_\bpc\()bpc_neon, export=1
|
||||
mla v14.4s, v19.4s, v31.4s // * 3 -> b
|
||||
mla v15.4s, v20.4s, v31.4s
|
||||
|
||||
umlal v8.4s, v4.4h, v25.4h // b + a * src
|
||||
umlal2 v9.4s, v4.8h, v25.8h
|
||||
umlal v14.4s, v0.4h, v26.4h // b + a * src
|
||||
umlal2 v15.4s, v0.8h, v26.8h
|
||||
umlsl v8.4s, v4.4h, v25.4h // b - a * src
|
||||
umlsl2 v9.4s, v4.8h, v25.8h
|
||||
umlsl v14.4s, v0.4h, v26.4h // b - a * src
|
||||
umlsl2 v15.4s, v0.8h, v26.8h
|
||||
mov v0.16b, v1.16b
|
||||
rshrn v8.4h, v8.4s, #9
|
||||
rshrn2 v8.8h, v9.4s, #9
|
||||
@@ -292,8 +292,8 @@ function sgr_finish_weighted1_\bpc\()bpc_neon, export=1
|
||||
uxtl v19.8h, v19.8b // src
|
||||
.endif
|
||||
mov v0.16b, v1.16b
|
||||
umlal v25.4s, v2.4h, v19.4h // b + a * src
|
||||
umlal2 v26.4s, v2.8h, v19.8h
|
||||
umlsl v25.4s, v2.4h, v19.4h // b - a * src
|
||||
umlsl2 v26.4s, v2.8h, v19.8h
|
||||
mov v2.16b, v3.16b
|
||||
rshrn v25.4h, v25.4s, #9
|
||||
rshrn2 v25.8h, v26.4s, #9
|
||||
@@ -301,30 +301,25 @@ function sgr_finish_weighted1_\bpc\()bpc_neon, export=1
|
||||
subs w3, w3, #8
|
||||
|
||||
// weighted1
|
||||
shl v19.8h, v19.8h, #4 // u
|
||||
mov v4.16b, v5.16b
|
||||
|
||||
sub v25.8h, v25.8h, v19.8h // t1 - u
|
||||
ld1 {v1.8h}, [x9], #16
|
||||
ushll v26.4s, v19.4h, #7 // u << 7
|
||||
ushll2 v27.4s, v19.8h, #7 // u << 7
|
||||
ld1 {v3.8h}, [x10], #16
|
||||
smlal v26.4s, v25.4h, v31.4h // v
|
||||
smlal2 v27.4s, v25.8h, v31.8h // v
|
||||
smull v26.4s, v25.4h, v31.4h // v = t1 * w1
|
||||
smull2 v27.4s, v25.8h, v31.8h
|
||||
ld1 {v5.8h}, [x2], #16
|
||||
.if \bpc == 8
|
||||
rshrn v26.4h, v26.4s, #11
|
||||
rshrn2 v26.8h, v27.4s, #11
|
||||
usqadd v19.8h, v26.8h
|
||||
.if \bpc == 8
|
||||
mov v16.16b, v18.16b
|
||||
sqxtun v26.8b, v26.8h
|
||||
sqxtun v26.8b, v19.8h
|
||||
mov v19.16b, v21.16b
|
||||
mov v22.16b, v24.16b
|
||||
st1 {v26.8b}, [x0], #8
|
||||
.else
|
||||
sqrshrun v26.4h, v26.4s, #11
|
||||
sqrshrun2 v26.8h, v27.4s, #11
|
||||
mov v16.16b, v18.16b
|
||||
umin v26.8h, v26.8h, v30.8h
|
||||
umin v26.8h, v19.8h, v30.8h
|
||||
mov v19.16b, v21.16b
|
||||
mov v22.16b, v24.16b
|
||||
st1 {v26.8h}, [x0], #16
|
||||
@@ -424,10 +419,10 @@ function sgr_finish_filter2_2rows_\bpc\()bpc_neon, export=1
|
||||
uxtl v31.8h, v31.8b
|
||||
uxtl v30.8h, v30.8b
|
||||
.endif
|
||||
umlal v16.4s, v0.4h, v31.4h // b + a * src
|
||||
umlal2 v17.4s, v0.8h, v31.8h
|
||||
umlal v9.4s, v8.4h, v30.4h // b + a * src
|
||||
umlal2 v10.4s, v8.8h, v30.8h
|
||||
umlsl v16.4s, v0.4h, v31.4h // b - a * src
|
||||
umlsl2 v17.4s, v0.8h, v31.8h
|
||||
umlsl v9.4s, v8.4h, v30.4h // b - a * src
|
||||
umlsl2 v10.4s, v8.8h, v30.8h
|
||||
mov v0.16b, v1.16b
|
||||
rshrn v16.4h, v16.4s, #9
|
||||
rshrn2 v16.8h, v17.4s, #9
|
||||
@@ -541,10 +536,10 @@ function sgr_finish_weighted2_\bpc\()bpc_neon, export=1
|
||||
uxtl v31.8h, v31.8b
|
||||
uxtl v30.8h, v30.8b
|
||||
.endif
|
||||
umlal v16.4s, v0.4h, v31.4h // b + a * src
|
||||
umlal2 v17.4s, v0.8h, v31.8h
|
||||
umlal v9.4s, v8.4h, v30.4h // b + a * src
|
||||
umlal2 v10.4s, v8.8h, v30.8h
|
||||
umlsl v16.4s, v0.4h, v31.4h // b - a * src
|
||||
umlsl2 v17.4s, v0.8h, v31.8h
|
||||
umlsl v9.4s, v8.4h, v30.4h // b - a * src
|
||||
umlsl2 v10.4s, v8.8h, v30.8h
|
||||
mov v0.16b, v1.16b
|
||||
rshrn v16.4h, v16.4s, #9
|
||||
rshrn2 v16.8h, v17.4s, #9
|
||||
@@ -554,40 +549,30 @@ function sgr_finish_weighted2_\bpc\()bpc_neon, export=1
|
||||
subs w4, w4, #8
|
||||
|
||||
// weighted1
|
||||
shl v31.8h, v31.8h, #4 // u
|
||||
shl v30.8h, v30.8h, #4
|
||||
mov v2.16b, v3.16b
|
||||
|
||||
sub v16.8h, v16.8h, v31.8h // t1 - u
|
||||
sub v9.8h, v9.8h, v30.8h
|
||||
ld1 {v1.8h}, [x3], #16
|
||||
ushll v22.4s, v31.4h, #7 // u << 7
|
||||
ushll2 v23.4s, v31.8h, #7
|
||||
ushll v24.4s, v30.4h, #7
|
||||
ushll2 v25.4s, v30.8h, #7
|
||||
ld1 {v3.8h}, [x8], #16
|
||||
smlal v22.4s, v16.4h, v14.4h // v
|
||||
smlal2 v23.4s, v16.8h, v14.8h
|
||||
smull v22.4s, v16.4h, v14.4h // v
|
||||
smull2 v23.4s, v16.8h, v14.8h
|
||||
mov v16.16b, v18.16b
|
||||
smlal v24.4s, v9.4h, v14.4h
|
||||
smlal2 v25.4s, v9.8h, v14.8h
|
||||
smull v24.4s, v9.4h, v14.4h
|
||||
smull2 v25.4s, v9.8h, v14.8h
|
||||
mov v19.16b, v21.16b
|
||||
.if \bpc == 8
|
||||
rshrn v22.4h, v22.4s, #11
|
||||
rshrn2 v22.8h, v23.4s, #11
|
||||
rshrn v23.4h, v24.4s, #11
|
||||
rshrn2 v23.8h, v25.4s, #11
|
||||
sqxtun v22.8b, v22.8h
|
||||
sqxtun v23.8b, v23.8h
|
||||
usqadd v31.8h, v22.8h
|
||||
usqadd v30.8h, v23.8h
|
||||
.if \bpc == 8
|
||||
sqxtun v22.8b, v31.8h
|
||||
sqxtun v23.8b, v30.8h
|
||||
st1 {v22.8b}, [x0], #8
|
||||
st1 {v23.8b}, [x1], #8
|
||||
.else
|
||||
sqrshrun v22.4h, v22.4s, #11
|
||||
sqrshrun2 v22.8h, v23.4s, #11
|
||||
sqrshrun v23.4h, v24.4s, #11
|
||||
sqrshrun2 v23.8h, v25.4s, #11
|
||||
umin v22.8h, v22.8h, v15.8h
|
||||
umin v23.8h, v23.8h, v15.8h
|
||||
umin v22.8h, v31.8h, v15.8h
|
||||
umin v23.8h, v30.8h, v15.8h
|
||||
st1 {v22.8h}, [x0], #16
|
||||
st1 {v23.8h}, [x1], #16
|
||||
.endif
|
||||
@@ -605,146 +590,114 @@ function sgr_finish_weighted2_\bpc\()bpc_neon, export=1
|
||||
endfunc
|
||||
|
||||
// void dav1d_sgr_weighted2_Xbpc_neon(pixel *dst, const ptrdiff_t stride,
|
||||
// const pixel *src, const ptrdiff_t src_stride,
|
||||
// const int16_t *t1, const int16_t *t2,
|
||||
// const int w, const int h,
|
||||
// const int16_t wt[2], const int bitdepth_max);
|
||||
function sgr_weighted2_\bpc\()bpc_neon, export=1
|
||||
.if \bpc == 8
|
||||
ldr x8, [sp]
|
||||
.else
|
||||
ldp x8, x9, [sp]
|
||||
.endif
|
||||
cmp w7, #2
|
||||
cmp w5, #2
|
||||
add x10, x0, x1
|
||||
add x11, x2, x3
|
||||
add x12, x4, #2*FILTER_OUT_STRIDE
|
||||
add x13, x5, #2*FILTER_OUT_STRIDE
|
||||
ld2r {v30.8h, v31.8h}, [x8] // wt[0], wt[1]
|
||||
add x12, x2, #2*FILTER_OUT_STRIDE
|
||||
add x13, x3, #2*FILTER_OUT_STRIDE
|
||||
ld2r {v30.8h, v31.8h}, [x6] // wt[0], wt[1]
|
||||
.if \bpc == 16
|
||||
dup v29.8h, w9
|
||||
dup v29.8h, w7
|
||||
.endif
|
||||
mov x8, #4*FILTER_OUT_STRIDE
|
||||
lsl x1, x1, #1
|
||||
lsl x3, x3, #1
|
||||
add x9, x6, #7
|
||||
add w9, w4, #7
|
||||
bic x9, x9, #7 // Aligned width
|
||||
.if \bpc == 8
|
||||
sub x1, x1, x9
|
||||
sub x3, x3, x9
|
||||
.else
|
||||
sub x1, x1, x9, lsl #1
|
||||
sub x3, x3, x9, lsl #1
|
||||
.endif
|
||||
sub x8, x8, x9, lsl #1
|
||||
mov w9, w6
|
||||
mov w9, w4
|
||||
b.lt 2f
|
||||
1:
|
||||
.if \bpc == 8
|
||||
ld1 {v0.8b}, [x2], #8
|
||||
ld1 {v16.8b}, [x11], #8
|
||||
ld1 {v0.8b}, [x0]
|
||||
ld1 {v16.8b}, [x10]
|
||||
.else
|
||||
ld1 {v0.8h}, [x2], #16
|
||||
ld1 {v16.8h}, [x11], #16
|
||||
ld1 {v0.8h}, [x0]
|
||||
ld1 {v16.8h}, [x10]
|
||||
.endif
|
||||
ld1 {v1.8h}, [x4], #16
|
||||
ld1 {v1.8h}, [x2], #16
|
||||
ld1 {v17.8h}, [x12], #16
|
||||
ld1 {v2.8h}, [x5], #16
|
||||
ld1 {v2.8h}, [x3], #16
|
||||
ld1 {v18.8h}, [x13], #16
|
||||
subs w6, w6, #8
|
||||
subs w4, w4, #8
|
||||
.if \bpc == 8
|
||||
ushll v0.8h, v0.8b, #4 // u
|
||||
ushll v16.8h, v16.8b, #4 // u
|
||||
.else
|
||||
shl v0.8h, v0.8h, #4 // u
|
||||
shl v16.8h, v16.8h, #4 // u
|
||||
uxtl v0.8h, v0.8b
|
||||
uxtl v16.8h, v16.8b
|
||||
.endif
|
||||
sub v1.8h, v1.8h, v0.8h // t1 - u
|
||||
sub v2.8h, v2.8h, v0.8h // t2 - u
|
||||
sub v17.8h, v17.8h, v16.8h // t1 - u
|
||||
sub v18.8h, v18.8h, v16.8h // t2 - u
|
||||
ushll v3.4s, v0.4h, #7 // u << 7
|
||||
ushll2 v4.4s, v0.8h, #7 // u << 7
|
||||
ushll v19.4s, v16.4h, #7 // u << 7
|
||||
ushll2 v20.4s, v16.8h, #7 // u << 7
|
||||
smlal v3.4s, v1.4h, v30.4h // wt[0] * (t1 - u)
|
||||
smlal v3.4s, v2.4h, v31.4h // wt[1] * (t2 - u)
|
||||
smlal2 v4.4s, v1.8h, v30.8h // wt[0] * (t1 - u)
|
||||
smlal2 v4.4s, v2.8h, v31.8h // wt[1] * (t2 - u)
|
||||
smlal v19.4s, v17.4h, v30.4h // wt[0] * (t1 - u)
|
||||
smlal v19.4s, v18.4h, v31.4h // wt[1] * (t2 - u)
|
||||
smlal2 v20.4s, v17.8h, v30.8h // wt[0] * (t1 - u)
|
||||
smlal2 v20.4s, v18.8h, v31.8h // wt[1] * (t2 - u)
|
||||
.if \bpc == 8
|
||||
smull v3.4s, v1.4h, v30.4h // wt[0] * t1
|
||||
smlal v3.4s, v2.4h, v31.4h // wt[1] * t2
|
||||
smull2 v4.4s, v1.8h, v30.8h // wt[0] * t1
|
||||
smlal2 v4.4s, v2.8h, v31.8h // wt[1] * t2
|
||||
smull v19.4s, v17.4h, v30.4h // wt[0] * t1
|
||||
smlal v19.4s, v18.4h, v31.4h // wt[1] * t2
|
||||
smull2 v20.4s, v17.8h, v30.8h // wt[0] * t1
|
||||
smlal2 v20.4s, v18.8h, v31.8h // wt[1] * t2
|
||||
rshrn v3.4h, v3.4s, #11
|
||||
rshrn2 v3.8h, v4.4s, #11
|
||||
rshrn v19.4h, v19.4s, #11
|
||||
rshrn2 v19.8h, v20.4s, #11
|
||||
sqxtun v3.8b, v3.8h
|
||||
sqxtun v19.8b, v19.8h
|
||||
usqadd v0.8h, v3.8h
|
||||
usqadd v16.8h, v19.8h
|
||||
.if \bpc == 8
|
||||
sqxtun v3.8b, v0.8h
|
||||
sqxtun v19.8b, v16.8h
|
||||
st1 {v3.8b}, [x0], #8
|
||||
st1 {v19.8b}, [x10], #8
|
||||
.else
|
||||
sqrshrun v3.4h, v3.4s, #11
|
||||
sqrshrun2 v3.8h, v4.4s, #11
|
||||
sqrshrun v19.4h, v19.4s, #11
|
||||
sqrshrun2 v19.8h, v20.4s, #11
|
||||
umin v3.8h, v3.8h, v29.8h
|
||||
umin v19.8h, v19.8h, v29.8h
|
||||
umin v3.8h, v0.8h, v29.8h
|
||||
umin v19.8h, v16.8h, v29.8h
|
||||
st1 {v3.8h}, [x0], #16
|
||||
st1 {v19.8h}, [x10], #16
|
||||
.endif
|
||||
b.gt 1b
|
||||
|
||||
subs w7, w7, #2
|
||||
cmp w7, #1
|
||||
subs w5, w5, #2
|
||||
cmp w5, #1
|
||||
b.lt 0f
|
||||
mov w6, w9
|
||||
mov w4, w9
|
||||
add x0, x0, x1
|
||||
add x10, x10, x1
|
||||
add x2, x2, x3
|
||||
add x11, x11, x3
|
||||
add x4, x4, x8
|
||||
add x2, x2, x8
|
||||
add x12, x12, x8
|
||||
add x5, x5, x8
|
||||
add x3, x3, x8
|
||||
add x13, x13, x8
|
||||
b.eq 2f
|
||||
b 1b
|
||||
|
||||
2:
|
||||
.if \bpc == 8
|
||||
ld1 {v0.8b}, [x2], #8
|
||||
ld1 {v0.8b}, [x0]
|
||||
.else
|
||||
ld1 {v0.8h}, [x2], #16
|
||||
ld1 {v0.8h}, [x0]
|
||||
.endif
|
||||
ld1 {v1.8h}, [x4], #16
|
||||
ld1 {v2.8h}, [x5], #16
|
||||
subs w6, w6, #8
|
||||
ld1 {v1.8h}, [x2], #16
|
||||
ld1 {v2.8h}, [x3], #16
|
||||
subs w4, w4, #8
|
||||
.if \bpc == 8
|
||||
ushll v0.8h, v0.8b, #4 // u
|
||||
.else
|
||||
shl v0.8h, v0.8h, #4 // u
|
||||
uxtl v0.8h, v0.8b
|
||||
.endif
|
||||
sub v1.8h, v1.8h, v0.8h // t1 - u
|
||||
sub v2.8h, v2.8h, v0.8h // t2 - u
|
||||
ushll v3.4s, v0.4h, #7 // u << 7
|
||||
ushll2 v4.4s, v0.8h, #7 // u << 7
|
||||
smlal v3.4s, v1.4h, v30.4h // wt[0] * (t1 - u)
|
||||
smlal v3.4s, v2.4h, v31.4h // wt[1] * (t2 - u)
|
||||
smlal2 v4.4s, v1.8h, v30.8h // wt[0] * (t1 - u)
|
||||
smlal2 v4.4s, v2.8h, v31.8h // wt[1] * (t2 - u)
|
||||
.if \bpc == 8
|
||||
smull v3.4s, v1.4h, v30.4h // wt[0] * t1
|
||||
smlal v3.4s, v2.4h, v31.4h // wt[1] * t2
|
||||
smull2 v4.4s, v1.8h, v30.8h // wt[0] * t1
|
||||
smlal2 v4.4s, v2.8h, v31.8h // wt[1] * t2
|
||||
rshrn v3.4h, v3.4s, #11
|
||||
rshrn2 v3.8h, v4.4s, #11
|
||||
sqxtun v3.8b, v3.8h
|
||||
usqadd v0.8h, v3.8h
|
||||
.if \bpc == 8
|
||||
sqxtun v3.8b, v0.8h
|
||||
st1 {v3.8b}, [x0], #8
|
||||
.else
|
||||
sqrshrun v3.4h, v3.4s, #11
|
||||
sqrshrun2 v3.8h, v4.4s, #11
|
||||
umin v3.8h, v3.8h, v29.8h
|
||||
umin v3.8h, v0.8h, v29.8h
|
||||
st1 {v3.8h}, [x0], #16
|
||||
.endif
|
||||
b.gt 1b
|
||||
b.gt 2b
|
||||
0:
|
||||
ret
|
||||
endfunc
|
||||
|
||||
+328
-298
File diff suppressed because it is too large
Load Diff
+376
-307
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+338
-74
@@ -45,34 +45,39 @@ ENABLE_DOTPROD
|
||||
#define LOOP_ALIGN 2
|
||||
|
||||
|
||||
// Lookup table used to help conversion of shifted 32-bit values to 8-bit.
|
||||
.align 4
|
||||
L(hv_tbl_neon_dotprod):
|
||||
const h_tbl_neon_dotprod, align=4
|
||||
// Shuffle indices to permute horizontal samples in preparation for
|
||||
// input to SDOT instructions. The 8-tap horizontal convolution uses
|
||||
// sample indices in the interval of [-3, 4] relative to the current
|
||||
// sample position.
|
||||
.byte 0, 1, 2, 3, 1, 2, 3, 4, 2, 3, 4, 5, 3, 4, 5, 6
|
||||
.byte 4, 5, 6, 7, 5, 6, 7, 8, 6, 7, 8, 9, 7, 8, 9, 10
|
||||
.byte 8, 9, 10, 11, 9, 10, 11, 12, 10, 11, 12, 13, 11, 12, 13, 14
|
||||
|
||||
// Shuffle indices to permute horizontal samples in preparation for
|
||||
// input to USMMLA instructions.
|
||||
#define OFFSET_USMMLA 48
|
||||
.byte 0, 1, 2, 3, 4, 5, 6, 7, 2, 3, 4, 5, 6, 7, 8, 9
|
||||
.byte 4, 5, 6, 7, 8, 9, 10, 11, 6, 7, 8, 9, 10, 11, 12, 13
|
||||
|
||||
// Lookup table used to help conversion of shifted 32-bit values to 8-bit.
|
||||
#define OFFSET_CVT_32_8 80
|
||||
.byte 1, 2, 5, 6, 9, 10, 13, 14, 17, 18, 21, 22, 25, 26, 29, 30
|
||||
endconst
|
||||
|
||||
// Shuffle indices to permute horizontal samples in preparation for input to
|
||||
// SDOT instructions. The 8-tap horizontal convolution uses sample indices in the
|
||||
// interval of [-3, 4] relative to the current sample position. We load samples
|
||||
// from index value -4 to keep loads word aligned, so the shuffle bytes are
|
||||
// translated by 1 to handle this.
|
||||
.align 4
|
||||
L(h_tbl_neon_dotprod):
|
||||
.byte 1, 2, 3, 4, 2, 3, 4, 5, 3, 4, 5, 6, 4, 5, 6, 7
|
||||
.byte 5, 6, 7, 8, 6, 7, 8, 9, 7, 8, 9, 10, 8, 9, 10, 11
|
||||
.byte 9, 10, 11, 12, 10, 11, 12, 13, 11, 12, 13, 14, 12, 13, 14, 15
|
||||
|
||||
// Vertical convolutions are also using SDOT instructions, where a 128-bit
|
||||
// register contains a transposed 4x4 matrix of values. Subsequent iterations of
|
||||
// the vertical convolution can reuse the 3x4 sub-matrix from the previous loop
|
||||
// iteration. These shuffle indices shift and merge this 4x4 matrix with the
|
||||
// values of a new line.
|
||||
.align 4
|
||||
L(v_tbl_neon_dotprod):
|
||||
const v_tbl_neon_dotprod, align=4
|
||||
// Vertical convolutions are also using SDOT instructions, where a
|
||||
// 128-bit register contains a transposed 4x4 matrix of values.
|
||||
// Subsequent iterations of the vertical convolution can reuse the
|
||||
// 3x4 sub-matrix from the previous loop iteration. These shuffle
|
||||
// indices shift and merge this 4x4 matrix with the values of a new
|
||||
// line.
|
||||
.byte 1, 2, 3, 16, 5, 6, 7, 20, 9, 10, 11, 24, 13, 14, 15, 28
|
||||
.byte 1, 2, 3, 16, 5, 6, 7, 17, 9, 10, 11, 18, 13, 14, 15, 19
|
||||
.byte 1, 2, 3, 20, 5, 6, 7, 21, 9, 10, 11, 22, 13, 14, 15, 23
|
||||
.byte 1, 2, 3, 24, 5, 6, 7, 25, 9, 10, 11, 26, 13, 14, 15, 27
|
||||
.byte 1, 2, 3, 28, 5, 6, 7, 29, 9, 10, 11, 30, 13, 14, 15, 31
|
||||
endconst
|
||||
|
||||
|
||||
.macro make_8tap_fn op, type, type_h, type_v, isa, jump=1
|
||||
@@ -111,24 +116,24 @@ function \type\()_8tap_\isa, align=FUNC_ALIGN
|
||||
.align JUMP_ALIGN
|
||||
L(\type\()_8tap_v_\isa):
|
||||
madd \my, \my, w11, w10
|
||||
ldr q6, L(v_tbl_neon_dotprod)
|
||||
movrel x13, v_tbl_neon_dotprod
|
||||
sub \src, \src, \s_strd
|
||||
.ifc \isa, neon_dotprod
|
||||
.ifc \type, prep
|
||||
mov w8, 0x2002 // FILTER_WEIGHT * 128 + rounding
|
||||
mov w8, #0x2002 // FILTER_WEIGHT * 128 + rounding
|
||||
dup v4.4s, w8
|
||||
.else
|
||||
movi v4.4s, #32, lsl 8 // FILTER_WEIGHT * 128, bias for SDOT
|
||||
movi v4.4s, #32, lsl #8 // FILTER_WEIGHT * 128, bias for SDOT
|
||||
.endif
|
||||
.endif
|
||||
ubfx w11, \my, #7, #7
|
||||
and \my, \my, #0x7F
|
||||
ldr q28, L(v_tbl_neon_dotprod) + 16
|
||||
ldp q6, q28, [x13]
|
||||
cmp \h, #4
|
||||
csel \my, \my, w11, le
|
||||
sub \src, \src, \s_strd, lsl #1 // src - s_strd * 3
|
||||
add \xmy, x12, \xmy, lsl #3 // subpel V filter address
|
||||
ldr q29, L(v_tbl_neon_dotprod) + 32
|
||||
ldr q29, [x13, #32]
|
||||
.ifc \isa, neon_dotprod
|
||||
movi v5.16b, #128
|
||||
.endif
|
||||
@@ -139,8 +144,7 @@ L(\type\()_8tap_v_\isa):
|
||||
|
||||
// .align JUMP_ALIGN // fallthrough
|
||||
160: // V - 16xN+
|
||||
ldr q30, L(v_tbl_neon_dotprod) + 48
|
||||
ldr q31, L(v_tbl_neon_dotprod) + 64
|
||||
ldp q30, q31, [x13, #48]
|
||||
.ifc \type, prep
|
||||
add \wd_strd, \w, \w
|
||||
.endif
|
||||
@@ -678,18 +682,19 @@ L(\type\()_8tap_v_\isa):
|
||||
L(\type\()_8tap_h_hv_\isa):
|
||||
madd \mx, \mx, w11, w9
|
||||
madd w14, \my, w11, w10 // for HV
|
||||
ldr q28, L(h_tbl_neon_dotprod)
|
||||
.ifc \isa, neon_dotprod
|
||||
mov w13, 0x2002 // FILTER_WEIGHT * 128 + rounding
|
||||
mov w13, #0x2002 // FILTER_WEIGHT * 128 + rounding
|
||||
dup v27.4s, w13 // put H overrides this
|
||||
.endif
|
||||
sub \src, \src, #4 // src - 4
|
||||
ubfx w9, \mx, #7, #7
|
||||
movrel x13, h_tbl_neon_dotprod
|
||||
sub \src, \src, #3 // src - 3
|
||||
ldr q28, [x13] // for 4-tap & 8-tap H filters
|
||||
ubfx w15, \mx, #7, #7
|
||||
and \mx, \mx, #0x7F
|
||||
ubfx w11, w14, #7, #7 // for HV
|
||||
and w14, w14, #0x7F // for HV
|
||||
cmp \w, #4
|
||||
csel \mx, \mx, w9, le
|
||||
csel \mx, \mx, w15, le
|
||||
add \xmx, x12, \xmx, lsl #3 // subpel H filter address
|
||||
.ifc \isa, neon_dotprod
|
||||
movi v24.16b, #128
|
||||
@@ -699,19 +704,19 @@ L(\type\()_8tap_h_hv_\isa):
|
||||
// HV cases
|
||||
cmp \h, #4
|
||||
csel w14, w14, w11, le
|
||||
sub \src, \src, \s_strd, lsl #1 // src - s_strd * 2 - 4
|
||||
sub \src, \src, \s_strd, lsl #1 // src - s_strd * 2 - 3
|
||||
add \xmy, x12, x14, lsl #3 // subpel V filter address
|
||||
mov x15, x30
|
||||
ldr d7, [\xmy]
|
||||
.ifc \type, put
|
||||
ldr q25, L(hv_tbl_neon_dotprod)
|
||||
.endif
|
||||
ldr q25, [x13, #(OFFSET_CVT_32_8)] // LUT to help conversion
|
||||
.endif // of 32b values to 8b
|
||||
sxtl v7.8h, v7.8b
|
||||
cmp w10, SHARP1
|
||||
cmp w10, #SHARP1
|
||||
b.ne L(\type\()_6tap_hv_\isa) // vertical != SHARP1
|
||||
|
||||
// HV 8-tap cases
|
||||
sub \src, \src, \s_strd // src - s_strd * 3 - 4
|
||||
sub \src, \src, \s_strd // src - s_strd * 3 - 3
|
||||
cmp \w, #4
|
||||
b.eq 40f
|
||||
.ifc \type, put
|
||||
@@ -720,8 +725,7 @@ L(\type\()_8tap_h_hv_\isa):
|
||||
|
||||
// .align JUMP_ALIGN // fallthrough
|
||||
80: // HV8 - 8xN+
|
||||
ldr q29, L(h_tbl_neon_dotprod) + 16
|
||||
ldr q30, L(h_tbl_neon_dotprod) + 32
|
||||
ldp q29, q30, [x13, #16]
|
||||
ldr d26, [\xmx]
|
||||
.ifc \type, prep
|
||||
add \wd_strd, \w, \w
|
||||
@@ -862,7 +866,7 @@ L(\type\()_8tap_h_hv_\isa):
|
||||
|
||||
.align JUMP_ALIGN
|
||||
40: // HV8 - 4xN
|
||||
ldr s26, [\xmx, #2]
|
||||
ldur s26, [\xmx, #2]
|
||||
add \src, \src, #2
|
||||
|
||||
bl L(\type\()_hv_filter4_\isa)
|
||||
@@ -932,7 +936,7 @@ L(\type\()_8tap_h_hv_\isa):
|
||||
.ifc \type, put
|
||||
.align JUMP_ALIGN
|
||||
20: // HV8 - 2xN
|
||||
ldr s26, [\xmx, #2]
|
||||
ldur s26, [\xmx, #2]
|
||||
add \src, \src, #2
|
||||
|
||||
bl L(\type\()_hv_filter4_\isa)
|
||||
@@ -1007,12 +1011,91 @@ L(\type\()_6tap_hv_\isa):
|
||||
|
||||
// .align JUMP_ALIGN // fallthrough
|
||||
80: // HV6 - 8xN+
|
||||
ldr q29, L(h_tbl_neon_dotprod) + 16
|
||||
ldr q30, L(h_tbl_neon_dotprod) + 32
|
||||
ldr d26, [\xmx]
|
||||
.ifc \type, prep
|
||||
add \wd_strd, \w, \w
|
||||
.endif
|
||||
.ifc \isa, neon_i8mm
|
||||
cmp w9, #SHARP1
|
||||
b.eq 88f // horizontal == SHARP1
|
||||
|
||||
ldp q29, q30, [x13, #(OFFSET_USMMLA)]
|
||||
ext v0.8b, v26.8b, v26.8b, #7
|
||||
ins v26.d[1], v0.d[0]
|
||||
|
||||
.align LOOP_ALIGN
|
||||
81:
|
||||
mov \lsrc, \src
|
||||
mov \ldst, \dst
|
||||
mov w8, \h
|
||||
|
||||
bl L(\type\()_hv_filter6_neon_i8mm)
|
||||
srshr v16.8h, v22.8h, #2
|
||||
bl L(\type\()_hv_filter6_neon_i8mm)
|
||||
srshr v17.8h, v22.8h, #2
|
||||
bl L(\type\()_hv_filter6_neon_i8mm)
|
||||
srshr v18.8h, v22.8h, #2
|
||||
bl L(\type\()_hv_filter6_neon_i8mm)
|
||||
srshr v19.8h, v22.8h, #2
|
||||
bl L(\type\()_hv_filter6_neon_i8mm)
|
||||
srshr v20.8h, v22.8h, #2
|
||||
|
||||
.align LOOP_ALIGN
|
||||
8:
|
||||
ld1 {v23.16b}, [\lsrc], \s_strd
|
||||
|
||||
smull v0.4s, v16.4h, v7.h[1]
|
||||
smull2 v1.4s, v16.8h, v7.h[1]
|
||||
mov v16.16b, v17.16b
|
||||
movi v5.4s, #0
|
||||
movi v6.4s, #0
|
||||
tbl v2.16b, {v23.16b}, v29.16b
|
||||
tbl v3.16b, {v23.16b}, v30.16b
|
||||
|
||||
smlal v0.4s, v17.4h, v7.h[2]
|
||||
smlal2 v1.4s, v17.8h, v7.h[2]
|
||||
mov v17.16b, v18.16b
|
||||
|
||||
usmmla v5.4s, v2.16b, v26.16b
|
||||
usmmla v6.4s, v3.16b, v26.16b
|
||||
|
||||
smlal v0.4s, v18.4h, v7.h[3]
|
||||
smlal2 v1.4s, v18.8h, v7.h[3]
|
||||
mov v18.16b, v19.16b
|
||||
subs w8, w8, #1
|
||||
|
||||
smlal v0.4s, v19.4h, v7.h[4]
|
||||
smlal2 v1.4s, v19.8h, v7.h[4]
|
||||
uzp1 v23.8h, v5.8h, v6.8h
|
||||
mov v19.16b, v20.16b
|
||||
|
||||
smlal v0.4s, v20.4h, v7.h[5]
|
||||
smlal2 v1.4s, v20.8h, v7.h[5]
|
||||
srshr v20.8h, v23.8h, #2
|
||||
smlal v0.4s, v20.4h, v7.h[6]
|
||||
smlal2 v1.4s, v20.8h, v7.h[6]
|
||||
.ifc \type, prep
|
||||
rshrn v0.4h, v0.4s, #6
|
||||
rshrn2 v0.8h, v1.4s, #6
|
||||
st1 {v0.8h}, [\ldst], \d_strd
|
||||
b.gt 8b
|
||||
add \dst, \dst, #16
|
||||
.else
|
||||
tbl v0.16b, {v0.16b, v1.16b}, v25.16b
|
||||
sqrshrun v0.8b, v0.8h, #2
|
||||
st1 {v0.8b}, [\ldst], \d_strd
|
||||
b.gt 8b
|
||||
add \dst, \dst, #8
|
||||
.endif
|
||||
add \src, \src, #8
|
||||
subs \w, \w, #8
|
||||
b.gt 81b
|
||||
ret x15
|
||||
|
||||
.align JUMP_ALIGN
|
||||
88:
|
||||
.endif // neon_i8mm
|
||||
ldp q29, q30, [x13, #16]
|
||||
|
||||
.align LOOP_ALIGN
|
||||
81:
|
||||
@@ -1044,8 +1127,8 @@ L(\type\()_6tap_hv_\isa):
|
||||
.endif
|
||||
.align LOOP_ALIGN
|
||||
8:
|
||||
ldr q23, [\xmy]
|
||||
add \xmy, \xmy, \s_strd
|
||||
ldr q23, [\lsrc]
|
||||
add \lsrc, \lsrc, \s_strd
|
||||
|
||||
smull v0.4s, v16.4h, v7.h[1]
|
||||
smull2 v1.4s, v16.8h, v7.h[1]
|
||||
@@ -1132,6 +1215,20 @@ L(\type\()_hv_filter8_\isa):
|
||||
uzp1 v22.8h, v22.8h, v23.8h
|
||||
ret
|
||||
|
||||
.ifc \isa, neon_i8mm
|
||||
.align FUNC_ALIGN
|
||||
L(\type\()_hv_filter6_neon_i8mm):
|
||||
ld1 {v4.16b}, [\lsrc], \s_strd
|
||||
movi v22.4s, #0
|
||||
movi v23.4s, #0
|
||||
tbl v2.16b, {v4.16b}, v29.16b
|
||||
tbl v3.16b, {v4.16b}, v30.16b
|
||||
usmmla v22.4s, v2.16b, v26.16b
|
||||
usmmla v23.4s, v3.16b, v26.16b
|
||||
uzp1 v22.8h, v22.8h, v23.8h
|
||||
ret
|
||||
.endif
|
||||
|
||||
.align FUNC_ALIGN
|
||||
L(\type\()_hv_filter4_\isa):
|
||||
ld1 {v4.8b}, [\src], \s_strd
|
||||
@@ -1147,7 +1244,7 @@ L(\type\()_hv_filter4_\isa):
|
||||
|
||||
.align JUMP_ALIGN
|
||||
40: // HV6 - 4xN
|
||||
ldr s26, [\xmx, #2]
|
||||
ldur s26, [\xmx, #2]
|
||||
add \src, \src, #2
|
||||
|
||||
bl L(\type\()_hv_filter4_\isa)
|
||||
@@ -1208,7 +1305,7 @@ L(\type\()_hv_filter4_\isa):
|
||||
.ifc \type, put
|
||||
.align JUMP_ALIGN
|
||||
20: // HV6 - 2xN
|
||||
ldr s26, [\xmx, #2]
|
||||
ldur s26, [\xmx, #2]
|
||||
add \src, \src, #2
|
||||
|
||||
bl L(\type\()_hv_filter4_\isa)
|
||||
@@ -1268,8 +1365,8 @@ L(\type\()_hv_filter4_\isa):
|
||||
|
||||
.align JUMP_ALIGN
|
||||
L(\type\()_8tap_h_\isa):
|
||||
adr x9, L(\type\()_8tap_h_\isa\()_tbl)
|
||||
ldrh w8, [x9, x8, lsl #1]
|
||||
movrel x11, \type\()_8tap_h_\isa\()_tbl
|
||||
ldrsw x8, [x11, x8, lsl #2]
|
||||
.ifc \type, put
|
||||
.ifc \isa, neon_i8mm
|
||||
movi v27.4s, #34 // special rounding
|
||||
@@ -1278,15 +1375,15 @@ L(\type\()_8tap_h_\isa):
|
||||
dup v27.4s, w10
|
||||
.endif
|
||||
.endif
|
||||
sub x9, x9, x8
|
||||
br x9
|
||||
add x11, x11, x8
|
||||
br x11
|
||||
|
||||
.ifc \type, put
|
||||
.align JUMP_ALIGN
|
||||
20: // H - 2xN
|
||||
AARCH64_VALID_JUMP_TARGET
|
||||
add \src, \src, #2
|
||||
ldr s26, [\xmx, #2]
|
||||
ldur s26, [\xmx, #2]
|
||||
|
||||
.align LOOP_ALIGN
|
||||
2:
|
||||
@@ -1323,7 +1420,7 @@ L(\type\()_8tap_h_\isa):
|
||||
40: // H - 4xN
|
||||
AARCH64_VALID_JUMP_TARGET
|
||||
add \src, \src, #2
|
||||
ldr s26, [\xmx, #2]
|
||||
ldur s26, [\xmx, #2]
|
||||
|
||||
.align LOOP_ALIGN
|
||||
4:
|
||||
@@ -1372,9 +1469,63 @@ L(\type\()_8tap_h_\isa):
|
||||
.align JUMP_ALIGN
|
||||
80: // H - 8xN
|
||||
AARCH64_VALID_JUMP_TARGET
|
||||
ldr q29, L(h_tbl_neon_dotprod) + 16
|
||||
ldr q30, L(h_tbl_neon_dotprod) + 32
|
||||
ldr d26, [\xmx]
|
||||
.ifc \isa, neon_i8mm
|
||||
cmp w9, #SHARP1
|
||||
b.eq 88f // horizontal == SHARP1
|
||||
|
||||
ldp q29, q30, [x13, #(OFFSET_USMMLA)]
|
||||
ext v0.8b, v26.8b, v26.8b, #7
|
||||
ins v26.d[1], v0.d[0]
|
||||
|
||||
.align LOOP_ALIGN
|
||||
8:
|
||||
ldr q0, [\src]
|
||||
ldr q16, [\src, \s_strd]
|
||||
add \src, \src, \s_strd, lsl #1
|
||||
.ifc \type, prep
|
||||
movi v4.4s, #0
|
||||
movi v5.4s, #0
|
||||
movi v20.4s, #0
|
||||
movi v21.4s, #0
|
||||
.else
|
||||
mov v4.16b, v27.16b
|
||||
mov v5.16b, v27.16b
|
||||
mov v20.16b, v27.16b
|
||||
mov v21.16b, v27.16b
|
||||
.endif
|
||||
tbl v1.16b, {v0.16b}, v29.16b
|
||||
tbl v2.16b, {v0.16b}, v30.16b
|
||||
tbl v17.16b, {v16.16b}, v29.16b
|
||||
tbl v18.16b, {v16.16b}, v30.16b
|
||||
|
||||
usmmla v4.4s, v1.16b, v26.16b
|
||||
usmmla v5.4s, v2.16b, v26.16b
|
||||
usmmla v20.4s, v17.16b, v26.16b
|
||||
usmmla v21.4s, v18.16b, v26.16b
|
||||
|
||||
uzp1 v4.8h, v4.8h, v5.8h
|
||||
uzp1 v20.8h, v20.8h, v21.8h
|
||||
.ifc \type, prep
|
||||
srshr v4.8h, v4.8h, #2
|
||||
srshr v20.8h, v20.8h, #2
|
||||
subs \h, \h, #2
|
||||
stp q4, q20, [\dst], #32
|
||||
.else // put
|
||||
sqshrun v4.8b, v4.8h, #6
|
||||
sqshrun v20.8b, v20.8h, #6
|
||||
subs \h, \h, #2
|
||||
str d4, [\dst]
|
||||
str d20, [\dst, \d_strd]
|
||||
add \dst, \dst, \d_strd, lsl #1
|
||||
.endif
|
||||
b.gt 8b
|
||||
ret
|
||||
|
||||
.align JUMP_ALIGN
|
||||
88:
|
||||
.endif // neon_i8mm
|
||||
ldp q29, q30, [x13, #16]
|
||||
|
||||
.align LOOP_ALIGN
|
||||
8:
|
||||
@@ -1438,14 +1589,66 @@ L(\type\()_8tap_h_\isa):
|
||||
.align JUMP_ALIGN
|
||||
160: // H - 16xN
|
||||
AARCH64_VALID_JUMP_TARGET
|
||||
ldr q29, L(h_tbl_neon_dotprod) + 16
|
||||
ldr q30, L(h_tbl_neon_dotprod) + 32
|
||||
ldr d26, [\xmx]
|
||||
.ifc \isa, neon_i8mm
|
||||
cmp w9, #SHARP1
|
||||
b.eq 168f // horizontal == SHARP1
|
||||
|
||||
ldp q29, q30, [x13, #(OFFSET_USMMLA)]
|
||||
ext v0.8b, v26.8b, v26.8b, #7
|
||||
ins v26.d[1], v0.d[0]
|
||||
|
||||
.align LOOP_ALIGN
|
||||
16:
|
||||
ldr q16, [\src]
|
||||
ldr q17, [\src, #12] // avoid 2 register TBL for small cores
|
||||
ldur q17, [\src, #8] // avoid 2 register TBL for small cores
|
||||
add \src, \src, \s_strd
|
||||
.ifc \type, prep
|
||||
movi v6.4s, #0
|
||||
movi v7.4s, #0
|
||||
movi v22.4s, #0
|
||||
movi v23.4s, #0
|
||||
.else
|
||||
mov v6.16b, v27.16b
|
||||
mov v7.16b, v27.16b
|
||||
mov v22.16b, v27.16b
|
||||
mov v23.16b, v27.16b
|
||||
.endif
|
||||
tbl v0.16b, {v16.16b}, v29.16b
|
||||
tbl v1.16b, {v16.16b}, v30.16b
|
||||
tbl v2.16b, {v17.16b}, v29.16b
|
||||
tbl v3.16b, {v17.16b}, v30.16b
|
||||
|
||||
usmmla v6.4s, v0.16b, v26.16b
|
||||
usmmla v7.4s, v1.16b, v26.16b
|
||||
usmmla v22.4s, v2.16b, v26.16b
|
||||
usmmla v23.4s, v3.16b, v26.16b
|
||||
|
||||
uzp1 v6.8h, v6.8h, v7.8h
|
||||
uzp1 v22.8h, v22.8h, v23.8h
|
||||
.ifc \type, prep
|
||||
srshr v6.8h, v6.8h, #2
|
||||
srshr v22.8h, v22.8h, #2
|
||||
subs \h, \h, #1
|
||||
stp q6, q22, [\dst], #32
|
||||
.else // put
|
||||
sqshrun v6.8b, v6.8h, #6
|
||||
sqshrun2 v6.16b, v22.8h, #6
|
||||
subs \h, \h, #1
|
||||
st1 {v6.16b}, [\dst], \d_strd
|
||||
.endif
|
||||
b.gt 16b
|
||||
ret
|
||||
|
||||
.align JUMP_ALIGN
|
||||
168:
|
||||
.endif // neon_i8mm
|
||||
ldp q29, q30, [x13, #16]
|
||||
|
||||
.align LOOP_ALIGN
|
||||
16:
|
||||
ldr q16, [\src]
|
||||
ldur q17, [\src, #12] // avoid 2 register TBL for small cores
|
||||
add \src, \src, \s_strd
|
||||
.ifc \type\()_\isa, prep_neon_i8mm
|
||||
movi v6.4s, #0
|
||||
@@ -1503,8 +1706,6 @@ L(\type\()_8tap_h_\isa):
|
||||
640:
|
||||
1280:
|
||||
AARCH64_VALID_JUMP_TARGET
|
||||
ldr q29, L(h_tbl_neon_dotprod) + 16
|
||||
ldr q30, L(h_tbl_neon_dotprod) + 32
|
||||
ldr d26, [\xmx]
|
||||
.ifc \type, put
|
||||
sub \d_strd, \d_strd, \w, uxtw
|
||||
@@ -1512,10 +1713,73 @@ L(\type\()_8tap_h_\isa):
|
||||
sub \s_strd, \s_strd, \w, uxtw
|
||||
mov w8, \w
|
||||
|
||||
.ifc \isa, neon_i8mm
|
||||
cmp w9, #SHARP1
|
||||
b.eq 328f // horizontal == SHARP1
|
||||
|
||||
ldp q29, q30, [x13, #(OFFSET_USMMLA)]
|
||||
ext v0.8b, v26.8b, v26.8b, #7
|
||||
ins v26.d[1], v0.d[0]
|
||||
|
||||
.align LOOP_ALIGN
|
||||
32:
|
||||
ldr q16, [\src]
|
||||
ldr q17, [\src, #12] // avoid 2 register TBL for small cores
|
||||
ldur q17, [\src, #8] // avoid 2 register TBL for small cores
|
||||
add \src, \src, #16
|
||||
.ifc \type, prep
|
||||
movi v6.4s, #0
|
||||
movi v7.4s, #0
|
||||
movi v22.4s, #0
|
||||
movi v23.4s, #0
|
||||
.else
|
||||
mov v6.16b, v27.16b
|
||||
mov v7.16b, v27.16b
|
||||
mov v22.16b, v27.16b
|
||||
mov v23.16b, v27.16b
|
||||
.endif
|
||||
tbl v0.16b, {v16.16b}, v29.16b
|
||||
tbl v1.16b, {v16.16b}, v30.16b
|
||||
tbl v2.16b, {v17.16b}, v29.16b
|
||||
tbl v3.16b, {v17.16b}, v30.16b
|
||||
|
||||
usmmla v6.4s, v0.16b, v26.16b
|
||||
usmmla v7.4s, v1.16b, v26.16b
|
||||
usmmla v22.4s, v2.16b, v26.16b
|
||||
usmmla v23.4s, v3.16b, v26.16b
|
||||
|
||||
uzp1 v6.8h, v6.8h, v7.8h
|
||||
uzp1 v22.8h, v22.8h, v23.8h
|
||||
.ifc \type, prep
|
||||
srshr v6.8h, v6.8h, #2
|
||||
srshr v22.8h, v22.8h, #2
|
||||
subs w8, w8, #16
|
||||
stp q6, q22, [\dst], #32
|
||||
.else // put
|
||||
sqshrun v6.8b, v6.8h, #6
|
||||
sqshrun2 v6.16b, v22.8h, #6
|
||||
subs w8, w8, #16
|
||||
str q6, [\dst], #16
|
||||
.endif
|
||||
b.gt 32b
|
||||
|
||||
add \src, \src, \s_strd
|
||||
.ifc \type, put
|
||||
add \dst, \dst, \d_strd
|
||||
.endif
|
||||
mov w8, \w
|
||||
subs \h, \h, #1
|
||||
b.gt 32b
|
||||
ret
|
||||
|
||||
.align JUMP_ALIGN
|
||||
328:
|
||||
.endif // neon_i8mm
|
||||
ldp q29, q30, [x13, #16]
|
||||
|
||||
.align LOOP_ALIGN
|
||||
32:
|
||||
ldr q16, [\src]
|
||||
ldur q17, [\src, #12] // avoid 2 register TBL for small cores
|
||||
add \src, \src, #16
|
||||
.ifc \type\()_\isa, prep_neon_i8mm
|
||||
movi v6.4s, #0
|
||||
@@ -1575,19 +1839,19 @@ L(\type\()_8tap_h_\isa):
|
||||
subs \h, \h, #1
|
||||
b.gt 32b
|
||||
ret
|
||||
|
||||
L(\type\()_8tap_h_\isa\()_tbl):
|
||||
.hword (L(\type\()_8tap_h_\isa\()_tbl) - 1280b)
|
||||
.hword (L(\type\()_8tap_h_\isa\()_tbl) - 640b)
|
||||
.hword (L(\type\()_8tap_h_\isa\()_tbl) - 320b)
|
||||
.hword (L(\type\()_8tap_h_\isa\()_tbl) - 160b)
|
||||
.hword (L(\type\()_8tap_h_\isa\()_tbl) - 80b)
|
||||
.hword (L(\type\()_8tap_h_\isa\()_tbl) - 40b)
|
||||
.ifc \type, put
|
||||
.hword (L(\type\()_8tap_h_\isa\()_tbl) - 20b)
|
||||
.hword 0
|
||||
.endif
|
||||
endfunc
|
||||
|
||||
jumptable \type\()_8tap_h_\isa\()_tbl
|
||||
.word 1280b - \type\()_8tap_h_\isa\()_tbl
|
||||
.word 640b - \type\()_8tap_h_\isa\()_tbl
|
||||
.word 320b - \type\()_8tap_h_\isa\()_tbl
|
||||
.word 160b - \type\()_8tap_h_\isa\()_tbl
|
||||
.word 80b - \type\()_8tap_h_\isa\()_tbl
|
||||
.word 40b - \type\()_8tap_h_\isa\()_tbl
|
||||
.ifc \type, put
|
||||
.word 20b - \type\()_8tap_h_\isa\()_tbl
|
||||
.endif
|
||||
endjumptable
|
||||
.endm
|
||||
|
||||
// dst(x0), d_strd(x7), src(x1), s_strd(x2), w(w3), h(w4), mx(w5), my(w6)
|
||||
|
||||
+3
-3
@@ -250,7 +250,7 @@ function msac_decode_symbol_adapt4_neon, export=1
|
||||
ret
|
||||
1:
|
||||
lsr w15, w15, #4
|
||||
b L(refill)
|
||||
b L(refill)
|
||||
.elseif \n == 8
|
||||
ldr w6, [x0, #CNT]
|
||||
tbl v30.8b, {v30.16b}, v31.8b
|
||||
@@ -275,7 +275,7 @@ function msac_decode_symbol_adapt4_neon, export=1
|
||||
1:
|
||||
lsr w15, w15, #1 // ret
|
||||
mov x7, v28.d[0]
|
||||
b L(refill)
|
||||
b L(refill)
|
||||
.elseif \n == 16
|
||||
add x8, sp, w15, sxtw #1
|
||||
ldrh w3, [x8, #48] // v
|
||||
@@ -298,7 +298,7 @@ function msac_decode_symbol_adapt4_neon, export=1
|
||||
ret
|
||||
1:
|
||||
add w15, w15, #\n // ret
|
||||
b L(refill)
|
||||
b L(refill)
|
||||
.endif
|
||||
.endm
|
||||
|
||||
|
||||
+323
-69
@@ -25,22 +25,25 @@
|
||||
* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
#include "src/arm/asm-offsets.h"
|
||||
#include "src/arm/asm.S"
|
||||
#include "util.S"
|
||||
|
||||
#define INVALID_MV 0x80008000
|
||||
|
||||
// void dav1d_splat_mv_neon(refmvs_block **rr, const refmvs_block *rmv,
|
||||
// int bx4, int bw4, int bh4)
|
||||
|
||||
function splat_mv_neon, export=1
|
||||
ld1 {v3.16b}, [x1]
|
||||
clz w3, w3
|
||||
adr x5, L(splat_tbl)
|
||||
movrel x5, splat_tbl
|
||||
sub w3, w3, #26
|
||||
ext v2.16b, v3.16b, v3.16b, #12
|
||||
ldrh w3, [x5, w3, uxtw #1]
|
||||
ldrsw x3, [x5, w3, uxtw #2]
|
||||
add w2, w2, w2, lsl #1
|
||||
ext v0.16b, v2.16b, v3.16b, #4
|
||||
sub x3, x5, w3, uxtw
|
||||
add x3, x5, x3
|
||||
ext v1.16b, v2.16b, v3.16b, #8
|
||||
lsl w2, w2, #2
|
||||
ext v2.16b, v2.16b, v3.16b, #12
|
||||
@@ -80,16 +83,17 @@ function splat_mv_neon, export=1
|
||||
st1 {v0.16b, v1.16b, v2.16b}, [x1]
|
||||
b.gt 1b
|
||||
ret
|
||||
|
||||
L(splat_tbl):
|
||||
.hword L(splat_tbl) - 320b
|
||||
.hword L(splat_tbl) - 160b
|
||||
.hword L(splat_tbl) - 80b
|
||||
.hword L(splat_tbl) - 40b
|
||||
.hword L(splat_tbl) - 20b
|
||||
.hword L(splat_tbl) - 10b
|
||||
endfunc
|
||||
|
||||
jumptable splat_tbl
|
||||
.word 320b - splat_tbl
|
||||
.word 160b - splat_tbl
|
||||
.word 80b - splat_tbl
|
||||
.word 40b - splat_tbl
|
||||
.word 20b - splat_tbl
|
||||
.word 10b - splat_tbl
|
||||
endjumptable
|
||||
|
||||
const mv_tbls, align=4
|
||||
.byte 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255
|
||||
.byte 0, 1, 2, 3, 8, 0, 1, 2, 3, 8, 0, 1, 2, 3, 8, 0
|
||||
@@ -112,7 +116,7 @@ function save_tmvs_neon, export=1
|
||||
|
||||
movi v30.8b, #0
|
||||
ld1 {v31.8b}, [x3]
|
||||
adr x8, L(save_tmvs_tbl)
|
||||
movrel x8, save_tmvs_tbl
|
||||
movrel x16, mask_mult
|
||||
movrel x13, mv_tbls
|
||||
ld1 {v29.8b}, [x16]
|
||||
@@ -129,17 +133,17 @@ function save_tmvs_neon, export=1
|
||||
and w9, w7, #30 // (y & 15) * 2
|
||||
ldr x9, [x2, w9, uxtw #3] // b = rr[(y & 15) * 2]
|
||||
add x9, x9, #12 // &b[... + 1]
|
||||
madd x10, x4, x14, x9 // end_cand_b = &b[col_end8*2 + 1]
|
||||
madd x9, x6, x14, x9 // cand_b = &b[x*2 + 1]
|
||||
madd x10, x4, x14, x9 // end_cand_b = &b[col_end8*2 + 1]
|
||||
madd x9, x6, x14, x9 // cand_b = &b[x*2 + 1]
|
||||
|
||||
madd x3, x6, x15, x0 // &rp[x]
|
||||
madd x3, x6, x15, x0 // &rp[x]
|
||||
|
||||
2:
|
||||
ldrb w11, [x9, #10] // cand_b->bs
|
||||
ld1 {v0.16b}, [x9] // cand_b->mv
|
||||
add x11, x8, w11, uxtw #2
|
||||
add x11, x8, w11, uxtw #3
|
||||
ldr h1, [x9, #8] // cand_b->ref
|
||||
ldrh w12, [x11] // bw8
|
||||
ldr w12, [x11] // bw8
|
||||
mov x15, x8
|
||||
add x9, x9, w12, uxtw #1 // cand_b += bw8*2
|
||||
cmp x9, x10
|
||||
@@ -149,9 +153,9 @@ function save_tmvs_neon, export=1
|
||||
ldrb w15, [x9, #10] // cand_b->bs
|
||||
add x16, x9, #8
|
||||
ld1 {v4.16b}, [x9] // cand_b->mv
|
||||
add x15, x8, w15, uxtw #2
|
||||
add x15, x8, w15, uxtw #3
|
||||
ld1 {v1.h}[1], [x16] // cand_b->ref
|
||||
ldrh w12, [x15] // bw8
|
||||
ldr w12, [x15] // bw8
|
||||
add x9, x9, w12, uxtw #1 // cand_b += bw8*2
|
||||
trn1 v2.2d, v0.2d, v4.2d
|
||||
|
||||
@@ -166,12 +170,12 @@ function save_tmvs_neon, export=1
|
||||
addp v1.4h, v1.4h, v1.4h // Combine condition for [1] and [0]
|
||||
umov w16, v1.h[0] // Extract case for first block
|
||||
umov w17, v1.h[1]
|
||||
ldrh w11, [x11, #2] // Fetch jump table entry
|
||||
ldrh w15, [x15, #2]
|
||||
ldrsw x11, [x11, #4] // Fetch jump table entry
|
||||
ldrsw x15, [x15, #4]
|
||||
ldr q1, [x13, w16, uxtw #4] // Load permutation table base on case
|
||||
ldr q5, [x13, w17, uxtw #4]
|
||||
sub x11, x8, w11, uxtw // Find jump table target
|
||||
sub x15, x8, w15, uxtw
|
||||
add x11, x8, x11 // Find jump table target
|
||||
add x15, x8, x15
|
||||
tbl v0.16b, {v0.16b}, v1.16b // Permute cand_b to output refmvs_temporal_block
|
||||
tbl v4.16b, {v4.16b}, v5.16b
|
||||
|
||||
@@ -243,50 +247,300 @@ function save_tmvs_neon, export=1
|
||||
str q2, [x3, #(16*5-16)]
|
||||
add x3, x3, #16*5
|
||||
ret
|
||||
|
||||
L(save_tmvs_tbl):
|
||||
.hword 16 * 12
|
||||
.hword L(save_tmvs_tbl) - 160b
|
||||
.hword 16 * 12
|
||||
.hword L(save_tmvs_tbl) - 160b
|
||||
.hword 8 * 12
|
||||
.hword L(save_tmvs_tbl) - 80b
|
||||
.hword 8 * 12
|
||||
.hword L(save_tmvs_tbl) - 80b
|
||||
.hword 8 * 12
|
||||
.hword L(save_tmvs_tbl) - 80b
|
||||
.hword 8 * 12
|
||||
.hword L(save_tmvs_tbl) - 80b
|
||||
.hword 4 * 12
|
||||
.hword L(save_tmvs_tbl) - 40b
|
||||
.hword 4 * 12
|
||||
.hword L(save_tmvs_tbl) - 40b
|
||||
.hword 4 * 12
|
||||
.hword L(save_tmvs_tbl) - 40b
|
||||
.hword 4 * 12
|
||||
.hword L(save_tmvs_tbl) - 40b
|
||||
.hword 2 * 12
|
||||
.hword L(save_tmvs_tbl) - 20b
|
||||
.hword 2 * 12
|
||||
.hword L(save_tmvs_tbl) - 20b
|
||||
.hword 2 * 12
|
||||
.hword L(save_tmvs_tbl) - 20b
|
||||
.hword 2 * 12
|
||||
.hword L(save_tmvs_tbl) - 20b
|
||||
.hword 2 * 12
|
||||
.hword L(save_tmvs_tbl) - 20b
|
||||
.hword 1 * 12
|
||||
.hword L(save_tmvs_tbl) - 10b
|
||||
.hword 1 * 12
|
||||
.hword L(save_tmvs_tbl) - 10b
|
||||
.hword 1 * 12
|
||||
.hword L(save_tmvs_tbl) - 10b
|
||||
.hword 1 * 12
|
||||
.hword L(save_tmvs_tbl) - 10b
|
||||
.hword 1 * 12
|
||||
.hword L(save_tmvs_tbl) - 10b
|
||||
.hword 1 * 12
|
||||
.hword L(save_tmvs_tbl) - 10b
|
||||
.hword 1 * 12
|
||||
.hword L(save_tmvs_tbl) - 10b
|
||||
endfunc
|
||||
|
||||
jumptable save_tmvs_tbl
|
||||
.word 16 * 12
|
||||
.word 160b - save_tmvs_tbl
|
||||
.word 16 * 12
|
||||
.word 160b - save_tmvs_tbl
|
||||
.word 8 * 12
|
||||
.word 80b - save_tmvs_tbl
|
||||
.word 8 * 12
|
||||
.word 80b - save_tmvs_tbl
|
||||
.word 8 * 12
|
||||
.word 80b - save_tmvs_tbl
|
||||
.word 8 * 12
|
||||
.word 80b - save_tmvs_tbl
|
||||
.word 4 * 12
|
||||
.word 40b - save_tmvs_tbl
|
||||
.word 4 * 12
|
||||
.word 40b - save_tmvs_tbl
|
||||
.word 4 * 12
|
||||
.word 40b - save_tmvs_tbl
|
||||
.word 4 * 12
|
||||
.word 40b - save_tmvs_tbl
|
||||
.word 2 * 12
|
||||
.word 20b - save_tmvs_tbl
|
||||
.word 2 * 12
|
||||
.word 20b - save_tmvs_tbl
|
||||
.word 2 * 12
|
||||
.word 20b - save_tmvs_tbl
|
||||
.word 2 * 12
|
||||
.word 20b - save_tmvs_tbl
|
||||
.word 2 * 12
|
||||
.word 20b - save_tmvs_tbl
|
||||
.word 1 * 12
|
||||
.word 10b - save_tmvs_tbl
|
||||
.word 1 * 12
|
||||
.word 10b - save_tmvs_tbl
|
||||
.word 1 * 12
|
||||
.word 10b - save_tmvs_tbl
|
||||
.word 1 * 12
|
||||
.word 10b - save_tmvs_tbl
|
||||
.word 1 * 12
|
||||
.word 10b - save_tmvs_tbl
|
||||
.word 1 * 12
|
||||
.word 10b - save_tmvs_tbl
|
||||
.word 1 * 12
|
||||
.word 10b - save_tmvs_tbl
|
||||
endjumptable
|
||||
|
||||
// void dav1d_load_tmvs_neon(const refmvs_frame *const rf, int tile_row_idx,
|
||||
// const int col_start8, const int col_end8,
|
||||
// const int row_start8, int row_end8)
|
||||
function load_tmvs_neon, export=1
|
||||
rf .req x0
|
||||
tile_row_idx .req w1
|
||||
col_start8 .req w2
|
||||
col_end8 .req w3
|
||||
row_start8 .req w4
|
||||
row_end8 .req w5
|
||||
col_start8i .req w6
|
||||
col_end8i .req w7
|
||||
rp_proj .req x8
|
||||
stride5 .req x9
|
||||
wstride5 .req w9
|
||||
stp x28, x27, [sp, #-96]!
|
||||
stp x26, x25, [sp, #16]
|
||||
stp x24, x23, [sp, #32]
|
||||
stp x22, x21, [sp, #48]
|
||||
stp x20, x19, [sp, #64]
|
||||
stp x29, x30, [sp, #80]
|
||||
|
||||
ldr w15, [rf, #RMVSF_N_TILE_THREADS]
|
||||
ldp w16, w17, [rf, #RMVSF_IW8] // include rf->ih8 too
|
||||
sub col_start8i, col_start8, #8 // col_start8 - 8
|
||||
add col_end8i, col_end8, #8 // col_end8 + 8
|
||||
ldr wstride5, [rf, #RMVSF_RP_STRIDE]
|
||||
ldr rp_proj, [rf, #RMVSF_RP_PROJ]
|
||||
|
||||
cmp w15, #1
|
||||
csel tile_row_idx, wzr, tile_row_idx, eq // if (rf->n_tile_threads == 1) tile_row_idx = 0
|
||||
|
||||
bic col_start8i, col_start8i, col_start8i, asr #31 // imax(col_start8 - 8, 0)
|
||||
cmp col_end8i, w16
|
||||
csel col_end8i, col_end8i, w16, lt // imin(col_end8 + 8, rf->iw8)
|
||||
|
||||
lsl tile_row_idx, tile_row_idx, #4 // 16 * tile_row_idx
|
||||
|
||||
cmp row_end8, w17
|
||||
csel row_end8, row_end8, w17, lt // imin(row_end8, rf->ih8)
|
||||
|
||||
add wstride5, wstride5, wstride5, lsl #2 // stride * sizeof(refmvs_temporal_block)
|
||||
and w15, row_start8, #15 // row_start8 & 15
|
||||
add w10, col_start8, col_start8, lsl #2 // col_start8 * sizeof(refmvs_temporal_block)
|
||||
smaddl rp_proj, tile_row_idx, wstride5, rp_proj // &rf->rp_proj[16 * stride * tile_row_idx]
|
||||
smaddl x10, w15, wstride5, x10 // ((row_start8 & 15) * stride + col_start8) * sizeof(refmvs_temporal_block)
|
||||
mov w15, #INVALID_MV
|
||||
sub w11, col_end8, col_start8 // xfill loop count
|
||||
add x10, x10, rp_proj // &rf->rp_proj[16 * stride * tile_row_idx + (row_start8 & 15) * stride + col_start8]
|
||||
add x15, x15, x15, lsl #40 // first 64b of 4 [INVALID_MV, 0]... patterns
|
||||
mov w17, #(INVALID_MV >> 8) // last 32b of 4 patterns
|
||||
sub w12, row_end8, row_start8 // yfill loop count
|
||||
ror x16, x15, #48 // second 64b of 4 patterns
|
||||
ldr w19, [rf, #RMVSF_N_MFMVS]
|
||||
|
||||
5: // yfill loop
|
||||
and w13, w11, #-4 // xfill 4x count by patterns
|
||||
mov x14, x10 // fill_ptr = row_ptr
|
||||
add x10, x10, stride5 // row_ptr += stride
|
||||
sub w12, w12, #1 // y--
|
||||
|
||||
cbz w13, 3f
|
||||
|
||||
4: // xfill loop 4x
|
||||
sub w13, w13, #4 // xfill 4x count -= 4
|
||||
stp x15, x16, [x14]
|
||||
str w17, [x14, #16]
|
||||
add x14, x14, #20 // fill_ptr += 4 * sizeof(refmvs_temporal_block)
|
||||
cbnz w13, 4b
|
||||
|
||||
3: // up to 3 residuals
|
||||
tbz w11, #1, 1f
|
||||
str x15, [x14]
|
||||
strh w16, [x14, #8]
|
||||
add x14, x14, #10 // fill_ptr += 2 * sizeof(refmvs_temporal_block)
|
||||
|
||||
1: // up to 1 residual
|
||||
tbz w11, #0, 2f
|
||||
str w15, [x14]
|
||||
2:
|
||||
cbnz w12, 5b // yfill loop
|
||||
|
||||
cbz w19, 11f // if (!rf->n_mfmvs) skip nloop
|
||||
|
||||
add x29, rf, #RMVSF_MFMV_REF2CUR
|
||||
mov w10, #0 // n = 0
|
||||
movi v3.2s, #255 // 0x3FFF >> 6, for MV clamp
|
||||
movrel x1, div_mult_tbl
|
||||
|
||||
10: // nloop
|
||||
ldrsb w16, [x29, x10] // ref2cur = rf->mfmv_ref2cur[n]
|
||||
cmp w16, #-32
|
||||
b.eq 9f // if (ref2cur == INVALID_REF2CUR) continue
|
||||
|
||||
add x17, x10, #(RMVSF_MFMV_REF - RMVSF_MFMV_REF2CUR) // n - (&rf->mfmv_ref - &rf->mfmv_ref2cur)
|
||||
mov x20, #4
|
||||
ldrb w17, [x29, x17] // ref = rf->mfmv_ref[n]
|
||||
ldr x13, [x29, #(RMVSF_RP_REF - RMVSF_MFMV_REF2CUR)]
|
||||
sub x21, x10, x10, lsl #3 // -(n * 7)
|
||||
smaddl x20, row_start8, wstride5, x20 // row_start8 * stride * sizeof(refmvs_temporal_block) + 4
|
||||
mov w12, row_start8 // y = row_start8
|
||||
add x28, x29, #(RMVSF_MFMV_REF2REF - RMVSF_MFMV_REF2CUR - 1) // &rf->mfmv_ref2ref - 1
|
||||
ldr x13, [x13, x17, lsl #3] // rf->rp_ref[ref]
|
||||
sub x28, x28, x21 // rf->mfmv_ref2ref[n] - 1
|
||||
sub w17, w17, #4 // ref_sign = ref - 4
|
||||
add x13, x13, x20 // r = &rf->rp_ref[ref][row_start8 * stride].ref
|
||||
dup v0.2s, w17 // ref_sign
|
||||
|
||||
5: // yloop
|
||||
and w14, w12, #-8 // y_sb_align = y & ~7
|
||||
mov w11, col_start8i // x = col_start8i
|
||||
add w15, w14, #8 // y_sb_align + 8
|
||||
cmp w14, row_start8
|
||||
csel w14, w14, row_start8, gt // imax(y_sb_align, row_start8)
|
||||
cmp w15, row_end8
|
||||
csel w15, w15, row_end8, lt // imin(y_sb_align + 8, row_end8)
|
||||
|
||||
4: // xloop
|
||||
add x23, x13, x11, lsl #2 // partial &r[x] address
|
||||
ldrb w22, [x23, x11] // b_ref = rb->ref
|
||||
cbz w22, 6f // if (!b_ref) continue
|
||||
|
||||
ldrb w24, [x28, x22] // ref2ref = rf->mfmv_ref2ref[n][b_ref - 1]
|
||||
cbz w24, 6f // if (!ref2ref) continue
|
||||
|
||||
ldrh w20, [x1, x24, lsl #1] // div_mult[ref2ref]
|
||||
add x23, x23, x11 // &r[x]
|
||||
mul w20, w20, w16 // frac = ref2cur * div_mult[ref2ref]
|
||||
|
||||
ldur s1, [x23, #-4] // mv{y, x} = rb->mv
|
||||
fmov s2, w20 // frac
|
||||
sxtl v1.4s, v1.4h
|
||||
mul v1.2s, v1.2s, v2.s[0] // offset{y, x} = frac * mv{y, x}
|
||||
|
||||
ssra v1.2s, v1.2s, #31 // offset{y, x} + (offset{y, x} >> 31)
|
||||
ldur w25, [x23, #-4] // b_mv = rb->mv
|
||||
srshr v1.2s, v1.2s, #14 // (offset{y, x} + (offset{y, x} >> 31) + 8192) >> 14
|
||||
|
||||
abs v2.2s, v1.2s // abs(offset{y, x})
|
||||
eor v1.8b, v1.8b, v0.8b // offset{y, x} ^ ref_sign
|
||||
|
||||
sshr v2.2s, v2.2s, #6 // abs(offset{y, x}) >> 6
|
||||
cmlt v1.2s, v1.2s, #0 // sign(offset{y, x} ^ ref_sign): -1 or 0
|
||||
umin v2.2s, v2.2s, v3.2s // iclip(abs(offset{y, x}) >> 6, 0, 0x3FFF >> 6)
|
||||
|
||||
neg v4.2s, v2.2s
|
||||
bsl v1.8b, v4.8b, v2.8b // apply_sign(iclip(abs(offset{y, x}) >> 6, 0, 0x3FFF >> 6))
|
||||
fmov x20, d1 // offset{y, x}
|
||||
|
||||
add w21, w12, w20 // pos_y = y + offset.y
|
||||
cmp w21, w14 // pos_y >= y_proj_start
|
||||
b.lt 1f
|
||||
cmp w21, w15 // pos_y < y_proj_end
|
||||
b.ge 1f
|
||||
add x26, x11, x20, asr #32 // pos_x = x + offset.x
|
||||
and w27, w21, #15 // pos_y & 15
|
||||
add x21, x26, x26, lsl #2 // pos_x * sizeof(refmvs_temporal_block)
|
||||
umaddl x27, w27, wstride5, rp_proj // &rp_proj[(pos_y & 15) * stride]
|
||||
add x27, x27, x21 // &rp_proj[(pos_y & 15) * stride + pos_x]
|
||||
|
||||
3: // copy loop
|
||||
and w20, w11, #-8 // x_sb_align = x & ~7
|
||||
sub w21, w20, #8 // x_sb_align - 8
|
||||
cmp w21, col_start8
|
||||
csel w21, w21, col_start8, gt // imax(x_sb_align - 8, col_start8)
|
||||
cmp w26, w21 // pos_x >= imax(x_sb_align - 8, col_start8)
|
||||
b.lt 2f
|
||||
add w20, w20, #16 // x_sb_align + 16
|
||||
cmp w20, col_end8
|
||||
csel w20, w20, col_end8, lt // imin(x_sb_align + 16, col_end8)
|
||||
cmp w26, w20 // pos_x < imin(x_sb_align + 16, col_end8)
|
||||
b.ge 2f
|
||||
str w25, [x27] // rp_proj[pos + pos_x].mv = rb->mv (b_mv)
|
||||
strb w24, [x27, #4] // rp_proj[pos + pos_x].ref = ref2ref
|
||||
|
||||
2: // search part of copy loop
|
||||
add w11, w11, #1 // x++
|
||||
cmp w11, col_end8i // if (++x >= col_end8i) break xloop
|
||||
b.ge 8f
|
||||
|
||||
ldrb w20, [x23, #5]! // rb++; rb->ref
|
||||
cmp w20, w22 // if (rb->ref != b_ref) break
|
||||
b.ne 7f
|
||||
|
||||
ldur w21, [x23, #-4] // rb->mv.n
|
||||
cmp w21, w25 // if (rb->mv.n != b_mv.n) break
|
||||
b.ne 7f
|
||||
|
||||
add w26, w26, #1 // pos_x++
|
||||
add x27, x27, #5 // advance &rp_proj[(pos_y & 15) * stride + pos_x]
|
||||
b 3b // copy loop
|
||||
|
||||
1: // search loop
|
||||
add w11, w11, #1 // x++
|
||||
cmp w11, col_end8i // if (++x >= col_end8i) break xloop
|
||||
b.ge 8f
|
||||
|
||||
ldrb w20, [x23, #5]! // rb++; rb->ref
|
||||
cmp w20, w22 // if (rb->ref != b_ref) break
|
||||
b.ne 7f
|
||||
|
||||
ldur w21, [x23, #-4] // rb->mv.n
|
||||
cmp w21, w25 // if (rb->mv.n == b_mv.n) continue
|
||||
b.eq 1b // search loop
|
||||
7:
|
||||
cmp w11, col_end8i // x < col_end8i
|
||||
b.lt 4b // xloop
|
||||
|
||||
6: // continue case of xloop
|
||||
add w11, w11, #1 // x++
|
||||
cmp w11, col_end8i // x < col_end8i
|
||||
b.lt 4b // xloop
|
||||
8:
|
||||
add w12, w12, #1 // y++
|
||||
add x13, x13, stride5 // r += stride
|
||||
cmp w12, row_end8 // y < row_end8
|
||||
b.lt 5b // yloop
|
||||
9:
|
||||
add w10, w10, #1
|
||||
cmp w10, w19 // n < rf->n_mfmvs
|
||||
b.lt 10b // nloop
|
||||
11:
|
||||
ldp x29, x30, [sp, #80]
|
||||
ldp x20, x19, [sp, #64]
|
||||
ldp x22, x21, [sp, #48]
|
||||
ldp x24, x23, [sp, #32]
|
||||
ldp x26, x25, [sp, #16]
|
||||
ldp x28, x27, [sp], #96
|
||||
ret
|
||||
.unreq rf
|
||||
.unreq tile_row_idx
|
||||
.unreq col_start8
|
||||
.unreq col_end8
|
||||
.unreq row_start8
|
||||
.unreq row_end8
|
||||
.unreq col_start8i
|
||||
.unreq col_end8i
|
||||
.unreq rp_proj
|
||||
.unreq stride5
|
||||
.unreq wstride5
|
||||
endfunc
|
||||
|
||||
const div_mult_tbl
|
||||
.hword 0, 16384, 8192, 5461, 4096, 3276, 2730, 2340
|
||||
.hword 2048, 1820, 1638, 1489, 1365, 1260, 1170, 1092
|
||||
.hword 1024, 963, 910, 862, 819, 780, 744, 712
|
||||
.hword 682, 655, 630, 606, 585, 564, 546, 528
|
||||
endconst
|
||||
|
||||
+1
-1
@@ -72,7 +72,7 @@
|
||||
.if \space > 8192
|
||||
// Here, we'd need to touch two (or more) pages while decrementing
|
||||
// the stack pointer.
|
||||
.error "sub_sp_align doesn't support values over 8K at the moment"
|
||||
.error "sub_sp doesn't support values over 8K at the moment"
|
||||
.elseif \space > 4096
|
||||
sub x16, sp, #4096
|
||||
ldr xzr, [x16]
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
/*
|
||||
* Copyright © 2024, VideoLAN and dav1d authors
|
||||
* All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
||||
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
* WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
|
||||
* ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
|
||||
* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
|
||||
* ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
|
||||
* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
#ifndef ARM_ARM_ARCH_H
|
||||
#define ARM_ARM_ARCH_H
|
||||
|
||||
/* Compatibility header to define __ARM_ARCH with older compilers */
|
||||
#ifndef __ARM_ARCH
|
||||
|
||||
#ifdef _M_ARM
|
||||
#define __ARM_ARCH _M_ARM
|
||||
|
||||
#elif defined(__ARM_ARCH_8A__) || defined(_M_ARM64)
|
||||
#define __ARM_ARCH 8
|
||||
|
||||
#elif defined(__ARM_ARCH_7__) || defined(__ARM_ARCH_7A__) || \
|
||||
defined(__ARM_ARCH_7EM__) || defined(__ARM_ARCH_7R__) || \
|
||||
defined(__ARM_ARCH_7M__) || defined(__ARM_ARCH_7S__)
|
||||
#define __ARM_ARCH 7
|
||||
|
||||
#elif defined(__ARM_ARCH_6__) || defined(__ARM_ARCH_6J__) || \
|
||||
defined(__ARM_ARCH_6K__) || defined(__ARM_ARCH_6T2__) || \
|
||||
defined(__ARM_ARCH_6Z__) || defined(__ARM_ARCH_6ZK__)
|
||||
#define __ARM_ARCH 6
|
||||
|
||||
#elif defined(__ARM_ARCH_5__) || defined(__ARM_ARCH_5T__) || \
|
||||
defined(__ARM_ARCH_5E__) || defined(__ARM_ARCH_5TE__)
|
||||
#define __ARM_ARCH 5
|
||||
|
||||
#elif defined(__ARM_ARCH_4__) || defined(__ARM_ARCH_4T__)
|
||||
#define __ARM_ARCH 4
|
||||
|
||||
#elif defined(__ARM_ARCH_3__) || defined(__ARM_ARCH_3M__)
|
||||
#define __ARM_ARCH 3
|
||||
|
||||
#elif defined(__ARM_ARCH_2__)
|
||||
#define __ARM_ARCH 2
|
||||
|
||||
#else
|
||||
#error Unknown ARM architecture version
|
||||
#endif
|
||||
|
||||
#endif /* !__ARM_ARCH */
|
||||
|
||||
#endif /* ARM_ARM_ARCH_H */
|
||||
@@ -27,6 +27,8 @@
|
||||
#ifndef ARM_ASM_OFFSETS_H
|
||||
#define ARM_ASM_OFFSETS_H
|
||||
|
||||
#include "config.h"
|
||||
|
||||
#define FGD_SEED 0
|
||||
#define FGD_AR_COEFF_LAG 92
|
||||
#define FGD_AR_COEFFS_Y 96
|
||||
@@ -40,4 +42,17 @@
|
||||
#define FGD_UV_OFFSET 204
|
||||
#define FGD_CLIP_TO_RESTRICTED_RANGE 216
|
||||
|
||||
#if ARCH_AARCH64
|
||||
#define RMVSF_IW8 16
|
||||
#define RMVSF_IH8 20
|
||||
#define RMVSF_MFMV_REF 53
|
||||
#define RMVSF_MFMV_REF2CUR 56
|
||||
#define RMVSF_MFMV_REF2REF 59
|
||||
#define RMVSF_N_MFMVS 80
|
||||
#define RMVSF_RP_REF 96
|
||||
#define RMVSF_RP_PROJ 104
|
||||
#define RMVSF_RP_STRIDE 112
|
||||
#define RMVSF_N_TILE_THREADS 128
|
||||
#endif
|
||||
|
||||
#endif /* ARM_ASM_OFFSETS_H */
|
||||
|
||||
+36
-4
@@ -147,7 +147,7 @@ DISABLE_SVE2
|
||||
*
|
||||
* References:
|
||||
* - "ELF for the Arm® 64-bit Architecture"
|
||||
* https: *github.com/ARM-software/abi-aa/blob/master/aaelf64/aaelf64.rst
|
||||
* https://github.com/ARM-software/abi-aa/blob/master/aaelf64/aaelf64.rst
|
||||
* - "Providing protection for complex software"
|
||||
* https://developer.arm.com/architectures/learn-the-architecture/providing-protection-for-complex-software
|
||||
*/
|
||||
@@ -193,8 +193,14 @@ DISABLE_SVE2
|
||||
|
||||
#endif /* !__ARM_FEATURE_PAC_DEFAULT */
|
||||
|
||||
#if defined(__ARM_FEATURE_GCS_DEFAULT) && __ARM_FEATURE_GCS_DEFAULT == 1
|
||||
#define GNU_PROPERTY_AARCH64_GCS (1 << 2)
|
||||
#else
|
||||
#define GNU_PROPERTY_AARCH64_GCS 0 /* No GCS */
|
||||
#endif
|
||||
|
||||
#if (GNU_PROPERTY_AARCH64_BTI != 0 || GNU_PROPERTY_AARCH64_PAC != 0) && defined(__ELF__)
|
||||
|
||||
#if (GNU_PROPERTY_AARCH64_BTI != 0 || GNU_PROPERTY_AARCH64_PAC != 0 || GNU_PROPERTY_AARCH64_GCS != 0) && defined(__ELF__)
|
||||
.pushsection .note.gnu.property, "a"
|
||||
.balign 8
|
||||
.long 4
|
||||
@@ -203,10 +209,10 @@ DISABLE_SVE2
|
||||
.asciz "GNU"
|
||||
.long 0xc0000000 /* GNU_PROPERTY_AARCH64_FEATURE_1_AND */
|
||||
.long 4
|
||||
.long (GNU_PROPERTY_AARCH64_BTI | GNU_PROPERTY_AARCH64_PAC)
|
||||
.long (GNU_PROPERTY_AARCH64_BTI | GNU_PROPERTY_AARCH64_PAC | GNU_PROPERTY_AARCH64_GCS)
|
||||
.long 0
|
||||
.popsection
|
||||
#endif /* (GNU_PROPERTY_AARCH64_BTI != 0 || GNU_PROPERTY_AARCH64_PAC != 0) && defined(__ELF__) */
|
||||
#endif /* (GNU_PROPERTY_AARCH64_BTI != 0 || GNU_PROPERTY_AARCH64_PAC != 0 || GNU_PROPERTY_AARCH64_GCS != 0) && defined(__ELF__) */
|
||||
#endif /* ARCH_AARCH64 */
|
||||
|
||||
#if ARCH_ARM
|
||||
@@ -323,6 +329,32 @@ EXTERN\name:
|
||||
\name:
|
||||
.endm
|
||||
|
||||
.macro jumptable name
|
||||
#ifdef _WIN32
|
||||
// MS armasm64 doesn't seem to be able to create relocations for subtraction
|
||||
// of labels in different sections; for armasm64 (and all of Windows for
|
||||
// simplicity), write the jump table in the text section, to allow calculating
|
||||
// differences at assembly time. See
|
||||
// https://developercommunity.visualstudio.com/t/armasm64-unable-to-create-cross-section/10722340
|
||||
// for reference. (LLVM can create such relocations, but checking for _WIN32
|
||||
// for simplicity, as execute-only memory isn't relevant on Windows at the
|
||||
// moment.)
|
||||
function \name
|
||||
#else
|
||||
// For other platforms, write jump tables in a const data section, to allow
|
||||
// working in environments where executable memory isn't readable.
|
||||
const \name
|
||||
#endif
|
||||
.endm
|
||||
|
||||
.macro endjumptable
|
||||
#ifdef _WIN32
|
||||
endfunc
|
||||
#else
|
||||
endconst
|
||||
#endif
|
||||
.endm
|
||||
|
||||
#ifdef __APPLE__
|
||||
#define L(x) L ## x
|
||||
#else
|
||||
|
||||
+63
-25
@@ -29,9 +29,10 @@
|
||||
|
||||
#include "common/attributes.h"
|
||||
|
||||
#include "src/cpu.h"
|
||||
#include "src/arm/cpu.h"
|
||||
|
||||
#if defined(HAVE_GETAUXVAL) || defined(HAVE_ELF_AUX_INFO)
|
||||
#if HAVE_GETAUXVAL || HAVE_ELF_AUX_INFO
|
||||
#include <sys/auxv.h>
|
||||
|
||||
#if ARCH_AARCH64
|
||||
@@ -42,17 +43,10 @@
|
||||
#define HWCAP2_AARCH64_I8MM (1 << 13)
|
||||
|
||||
COLD unsigned dav1d_get_cpu_flags_arm(void) {
|
||||
#ifdef HAVE_GETAUXVAL
|
||||
unsigned long hw_cap = getauxval(AT_HWCAP);
|
||||
unsigned long hw_cap2 = getauxval(AT_HWCAP2);
|
||||
#else
|
||||
unsigned long hw_cap = 0;
|
||||
unsigned long hw_cap2 = 0;
|
||||
elf_aux_info(AT_HWCAP, &hw_cap, sizeof(hw_cap));
|
||||
elf_aux_info(AT_HWCAP2, &hw_cap2, sizeof(hw_cap2));
|
||||
#endif
|
||||
unsigned long hw_cap = dav1d_getauxval(AT_HWCAP);
|
||||
unsigned long hw_cap2 = dav1d_getauxval(AT_HWCAP2);
|
||||
|
||||
unsigned flags = DAV1D_ARM_CPU_FLAG_NEON;
|
||||
unsigned flags = dav1d_get_default_cpu_flags();
|
||||
flags |= (hw_cap & HWCAP_AARCH64_ASIMDDP) ? DAV1D_ARM_CPU_FLAG_DOTPROD : 0;
|
||||
flags |= (hw_cap2 & HWCAP2_AARCH64_I8MM) ? DAV1D_ARM_CPU_FLAG_I8MM : 0;
|
||||
flags |= (hw_cap & HWCAP_AARCH64_SVE) ? DAV1D_ARM_CPU_FLAG_SVE : 0;
|
||||
@@ -68,14 +62,10 @@ COLD unsigned dav1d_get_cpu_flags_arm(void) {
|
||||
#define HWCAP_ARM_I8MM (1 << 27)
|
||||
|
||||
COLD unsigned dav1d_get_cpu_flags_arm(void) {
|
||||
#ifdef HAVE_GETAUXVAL
|
||||
unsigned long hw_cap = getauxval(AT_HWCAP);
|
||||
#else
|
||||
unsigned long hw_cap = 0;
|
||||
elf_aux_info(AT_HWCAP, &hw_cap, sizeof(hw_cap));
|
||||
#endif
|
||||
unsigned long hw_cap = dav1d_getauxval(AT_HWCAP);
|
||||
|
||||
unsigned flags = (hw_cap & HWCAP_ARM_NEON) ? DAV1D_ARM_CPU_FLAG_NEON : 0;
|
||||
unsigned flags = dav1d_get_default_cpu_flags();
|
||||
flags |= (hw_cap & HWCAP_ARM_NEON) ? DAV1D_ARM_CPU_FLAG_NEON : 0;
|
||||
flags |= (hw_cap & HWCAP_ARM_ASIMDDP) ? DAV1D_ARM_CPU_FLAG_DOTPROD : 0;
|
||||
flags |= (hw_cap & HWCAP_ARM_I8MM) ? DAV1D_ARM_CPU_FLAG_I8MM : 0;
|
||||
return flags;
|
||||
@@ -95,7 +85,7 @@ static int have_feature(const char *feature) {
|
||||
}
|
||||
|
||||
COLD unsigned dav1d_get_cpu_flags_arm(void) {
|
||||
unsigned flags = DAV1D_ARM_CPU_FLAG_NEON;
|
||||
unsigned flags = dav1d_get_default_cpu_flags();
|
||||
if (have_feature("hw.optional.arm.FEAT_DotProd"))
|
||||
flags |= DAV1D_ARM_CPU_FLAG_DOTPROD;
|
||||
if (have_feature("hw.optional.arm.FEAT_I8MM"))
|
||||
@@ -104,21 +94,68 @@ COLD unsigned dav1d_get_cpu_flags_arm(void) {
|
||||
return flags;
|
||||
}
|
||||
|
||||
#elif defined(__OpenBSD__) && ARCH_AARCH64
|
||||
#include <machine/armreg.h>
|
||||
#include <machine/cpu.h>
|
||||
#include <sys/types.h>
|
||||
#include <sys/sysctl.h>
|
||||
|
||||
COLD unsigned dav1d_get_cpu_flags_arm(void) {
|
||||
unsigned flags = dav1d_get_default_cpu_flags();
|
||||
|
||||
#ifdef CPU_ID_AA64ISAR0
|
||||
int mib[2];
|
||||
uint64_t isar0;
|
||||
uint64_t isar1;
|
||||
size_t len;
|
||||
|
||||
mib[0] = CTL_MACHDEP;
|
||||
mib[1] = CPU_ID_AA64ISAR0;
|
||||
len = sizeof(isar0);
|
||||
if (sysctl(mib, 2, &isar0, &len, NULL, 0) != -1) {
|
||||
if (ID_AA64ISAR0_DP(isar0) >= ID_AA64ISAR0_DP_IMPL)
|
||||
flags |= DAV1D_ARM_CPU_FLAG_DOTPROD;
|
||||
}
|
||||
|
||||
mib[0] = CTL_MACHDEP;
|
||||
mib[1] = CPU_ID_AA64ISAR1;
|
||||
len = sizeof(isar1);
|
||||
if (sysctl(mib, 2, &isar1, &len, NULL, 0) != -1) {
|
||||
#ifdef ID_AA64ISAR1_I8MM_IMPL
|
||||
if (ID_AA64ISAR1_I8MM(isar1) >= ID_AA64ISAR1_I8MM_IMPL)
|
||||
flags |= DAV1D_ARM_CPU_FLAG_I8MM;
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
return flags;
|
||||
}
|
||||
|
||||
#elif defined(_WIN32)
|
||||
#include <windows.h>
|
||||
|
||||
COLD unsigned dav1d_get_cpu_flags_arm(void) {
|
||||
unsigned flags = DAV1D_ARM_CPU_FLAG_NEON;
|
||||
unsigned flags = dav1d_get_default_cpu_flags();
|
||||
#ifdef PF_ARM_V82_DP_INSTRUCTIONS_AVAILABLE
|
||||
if (IsProcessorFeaturePresent(PF_ARM_V82_DP_INSTRUCTIONS_AVAILABLE))
|
||||
flags |= DAV1D_ARM_CPU_FLAG_DOTPROD;
|
||||
#endif
|
||||
/* No I8MM or SVE feature detection available on Windows at the time of
|
||||
* writing. */
|
||||
#ifdef PF_ARM_SVE_INSTRUCTIONS_AVAILABLE
|
||||
if (IsProcessorFeaturePresent(PF_ARM_SVE_INSTRUCTIONS_AVAILABLE))
|
||||
flags |= DAV1D_ARM_CPU_FLAG_SVE;
|
||||
#endif
|
||||
#ifdef PF_ARM_SVE2_INSTRUCTIONS_AVAILABLE
|
||||
if (IsProcessorFeaturePresent(PF_ARM_SVE2_INSTRUCTIONS_AVAILABLE))
|
||||
flags |= DAV1D_ARM_CPU_FLAG_SVE2;
|
||||
#endif
|
||||
#ifdef PF_ARM_V82_I8MM_INSTRUCTIONS_AVAILABLE
|
||||
if (IsProcessorFeaturePresent(PF_ARM_V82_I8MM_INSTRUCTIONS_AVAILABLE))
|
||||
flags |= DAV1D_ARM_CPU_FLAG_I8MM;
|
||||
#endif
|
||||
return flags;
|
||||
}
|
||||
|
||||
#elif defined(__ANDROID__)
|
||||
#elif defined(__ANDROID__) || defined(__linux__)
|
||||
#include <ctype.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
@@ -160,7 +197,8 @@ static unsigned parse_proc_cpuinfo(const char *flag) {
|
||||
}
|
||||
|
||||
COLD unsigned dav1d_get_cpu_flags_arm(void) {
|
||||
unsigned flags = parse_proc_cpuinfo("neon") ? DAV1D_ARM_CPU_FLAG_NEON : 0;
|
||||
unsigned flags = dav1d_get_default_cpu_flags();
|
||||
flags |= parse_proc_cpuinfo("neon") ? DAV1D_ARM_CPU_FLAG_NEON : 0;
|
||||
flags |= parse_proc_cpuinfo("asimd") ? DAV1D_ARM_CPU_FLAG_NEON : 0;
|
||||
flags |= parse_proc_cpuinfo("asimddp") ? DAV1D_ARM_CPU_FLAG_DOTPROD : 0;
|
||||
flags |= parse_proc_cpuinfo("i8mm") ? DAV1D_ARM_CPU_FLAG_I8MM : 0;
|
||||
@@ -174,7 +212,7 @@ COLD unsigned dav1d_get_cpu_flags_arm(void) {
|
||||
#else /* Unsupported OS */
|
||||
|
||||
COLD unsigned dav1d_get_cpu_flags_arm(void) {
|
||||
return 0;
|
||||
return dav1d_get_default_cpu_flags();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
+4
-1
@@ -49,7 +49,9 @@ decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_64x16, neon));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_64x32, neon));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_64x64, neon));
|
||||
|
||||
static ALWAYS_INLINE void itx_dsp_init_arm(Dav1dInvTxfmDSPContext *const c, int bpc) {
|
||||
static ALWAYS_INLINE void itx_dsp_init_arm(Dav1dInvTxfmDSPContext *const c, int bpc,
|
||||
int *const all_simd)
|
||||
{
|
||||
const unsigned flags = dav1d_get_cpu_flags();
|
||||
|
||||
if (!(flags & DAV1D_ARM_CPU_FLAG_NEON)) return;
|
||||
@@ -77,4 +79,5 @@ static ALWAYS_INLINE void itx_dsp_init_arm(Dav1dInvTxfmDSPContext *const c, int
|
||||
assign_itx1_fn (R, 64, 16, neon);
|
||||
assign_itx1_fn (R, 64, 32, neon);
|
||||
assign_itx1_fn ( , 64, 64, neon);
|
||||
*all_simd = 1;
|
||||
}
|
||||
|
||||
+276
-258
@@ -58,9 +58,8 @@ void BF(dav1d_wiener_filter5, neon)(pixel *p, const ptrdiff_t stride,
|
||||
// reference C version, but with the end result subtracted by
|
||||
// 1 << (bitdepth + 6 - round_bits_h).
|
||||
void BF(dav1d_wiener_filter_h, neon)(int16_t *dst, const pixel (*left)[4],
|
||||
const pixel *src, ptrdiff_t stride,
|
||||
const int16_t fh[8], intptr_t w,
|
||||
int h, enum LrEdgeFlags edges
|
||||
const pixel *src, const int16_t fh[8],
|
||||
const int w, const enum LrEdgeFlags edges
|
||||
HIGHBD_DECL_SUFFIX);
|
||||
// This calculates things slightly differently than the reference C version.
|
||||
// This version calculates roughly this:
|
||||
@@ -69,196 +68,173 @@ void BF(dav1d_wiener_filter_h, neon)(int16_t *dst, const pixel (*left)[4],
|
||||
// sum += mid[idx] * fv[i];
|
||||
// sum = (sum + rounding_off_v) >> round_bits_v;
|
||||
// This function assumes that the width is a multiple of 8.
|
||||
void BF(dav1d_wiener_filter_v, neon)(pixel *dst, ptrdiff_t stride,
|
||||
const int16_t *mid, int w, int h,
|
||||
const int16_t fv[8], enum LrEdgeFlags edges,
|
||||
ptrdiff_t mid_stride HIGHBD_DECL_SUFFIX);
|
||||
void BF(dav1d_wiener_filter_v, neon)(pixel *dst, int16_t **ptrs,
|
||||
const int16_t fv[8], const int w
|
||||
HIGHBD_DECL_SUFFIX);
|
||||
|
||||
static void wiener_filter_neon(pixel *const dst, const ptrdiff_t stride,
|
||||
const pixel (*const left)[4], const pixel *lpf,
|
||||
const int w, const int h,
|
||||
void BF(dav1d_wiener_filter_hv, neon)(pixel *dst, const pixel (*left)[4],
|
||||
const pixel *src,
|
||||
const int16_t filter[2][8],
|
||||
const int w, const enum LrEdgeFlags edges,
|
||||
int16_t **ptrs
|
||||
HIGHBD_DECL_SUFFIX);
|
||||
|
||||
static void wiener_filter_neon(pixel *p, const ptrdiff_t stride,
|
||||
const pixel (*left)[4], const pixel *lpf,
|
||||
const int w, int h,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
ALIGN_STK_16(int16_t, hor, 6 * 384,);
|
||||
int16_t *ptrs[7], *rows[6];
|
||||
for (int i = 0; i < 6; i++)
|
||||
rows[i] = &hor[i * 384];
|
||||
const int16_t (*const filter)[8] = params->filter;
|
||||
ALIGN_STK_16(int16_t, mid, 68 * 384,);
|
||||
int mid_stride = (w + 7) & ~7;
|
||||
const int16_t *fh = params->filter[0];
|
||||
const int16_t *fv = params->filter[1];
|
||||
const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
|
||||
|
||||
// Horizontal filter
|
||||
BF(dav1d_wiener_filter_h, neon)(&mid[2 * mid_stride], left, dst, stride,
|
||||
filter[0], w, h, edges HIGHBD_TAIL_SUFFIX);
|
||||
if (edges & LR_HAVE_TOP)
|
||||
BF(dav1d_wiener_filter_h, neon)(mid, NULL, lpf, stride,
|
||||
filter[0], w, 2, edges
|
||||
const pixel *src = p;
|
||||
if (edges & LR_HAVE_TOP) {
|
||||
ptrs[0] = rows[0];
|
||||
ptrs[1] = rows[0];
|
||||
ptrs[2] = rows[1];
|
||||
ptrs[3] = rows[2];
|
||||
ptrs[4] = rows[2];
|
||||
ptrs[5] = rows[2];
|
||||
|
||||
BF(dav1d_wiener_filter_h, neon)(rows[0], NULL, lpf, fh, w, edges
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
if (edges & LR_HAVE_BOTTOM)
|
||||
BF(dav1d_wiener_filter_h, neon)(&mid[(2 + h) * mid_stride], NULL,
|
||||
lpf + 6 * PXSTRIDE(stride),
|
||||
stride, filter[0], w, 2, edges
|
||||
lpf += PXSTRIDE(stride);
|
||||
BF(dav1d_wiener_filter_h, neon)(rows[1], NULL, lpf, fh, w, edges
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
|
||||
// Vertical filter
|
||||
BF(dav1d_wiener_filter_v, neon)(dst, stride, &mid[2*mid_stride],
|
||||
w, h, filter[1], edges,
|
||||
mid_stride * sizeof(*mid)
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
BF(dav1d_wiener_filter_h, neon)(rows[2], left, src, fh, w, edges
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
|
||||
if (--h <= 0)
|
||||
goto v1;
|
||||
|
||||
ptrs[4] = ptrs[5] = rows[3];
|
||||
BF(dav1d_wiener_filter_h, neon)(rows[3], left, src, fh, w, edges
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
|
||||
if (--h <= 0)
|
||||
goto v2;
|
||||
|
||||
ptrs[5] = rows[4];
|
||||
BF(dav1d_wiener_filter_h, neon)(rows[4], left, src, fh, w, edges
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
|
||||
if (--h <= 0)
|
||||
goto v3;
|
||||
} else {
|
||||
ptrs[0] = rows[0];
|
||||
ptrs[1] = rows[0];
|
||||
ptrs[2] = rows[0];
|
||||
ptrs[3] = rows[0];
|
||||
ptrs[4] = rows[0];
|
||||
ptrs[5] = rows[0];
|
||||
|
||||
BF(dav1d_wiener_filter_h, neon)(rows[0], left, src, fh, w, edges
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
|
||||
if (--h <= 0)
|
||||
goto v1;
|
||||
|
||||
ptrs[4] = ptrs[5] = rows[1];
|
||||
BF(dav1d_wiener_filter_h, neon)(rows[1], left, src, fh, w, edges
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
|
||||
if (--h <= 0)
|
||||
goto v2;
|
||||
|
||||
ptrs[5] = rows[2];
|
||||
BF(dav1d_wiener_filter_h, neon)(rows[2], left, src, fh, w, edges
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
|
||||
if (--h <= 0)
|
||||
goto v3;
|
||||
|
||||
ptrs[6] = rows[3];
|
||||
BF(dav1d_wiener_filter_hv, neon)(p, left, src, filter, w, edges, ptrs
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
p += PXSTRIDE(stride);
|
||||
|
||||
if (--h <= 0)
|
||||
goto v3;
|
||||
|
||||
ptrs[6] = rows[4];
|
||||
BF(dav1d_wiener_filter_hv, neon)(p, left, src, filter, w, edges, ptrs
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
p += PXSTRIDE(stride);
|
||||
|
||||
if (--h <= 0)
|
||||
goto v3;
|
||||
}
|
||||
|
||||
ptrs[6] = ptrs[5] + 384;
|
||||
do {
|
||||
BF(dav1d_wiener_filter_hv, neon)(p, left, src, filter, w, edges, ptrs
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
p += PXSTRIDE(stride);
|
||||
} while (--h > 0);
|
||||
|
||||
if (!(edges & LR_HAVE_BOTTOM))
|
||||
goto v3;
|
||||
|
||||
BF(dav1d_wiener_filter_hv, neon)(p, NULL, lpf_bottom, filter, w, edges, ptrs
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
lpf_bottom += PXSTRIDE(stride);
|
||||
p += PXSTRIDE(stride);
|
||||
|
||||
BF(dav1d_wiener_filter_hv, neon)(p, NULL, lpf_bottom, filter, w, edges, ptrs
|
||||
HIGHBD_TAIL_SUFFIX);
|
||||
p += PXSTRIDE(stride);
|
||||
v1:
|
||||
BF(dav1d_wiener_filter_v, neon)(p, ptrs, fv, w HIGHBD_TAIL_SUFFIX);
|
||||
|
||||
return;
|
||||
|
||||
v3:
|
||||
BF(dav1d_wiener_filter_v, neon)(p, ptrs, fv, w HIGHBD_TAIL_SUFFIX);
|
||||
p += PXSTRIDE(stride);
|
||||
v2:
|
||||
BF(dav1d_wiener_filter_v, neon)(p, ptrs, fv, w HIGHBD_TAIL_SUFFIX);
|
||||
p += PXSTRIDE(stride);
|
||||
goto v1;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if ARCH_ARM
|
||||
void BF(dav1d_sgr_box3_h, neon)(int32_t *sumsq, int16_t *sum,
|
||||
const pixel (*left)[4],
|
||||
const pixel *src, const ptrdiff_t stride,
|
||||
const int w, const int h,
|
||||
const enum LrEdgeFlags edges);
|
||||
void dav1d_sgr_box3_v_neon(int32_t *sumsq, int16_t *sum,
|
||||
const int w, const int h,
|
||||
const enum LrEdgeFlags edges);
|
||||
void dav1d_sgr_calc_ab1_neon(int32_t *a, int16_t *b,
|
||||
const int w, const int h, const int strength,
|
||||
const int bitdepth_max);
|
||||
void BF(dav1d_sgr_finish_filter1, neon)(int16_t *tmp,
|
||||
const pixel *src, const ptrdiff_t stride,
|
||||
const int32_t *a, const int16_t *b,
|
||||
const int w, const int h);
|
||||
|
||||
/* filter with a 3x3 box (radius=1) */
|
||||
static void dav1d_sgr_filter1_neon(int16_t *tmp,
|
||||
const pixel *src, const ptrdiff_t stride,
|
||||
const pixel (*left)[4], const pixel *lpf,
|
||||
const int w, const int h, const int strength,
|
||||
const enum LrEdgeFlags edges
|
||||
HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
ALIGN_STK_16(int32_t, sumsq_mem, (384 + 16) * 68 + 8,);
|
||||
int32_t *const sumsq = &sumsq_mem[(384 + 16) * 2 + 8], *const a = sumsq;
|
||||
ALIGN_STK_16(int16_t, sum_mem, (384 + 16) * 68 + 16,);
|
||||
int16_t *const sum = &sum_mem[(384 + 16) * 2 + 16], *const b = sum;
|
||||
|
||||
BF(dav1d_sgr_box3_h, neon)(sumsq, sum, left, src, stride, w, h, edges);
|
||||
if (edges & LR_HAVE_TOP)
|
||||
BF(dav1d_sgr_box3_h, neon)(&sumsq[-2 * (384 + 16)], &sum[-2 * (384 + 16)],
|
||||
NULL, lpf, stride, w, 2, edges);
|
||||
|
||||
if (edges & LR_HAVE_BOTTOM)
|
||||
BF(dav1d_sgr_box3_h, neon)(&sumsq[h * (384 + 16)], &sum[h * (384 + 16)],
|
||||
NULL, lpf + 6 * PXSTRIDE(stride),
|
||||
stride, w, 2, edges);
|
||||
|
||||
dav1d_sgr_box3_v_neon(sumsq, sum, w, h, edges);
|
||||
dav1d_sgr_calc_ab1_neon(a, b, w, h, strength, BITDEPTH_MAX);
|
||||
BF(dav1d_sgr_finish_filter1, neon)(tmp, src, stride, a, b, w, h);
|
||||
}
|
||||
|
||||
void BF(dav1d_sgr_box5_h, neon)(int32_t *sumsq, int16_t *sum,
|
||||
const pixel (*left)[4],
|
||||
const pixel *src, const ptrdiff_t stride,
|
||||
const int w, const int h,
|
||||
const enum LrEdgeFlags edges);
|
||||
void dav1d_sgr_box5_v_neon(int32_t *sumsq, int16_t *sum,
|
||||
const int w, const int h,
|
||||
const enum LrEdgeFlags edges);
|
||||
void dav1d_sgr_calc_ab2_neon(int32_t *a, int16_t *b,
|
||||
const int w, const int h, const int strength,
|
||||
const int bitdepth_max);
|
||||
void BF(dav1d_sgr_finish_filter2, neon)(int16_t *tmp,
|
||||
const pixel *src, const ptrdiff_t stride,
|
||||
const int32_t *a, const int16_t *b,
|
||||
const int w, const int h);
|
||||
|
||||
/* filter with a 5x5 box (radius=2) */
|
||||
static void dav1d_sgr_filter2_neon(int16_t *tmp,
|
||||
const pixel *src, const ptrdiff_t stride,
|
||||
const pixel (*left)[4], const pixel *lpf,
|
||||
const int w, const int h, const int strength,
|
||||
const enum LrEdgeFlags edges
|
||||
HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
ALIGN_STK_16(int32_t, sumsq_mem, (384 + 16) * 68 + 8,);
|
||||
int32_t *const sumsq = &sumsq_mem[(384 + 16) * 2 + 8], *const a = sumsq;
|
||||
ALIGN_STK_16(int16_t, sum_mem, (384 + 16) * 68 + 16,);
|
||||
int16_t *const sum = &sum_mem[(384 + 16) * 2 + 16], *const b = sum;
|
||||
|
||||
BF(dav1d_sgr_box5_h, neon)(sumsq, sum, left, src, stride, w, h, edges);
|
||||
if (edges & LR_HAVE_TOP)
|
||||
BF(dav1d_sgr_box5_h, neon)(&sumsq[-2 * (384 + 16)], &sum[-2 * (384 + 16)],
|
||||
NULL, lpf, stride, w, 2, edges);
|
||||
|
||||
if (edges & LR_HAVE_BOTTOM)
|
||||
BF(dav1d_sgr_box5_h, neon)(&sumsq[h * (384 + 16)], &sum[h * (384 + 16)],
|
||||
NULL, lpf + 6 * PXSTRIDE(stride),
|
||||
stride, w, 2, edges);
|
||||
|
||||
dav1d_sgr_box5_v_neon(sumsq, sum, w, h, edges);
|
||||
dav1d_sgr_calc_ab2_neon(a, b, w, h, strength, BITDEPTH_MAX);
|
||||
BF(dav1d_sgr_finish_filter2, neon)(tmp, src, stride, a, b, w, h);
|
||||
}
|
||||
|
||||
void BF(dav1d_sgr_weighted1, neon)(pixel *dst, const ptrdiff_t dst_stride,
|
||||
const pixel *src, const ptrdiff_t src_stride,
|
||||
const int16_t *t1, const int w, const int h,
|
||||
const int wt HIGHBD_DECL_SUFFIX);
|
||||
void BF(dav1d_sgr_weighted2, neon)(pixel *dst, const ptrdiff_t dst_stride,
|
||||
const pixel *src, const ptrdiff_t src_stride,
|
||||
const int16_t *t1, const int16_t *t2,
|
||||
const int w, const int h,
|
||||
const int16_t wt[2] HIGHBD_DECL_SUFFIX);
|
||||
|
||||
static void sgr_filter_5x5_neon(pixel *const dst, const ptrdiff_t stride,
|
||||
const pixel (*const left)[4], const pixel *lpf,
|
||||
const int w, const int h,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
ALIGN_STK_16(int16_t, tmp, 64 * 384,);
|
||||
dav1d_sgr_filter2_neon(tmp, dst, stride, left, lpf,
|
||||
w, h, params->sgr.s0, edges HIGHBD_TAIL_SUFFIX);
|
||||
BF(dav1d_sgr_weighted1, neon)(dst, stride, dst, stride,
|
||||
tmp, w, h, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
|
||||
}
|
||||
|
||||
static void sgr_filter_3x3_neon(pixel *const dst, const ptrdiff_t stride,
|
||||
const pixel (*const left)[4], const pixel *lpf,
|
||||
const int w, const int h,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
ALIGN_STK_16(int16_t, tmp, 64 * 384,);
|
||||
dav1d_sgr_filter1_neon(tmp, dst, stride, left, lpf,
|
||||
w, h, params->sgr.s1, edges HIGHBD_TAIL_SUFFIX);
|
||||
BF(dav1d_sgr_weighted1, neon)(dst, stride, dst, stride,
|
||||
tmp, w, h, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
|
||||
}
|
||||
|
||||
static void sgr_filter_mix_neon(pixel *const dst, const ptrdiff_t stride,
|
||||
const pixel (*const left)[4], const pixel *lpf,
|
||||
const int w, const int h,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
ALIGN_STK_16(int16_t, tmp1, 64 * 384,);
|
||||
ALIGN_STK_16(int16_t, tmp2, 64 * 384,);
|
||||
dav1d_sgr_filter2_neon(tmp1, dst, stride, left, lpf,
|
||||
w, h, params->sgr.s0, edges HIGHBD_TAIL_SUFFIX);
|
||||
dav1d_sgr_filter1_neon(tmp2, dst, stride, left, lpf,
|
||||
w, h, params->sgr.s1, edges HIGHBD_TAIL_SUFFIX);
|
||||
const int16_t wt[2] = { params->sgr.w0, params->sgr.w1 };
|
||||
BF(dav1d_sgr_weighted2, neon)(dst, stride, dst, stride,
|
||||
tmp1, tmp2, w, h, wt HIGHBD_TAIL_SUFFIX);
|
||||
}
|
||||
|
||||
#else
|
||||
static void rotate(int32_t **sumsq_ptrs, int16_t **sum_ptrs, int n) {
|
||||
static void rotate_neon(int32_t **sumsq_ptrs, int16_t **sum_ptrs, int n) {
|
||||
int32_t *tmp32 = sumsq_ptrs[0];
|
||||
int16_t *tmp16 = sum_ptrs[0];
|
||||
for (int i = 0; i < n - 1; i++) {
|
||||
sumsq_ptrs[i] = sumsq_ptrs[i+1];
|
||||
sum_ptrs[i] = sum_ptrs[i+1];
|
||||
sumsq_ptrs[i] = sumsq_ptrs[i + 1];
|
||||
sum_ptrs[i] = sum_ptrs[i + 1];
|
||||
}
|
||||
sumsq_ptrs[n - 1] = tmp32;
|
||||
sum_ptrs[n - 1] = tmp16;
|
||||
}
|
||||
static void rotate5_x2(int32_t **sumsq_ptrs, int16_t **sum_ptrs) {
|
||||
static void rotate5_x2_neon(int32_t **sumsq_ptrs, int16_t **sum_ptrs) {
|
||||
int32_t *tmp32[2];
|
||||
int16_t *tmp16[2];
|
||||
for (int i = 0; i < 2; i++) {
|
||||
@@ -266,8 +242,8 @@ static void rotate5_x2(int32_t **sumsq_ptrs, int16_t **sum_ptrs) {
|
||||
tmp16[i] = sum_ptrs[i];
|
||||
}
|
||||
for (int i = 0; i < 3; i++) {
|
||||
sumsq_ptrs[i] = sumsq_ptrs[i+2];
|
||||
sum_ptrs[i] = sum_ptrs[i+2];
|
||||
sumsq_ptrs[i] = sumsq_ptrs[i + 2];
|
||||
sum_ptrs[i] = sum_ptrs[i + 2];
|
||||
}
|
||||
for (int i = 0; i < 2; i++) {
|
||||
sumsq_ptrs[3 + i] = tmp32[i];
|
||||
@@ -275,18 +251,6 @@ static void rotate5_x2(int32_t **sumsq_ptrs, int16_t **sum_ptrs) {
|
||||
}
|
||||
}
|
||||
|
||||
static void rotate_ab_3(int32_t **A_ptrs, int16_t **B_ptrs) {
|
||||
rotate(A_ptrs, B_ptrs, 3);
|
||||
}
|
||||
|
||||
static void rotate_ab_2(int32_t **A_ptrs, int16_t **B_ptrs) {
|
||||
rotate(A_ptrs, B_ptrs, 2);
|
||||
}
|
||||
|
||||
static void rotate_ab_4(int32_t **A_ptrs, int16_t **B_ptrs) {
|
||||
rotate(A_ptrs, B_ptrs, 4);
|
||||
}
|
||||
|
||||
void BF(dav1d_sgr_box3_row_h, neon)(int32_t *sumsq, int16_t *sum,
|
||||
const pixel (*left)[4],
|
||||
const pixel *src, const int w,
|
||||
@@ -301,6 +265,24 @@ void BF(dav1d_sgr_box35_row_h, neon)(int32_t *sumsq3, int16_t *sum3,
|
||||
const pixel *src, const int w,
|
||||
const enum LrEdgeFlags edges);
|
||||
|
||||
#if ARCH_ARM
|
||||
void dav1d_sgr_box3_row_v_neon(int32_t **sumsq, int16_t **sum,
|
||||
int32_t *sumsq_out, int16_t *sum_out,
|
||||
const int w);
|
||||
void dav1d_sgr_box5_row_v_neon(int32_t **sumsq, int16_t **sum,
|
||||
int32_t *sumsq_out, int16_t *sum_out,
|
||||
const int w);
|
||||
void dav1d_sgr_calc_row_ab1_neon(int32_t *AA, int16_t *BB, int w, int s,
|
||||
int bitdepth_max);
|
||||
void dav1d_sgr_calc_row_ab2_neon(int32_t *AA, int16_t *BB, int w, int s,
|
||||
int bitdepth_max);
|
||||
void BF(dav1d_sgr_finish_filter_row1, neon)(int16_t *tmp, const pixel *src,
|
||||
int32_t **A_ptrs, int16_t **B_ptrs,
|
||||
const int w);
|
||||
void BF(dav1d_sgr_weighted_row1, neon)(pixel *dst, const int16_t *t1,
|
||||
const int w, const int wt
|
||||
HIGHBD_DECL_SUFFIX);
|
||||
#else
|
||||
void dav1d_sgr_box3_vert_neon(int32_t **sumsq, int16_t **sum,
|
||||
int32_t *AA, int16_t *BB,
|
||||
const int w, const int s,
|
||||
@@ -324,30 +306,40 @@ void BF(dav1d_sgr_finish_filter1_2rows, neon)(int16_t *tmp, const pixel *src,
|
||||
int32_t **A_ptrs,
|
||||
int16_t **B_ptrs,
|
||||
const int w, const int h);
|
||||
#endif
|
||||
void BF(dav1d_sgr_finish_filter2_2rows, neon)(int16_t *tmp, const pixel *src,
|
||||
const ptrdiff_t src_stride,
|
||||
int32_t **A_ptrs, int16_t **B_ptrs,
|
||||
const int w, const int h);
|
||||
void BF(dav1d_sgr_weighted2, neon)(pixel *dst, const ptrdiff_t dst_stride,
|
||||
const pixel *src, const ptrdiff_t src_stride,
|
||||
const int16_t *t1, const int16_t *t2,
|
||||
const int w, const int h,
|
||||
const int16_t wt[2] HIGHBD_DECL_SUFFIX);
|
||||
|
||||
static void sgr_box3_vert_neon(int32_t **sumsq, int16_t **sum,
|
||||
int32_t *sumsq_out, int16_t *sum_out,
|
||||
const int w, int s, int bitdepth_max) {
|
||||
const int w, const int s, const int bitdepth_max) {
|
||||
#if ARCH_ARM
|
||||
dav1d_sgr_box3_row_v_neon(sumsq, sum, sumsq_out, sum_out, w);
|
||||
dav1d_sgr_calc_row_ab1_neon(sumsq_out, sum_out, w, s, bitdepth_max);
|
||||
#else
|
||||
// box3_v + calc_ab1
|
||||
dav1d_sgr_box3_vert_neon(sumsq, sum, sumsq_out, sum_out, w, s, bitdepth_max);
|
||||
rotate(sumsq, sum, 3);
|
||||
#endif
|
||||
rotate_neon(sumsq, sum, 3);
|
||||
}
|
||||
|
||||
static void sgr_box5_vert_neon(int32_t **sumsq, int16_t **sum,
|
||||
int32_t *sumsq_out, int16_t *sum_out,
|
||||
const int w, int s, int bitdepth_max) {
|
||||
const int w, const int s, const int bitdepth_max) {
|
||||
#if ARCH_ARM
|
||||
dav1d_sgr_box5_row_v_neon(sumsq, sum, sumsq_out, sum_out, w);
|
||||
dav1d_sgr_calc_row_ab2_neon(sumsq_out, sum_out, w, s, bitdepth_max);
|
||||
#else
|
||||
// box5_v + calc_ab2
|
||||
dav1d_sgr_box5_vert_neon(sumsq, sum, sumsq_out, sum_out, w, s, bitdepth_max);
|
||||
rotate5_x2(sumsq, sum);
|
||||
#endif
|
||||
rotate5_x2_neon(sumsq, sum);
|
||||
}
|
||||
|
||||
static void sgr_box3_hv_neon(int32_t **sumsq, int16_t **sum,
|
||||
@@ -365,20 +357,41 @@ static void sgr_box3_hv_neon(int32_t **sumsq, int16_t **sum,
|
||||
static void sgr_finish1_neon(pixel **dst, const ptrdiff_t stride,
|
||||
int32_t **A_ptrs, int16_t **B_ptrs, const int w,
|
||||
const int w1 HIGHBD_DECL_SUFFIX) {
|
||||
#if ARCH_ARM
|
||||
ALIGN_STK_16(int16_t, tmp, 384,);
|
||||
|
||||
BF(dav1d_sgr_finish_filter_row1, neon)(tmp, *dst, A_ptrs, B_ptrs, w);
|
||||
BF(dav1d_sgr_weighted_row1, neon)(*dst, tmp, w, w1 HIGHBD_TAIL_SUFFIX);
|
||||
#else
|
||||
BF(dav1d_sgr_finish_weighted1, neon)(*dst, A_ptrs, B_ptrs,
|
||||
w, w1 HIGHBD_TAIL_SUFFIX);
|
||||
#endif
|
||||
*dst += PXSTRIDE(stride);
|
||||
rotate_ab_3(A_ptrs, B_ptrs);
|
||||
rotate_neon(A_ptrs, B_ptrs, 3);
|
||||
}
|
||||
|
||||
#define ARM_FILTER_OUT_STRIDE 384
|
||||
|
||||
static void sgr_finish2_neon(pixel **dst, const ptrdiff_t stride,
|
||||
int32_t **A_ptrs, int16_t **B_ptrs,
|
||||
const int w, const int h, const int w1
|
||||
HIGHBD_DECL_SUFFIX) {
|
||||
#if ARCH_ARM
|
||||
ALIGN_STK_16(int16_t, tmp, 2*ARM_FILTER_OUT_STRIDE,);
|
||||
|
||||
BF(dav1d_sgr_finish_filter2_2rows, neon)(tmp, *dst, stride, A_ptrs, B_ptrs, w, h);
|
||||
BF(dav1d_sgr_weighted_row1, neon)(*dst, tmp, w, w1 HIGHBD_TAIL_SUFFIX);
|
||||
*dst += PXSTRIDE(stride);
|
||||
if (h > 1) {
|
||||
BF(dav1d_sgr_weighted_row1, neon)(*dst, tmp + FILTER_OUT_STRIDE, w, w1 HIGHBD_TAIL_SUFFIX);
|
||||
*dst += PXSTRIDE(stride);
|
||||
}
|
||||
#else
|
||||
BF(dav1d_sgr_finish_weighted2, neon)(*dst, stride, A_ptrs, B_ptrs,
|
||||
w, h, w1 HIGHBD_TAIL_SUFFIX);
|
||||
*dst += 2*PXSTRIDE(stride);
|
||||
rotate_ab_2(A_ptrs, B_ptrs);
|
||||
#endif
|
||||
rotate_neon(A_ptrs, B_ptrs, 2);
|
||||
}
|
||||
|
||||
static void sgr_finish_mix_neon(pixel **dst, const ptrdiff_t stride,
|
||||
@@ -386,20 +399,26 @@ static void sgr_finish_mix_neon(pixel **dst, const ptrdiff_t stride,
|
||||
int32_t **A3_ptrs, int16_t **B3_ptrs,
|
||||
const int w, const int h,
|
||||
const int w0, const int w1 HIGHBD_DECL_SUFFIX) {
|
||||
#define FILTER_OUT_STRIDE 384
|
||||
ALIGN_STK_16(int16_t, tmp5, 2*FILTER_OUT_STRIDE,);
|
||||
ALIGN_STK_16(int16_t, tmp3, 2*FILTER_OUT_STRIDE,);
|
||||
ALIGN_STK_16(int16_t, tmp5, 2*ARM_FILTER_OUT_STRIDE,);
|
||||
ALIGN_STK_16(int16_t, tmp3, 2*ARM_FILTER_OUT_STRIDE,);
|
||||
|
||||
BF(dav1d_sgr_finish_filter2_2rows, neon)(tmp5, *dst, stride,
|
||||
A5_ptrs, B5_ptrs, w, h);
|
||||
#if ARCH_ARM
|
||||
BF(dav1d_sgr_finish_filter_row1, neon)(tmp3, *dst, A3_ptrs, B3_ptrs, w);
|
||||
BF(dav1d_sgr_finish_filter_row1, neon)(tmp3 + FILTER_OUT_STRIDE,
|
||||
*dst + PXSTRIDE(stride),
|
||||
&A3_ptrs[1], &B3_ptrs[1], w);
|
||||
#else
|
||||
BF(dav1d_sgr_finish_filter1_2rows, neon)(tmp3, *dst, stride,
|
||||
A3_ptrs, B3_ptrs, w, h);
|
||||
#endif
|
||||
const int16_t wt[2] = { w0, w1 };
|
||||
BF(dav1d_sgr_weighted2, neon)(*dst, stride, *dst, stride,
|
||||
BF(dav1d_sgr_weighted2, neon)(*dst, stride,
|
||||
tmp5, tmp3, w, h, wt HIGHBD_TAIL_SUFFIX);
|
||||
*dst += h*PXSTRIDE(stride);
|
||||
rotate_ab_2(A5_ptrs, B5_ptrs);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A5_ptrs, B5_ptrs, 2);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
}
|
||||
|
||||
|
||||
@@ -409,23 +428,23 @@ static void sgr_filter_3x3_neon(pixel *dst, const ptrdiff_t stride,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
#define BUF_STRIDE (384 + 16)
|
||||
ALIGN_STK_16(int32_t, sumsq_buf, BUF_STRIDE * 3 + 16,);
|
||||
ALIGN_STK_16(int16_t, sum_buf, BUF_STRIDE * 3 + 16,);
|
||||
#define ARM_BUF_STRIDE (384 + 16)
|
||||
ALIGN_STK_16(int32_t, sumsq_buf, ARM_BUF_STRIDE * 3 + 16,);
|
||||
ALIGN_STK_16(int16_t, sum_buf, ARM_BUF_STRIDE * 3 + 16,);
|
||||
int32_t *sumsq_ptrs[3], *sumsq_rows[3];
|
||||
int16_t *sum_ptrs[3], *sum_rows[3];
|
||||
for (int i = 0; i < 3; i++) {
|
||||
sumsq_rows[i] = &sumsq_buf[i * BUF_STRIDE];
|
||||
sum_rows[i] = &sum_buf[i * BUF_STRIDE];
|
||||
sumsq_rows[i] = &sumsq_buf[i * ARM_BUF_STRIDE];
|
||||
sum_rows[i] = &sum_buf[i * ARM_BUF_STRIDE];
|
||||
}
|
||||
|
||||
ALIGN_STK_16(int32_t, A_buf, BUF_STRIDE * 3 + 16,);
|
||||
ALIGN_STK_16(int16_t, B_buf, BUF_STRIDE * 3 + 16,);
|
||||
ALIGN_STK_16(int32_t, A_buf, ARM_BUF_STRIDE * 3 + 16,);
|
||||
ALIGN_STK_16(int16_t, B_buf, ARM_BUF_STRIDE * 3 + 16,);
|
||||
int32_t *A_ptrs[3];
|
||||
int16_t *B_ptrs[3];
|
||||
for (int i = 0; i < 3; i++) {
|
||||
A_ptrs[i] = &A_buf[i * BUF_STRIDE];
|
||||
B_ptrs[i] = &B_buf[i * BUF_STRIDE];
|
||||
A_ptrs[i] = &A_buf[i * ARM_BUF_STRIDE];
|
||||
B_ptrs[i] = &B_buf[i * ARM_BUF_STRIDE];
|
||||
}
|
||||
const pixel *src = dst;
|
||||
const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
|
||||
@@ -448,15 +467,16 @@ static void sgr_filter_3x3_neon(pixel *dst, const ptrdiff_t stride,
|
||||
left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
rotate_ab_3(A_ptrs, B_ptrs);
|
||||
rotate_neon(A_ptrs, B_ptrs, 3);
|
||||
|
||||
if (--h <= 0)
|
||||
goto vert_1;
|
||||
|
||||
sgr_box3_hv_neon(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2], left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
|
||||
sgr_box3_hv_neon(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
|
||||
left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
rotate_ab_3(A_ptrs, B_ptrs);
|
||||
rotate_neon(A_ptrs, B_ptrs, 3);
|
||||
|
||||
if (--h <= 0)
|
||||
goto vert_2;
|
||||
@@ -475,7 +495,7 @@ static void sgr_filter_3x3_neon(pixel *dst, const ptrdiff_t stride,
|
||||
|
||||
sgr_box3_vert_neon(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_3(A_ptrs, B_ptrs);
|
||||
rotate_neon(A_ptrs, B_ptrs, 3);
|
||||
|
||||
if (--h <= 0)
|
||||
goto vert_1;
|
||||
@@ -487,7 +507,7 @@ static void sgr_filter_3x3_neon(pixel *dst, const ptrdiff_t stride,
|
||||
left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
|
||||
left++;
|
||||
src += PXSTRIDE(stride);
|
||||
rotate_ab_3(A_ptrs, B_ptrs);
|
||||
rotate_neon(A_ptrs, B_ptrs, 3);
|
||||
|
||||
if (--h <= 0)
|
||||
goto vert_2;
|
||||
@@ -547,7 +567,7 @@ vert_1:
|
||||
sum_ptrs[2] = sum_ptrs[1];
|
||||
sgr_box3_vert_neon(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_3(A_ptrs, B_ptrs);
|
||||
rotate_neon(A_ptrs, B_ptrs, 3);
|
||||
goto output_1;
|
||||
}
|
||||
|
||||
@@ -557,22 +577,22 @@ static void sgr_filter_5x5_neon(pixel *dst, const ptrdiff_t stride,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
ALIGN_STK_16(int32_t, sumsq_buf, BUF_STRIDE * 5 + 16,);
|
||||
ALIGN_STK_16(int16_t, sum_buf, BUF_STRIDE * 5 + 16,);
|
||||
ALIGN_STK_16(int32_t, sumsq_buf, ARM_BUF_STRIDE * 5 + 16,);
|
||||
ALIGN_STK_16(int16_t, sum_buf, ARM_BUF_STRIDE * 5 + 16,);
|
||||
int32_t *sumsq_ptrs[5], *sumsq_rows[5];
|
||||
int16_t *sum_ptrs[5], *sum_rows[5];
|
||||
for (int i = 0; i < 5; i++) {
|
||||
sumsq_rows[i] = &sumsq_buf[i * BUF_STRIDE];
|
||||
sum_rows[i] = &sum_buf[i * BUF_STRIDE];
|
||||
sumsq_rows[i] = &sumsq_buf[i * ARM_BUF_STRIDE];
|
||||
sum_rows[i] = &sum_buf[i * ARM_BUF_STRIDE];
|
||||
}
|
||||
|
||||
ALIGN_STK_16(int32_t, A_buf, BUF_STRIDE * 2 + 16,);
|
||||
ALIGN_STK_16(int16_t, B_buf, BUF_STRIDE * 2 + 16,);
|
||||
ALIGN_STK_16(int32_t, A_buf, ARM_BUF_STRIDE * 2 + 16,);
|
||||
ALIGN_STK_16(int16_t, B_buf, ARM_BUF_STRIDE * 2 + 16,);
|
||||
int32_t *A_ptrs[2];
|
||||
int16_t *B_ptrs[2];
|
||||
for (int i = 0; i < 2; i++) {
|
||||
A_ptrs[i] = &A_buf[i * BUF_STRIDE];
|
||||
B_ptrs[i] = &B_buf[i * BUF_STRIDE];
|
||||
A_ptrs[i] = &A_buf[i * ARM_BUF_STRIDE];
|
||||
B_ptrs[i] = &B_buf[i * ARM_BUF_STRIDE];
|
||||
}
|
||||
const pixel *src = dst;
|
||||
const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
|
||||
@@ -609,7 +629,7 @@ static void sgr_filter_5x5_neon(pixel *dst, const ptrdiff_t stride,
|
||||
src += PXSTRIDE(stride);
|
||||
sgr_box5_vert_neon(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
|
||||
w, params->sgr.s0, BITDEPTH_MAX);
|
||||
rotate_ab_2(A_ptrs, B_ptrs);
|
||||
rotate_neon(A_ptrs, B_ptrs, 2);
|
||||
|
||||
if (--h <= 0)
|
||||
goto vert_2;
|
||||
@@ -648,7 +668,7 @@ static void sgr_filter_5x5_neon(pixel *dst, const ptrdiff_t stride,
|
||||
|
||||
sgr_box5_vert_neon(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
|
||||
w, params->sgr.s0, BITDEPTH_MAX);
|
||||
rotate_ab_2(A_ptrs, B_ptrs);
|
||||
rotate_neon(A_ptrs, B_ptrs, 2);
|
||||
|
||||
if (--h <= 0)
|
||||
goto vert_2;
|
||||
@@ -760,7 +780,7 @@ vert_1:
|
||||
|
||||
sgr_box5_vert_neon(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
|
||||
w, params->sgr.s0, BITDEPTH_MAX);
|
||||
rotate_ab_2(A_ptrs, B_ptrs);
|
||||
rotate_neon(A_ptrs, B_ptrs, 2);
|
||||
|
||||
goto output_1;
|
||||
}
|
||||
@@ -771,38 +791,38 @@ static void sgr_filter_mix_neon(pixel *dst, const ptrdiff_t stride,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
ALIGN_STK_16(int32_t, sumsq5_buf, BUF_STRIDE * 5 + 16,);
|
||||
ALIGN_STK_16(int16_t, sum5_buf, BUF_STRIDE * 5 + 16,);
|
||||
ALIGN_STK_16(int32_t, sumsq5_buf, ARM_BUF_STRIDE * 5 + 16,);
|
||||
ALIGN_STK_16(int16_t, sum5_buf, ARM_BUF_STRIDE * 5 + 16,);
|
||||
int32_t *sumsq5_ptrs[5], *sumsq5_rows[5];
|
||||
int16_t *sum5_ptrs[5], *sum5_rows[5];
|
||||
for (int i = 0; i < 5; i++) {
|
||||
sumsq5_rows[i] = &sumsq5_buf[i * BUF_STRIDE];
|
||||
sum5_rows[i] = &sum5_buf[i * BUF_STRIDE];
|
||||
sumsq5_rows[i] = &sumsq5_buf[i * ARM_BUF_STRIDE];
|
||||
sum5_rows[i] = &sum5_buf[i * ARM_BUF_STRIDE];
|
||||
}
|
||||
ALIGN_STK_16(int32_t, sumsq3_buf, BUF_STRIDE * 3 + 16,);
|
||||
ALIGN_STK_16(int16_t, sum3_buf, BUF_STRIDE * 3 + 16,);
|
||||
ALIGN_STK_16(int32_t, sumsq3_buf, ARM_BUF_STRIDE * 3 + 16,);
|
||||
ALIGN_STK_16(int16_t, sum3_buf, ARM_BUF_STRIDE * 3 + 16,);
|
||||
int32_t *sumsq3_ptrs[3], *sumsq3_rows[3];
|
||||
int16_t *sum3_ptrs[3], *sum3_rows[3];
|
||||
for (int i = 0; i < 3; i++) {
|
||||
sumsq3_rows[i] = &sumsq3_buf[i * BUF_STRIDE];
|
||||
sum3_rows[i] = &sum3_buf[i * BUF_STRIDE];
|
||||
sumsq3_rows[i] = &sumsq3_buf[i * ARM_BUF_STRIDE];
|
||||
sum3_rows[i] = &sum3_buf[i * ARM_BUF_STRIDE];
|
||||
}
|
||||
|
||||
ALIGN_STK_16(int32_t, A5_buf, BUF_STRIDE * 2 + 16,);
|
||||
ALIGN_STK_16(int16_t, B5_buf, BUF_STRIDE * 2 + 16,);
|
||||
ALIGN_STK_16(int32_t, A5_buf, ARM_BUF_STRIDE * 2 + 16,);
|
||||
ALIGN_STK_16(int16_t, B5_buf, ARM_BUF_STRIDE * 2 + 16,);
|
||||
int32_t *A5_ptrs[2];
|
||||
int16_t *B5_ptrs[2];
|
||||
for (int i = 0; i < 2; i++) {
|
||||
A5_ptrs[i] = &A5_buf[i * BUF_STRIDE];
|
||||
B5_ptrs[i] = &B5_buf[i * BUF_STRIDE];
|
||||
A5_ptrs[i] = &A5_buf[i * ARM_BUF_STRIDE];
|
||||
B5_ptrs[i] = &B5_buf[i * ARM_BUF_STRIDE];
|
||||
}
|
||||
ALIGN_STK_16(int32_t, A3_buf, BUF_STRIDE * 4 + 16,);
|
||||
ALIGN_STK_16(int16_t, B3_buf, BUF_STRIDE * 4 + 16,);
|
||||
ALIGN_STK_16(int32_t, A3_buf, ARM_BUF_STRIDE * 4 + 16,);
|
||||
ALIGN_STK_16(int16_t, B3_buf, ARM_BUF_STRIDE * 4 + 16,);
|
||||
int32_t *A3_ptrs[4];
|
||||
int16_t *B3_ptrs[4];
|
||||
for (int i = 0; i < 4; i++) {
|
||||
A3_ptrs[i] = &A3_buf[i * BUF_STRIDE];
|
||||
B3_ptrs[i] = &B3_buf[i * BUF_STRIDE];
|
||||
A3_ptrs[i] = &A3_buf[i * ARM_BUF_STRIDE];
|
||||
B3_ptrs[i] = &B3_buf[i * ARM_BUF_STRIDE];
|
||||
}
|
||||
const pixel *src = dst;
|
||||
const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
|
||||
@@ -842,7 +862,7 @@ static void sgr_filter_mix_neon(pixel *dst, const ptrdiff_t stride,
|
||||
|
||||
sgr_box3_vert_neon(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
|
||||
if (--h <= 0)
|
||||
goto vert_1;
|
||||
@@ -854,10 +874,10 @@ static void sgr_filter_mix_neon(pixel *dst, const ptrdiff_t stride,
|
||||
src += PXSTRIDE(stride);
|
||||
sgr_box5_vert_neon(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
|
||||
w, params->sgr.s0, BITDEPTH_MAX);
|
||||
rotate_ab_2(A5_ptrs, B5_ptrs);
|
||||
rotate_neon(A5_ptrs, B5_ptrs, 2);
|
||||
sgr_box3_vert_neon(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
|
||||
if (--h <= 0)
|
||||
goto vert_2;
|
||||
@@ -893,7 +913,7 @@ static void sgr_filter_mix_neon(pixel *dst, const ptrdiff_t stride,
|
||||
|
||||
sgr_box3_vert_neon(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
|
||||
if (--h <= 0)
|
||||
goto vert_1;
|
||||
@@ -912,10 +932,10 @@ static void sgr_filter_mix_neon(pixel *dst, const ptrdiff_t stride,
|
||||
|
||||
sgr_box5_vert_neon(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
|
||||
w, params->sgr.s0, BITDEPTH_MAX);
|
||||
rotate_ab_2(A5_ptrs, B5_ptrs);
|
||||
rotate_neon(A5_ptrs, B5_ptrs, 2);
|
||||
sgr_box3_vert_neon(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
|
||||
if (--h <= 0)
|
||||
goto vert_2;
|
||||
@@ -936,7 +956,7 @@ static void sgr_filter_mix_neon(pixel *dst, const ptrdiff_t stride,
|
||||
|
||||
sgr_box3_vert_neon(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
|
||||
if (--h <= 0)
|
||||
goto odd;
|
||||
@@ -973,7 +993,7 @@ static void sgr_filter_mix_neon(pixel *dst, const ptrdiff_t stride,
|
||||
|
||||
sgr_box3_vert_neon(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
|
||||
if (--h <= 0)
|
||||
goto odd;
|
||||
@@ -1002,7 +1022,7 @@ static void sgr_filter_mix_neon(pixel *dst, const ptrdiff_t stride,
|
||||
lpf_bottom += PXSTRIDE(stride);
|
||||
sgr_box3_vert_neon(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
|
||||
BF(dav1d_sgr_box35_row_h, neon)(sumsq3_ptrs[2], sum3_ptrs[2],
|
||||
sumsq5_ptrs[4], sum5_ptrs[4],
|
||||
@@ -1029,7 +1049,7 @@ vert_2:
|
||||
sum3_ptrs[2] = sum3_ptrs[1];
|
||||
sgr_box3_vert_neon(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
|
||||
sumsq3_ptrs[2] = sumsq3_ptrs[1];
|
||||
sum3_ptrs[2] = sum3_ptrs[1];
|
||||
@@ -1066,7 +1086,7 @@ output_1:
|
||||
w, params->sgr.s0, BITDEPTH_MAX);
|
||||
sgr_box3_vert_neon(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
// Output only one row
|
||||
sgr_finish_mix_neon(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
|
||||
w, 1, params->sgr.w0, params->sgr.w1
|
||||
@@ -1083,16 +1103,14 @@ vert_1:
|
||||
|
||||
sgr_box5_vert_neon(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
|
||||
w, params->sgr.s0, BITDEPTH_MAX);
|
||||
rotate_ab_2(A5_ptrs, B5_ptrs);
|
||||
rotate_neon(A5_ptrs, B5_ptrs, 2);
|
||||
sgr_box3_vert_neon(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
|
||||
w, params->sgr.s1, BITDEPTH_MAX);
|
||||
rotate_ab_4(A3_ptrs, B3_ptrs);
|
||||
rotate_neon(A3_ptrs, B3_ptrs, 4);
|
||||
|
||||
goto output_1;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
|
||||
static ALWAYS_INLINE void loop_restoration_dsp_init_arm(Dav1dLoopRestorationDSPContext *const c, int bpc) {
|
||||
const unsigned flags = dav1d_get_cpu_flags();
|
||||
|
||||
+19
-38
@@ -30,39 +30,10 @@
|
||||
#include "src/mc.h"
|
||||
#include "src/cpu.h"
|
||||
|
||||
#define decl_8tap_gen(decl_name, fn_name, opt) \
|
||||
decl_##decl_name##_fn(BF(dav1d_##fn_name##_8tap_regular, opt)); \
|
||||
decl_##decl_name##_fn(BF(dav1d_##fn_name##_8tap_regular_smooth, opt)); \
|
||||
decl_##decl_name##_fn(BF(dav1d_##fn_name##_8tap_regular_sharp, opt)); \
|
||||
decl_##decl_name##_fn(BF(dav1d_##fn_name##_8tap_smooth_regular, opt)); \
|
||||
decl_##decl_name##_fn(BF(dav1d_##fn_name##_8tap_smooth, opt)); \
|
||||
decl_##decl_name##_fn(BF(dav1d_##fn_name##_8tap_smooth_sharp, opt)); \
|
||||
decl_##decl_name##_fn(BF(dav1d_##fn_name##_8tap_sharp_regular, opt)); \
|
||||
decl_##decl_name##_fn(BF(dav1d_##fn_name##_8tap_sharp_smooth, opt)); \
|
||||
decl_##decl_name##_fn(BF(dav1d_##fn_name##_8tap_sharp, opt))
|
||||
|
||||
#define decl_8tap_fns(opt) \
|
||||
decl_8tap_gen(mc, put, opt); \
|
||||
decl_8tap_gen(mct, prep, opt)
|
||||
|
||||
#define init_8tap_gen(name, opt) \
|
||||
init_##name##_fn(FILTER_2D_8TAP_REGULAR, 8tap_regular, opt); \
|
||||
init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
|
||||
init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP, 8tap_regular_sharp, opt); \
|
||||
init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
|
||||
init_##name##_fn(FILTER_2D_8TAP_SMOOTH, 8tap_smooth, opt); \
|
||||
init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP, 8tap_smooth_sharp, opt); \
|
||||
init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR, 8tap_sharp_regular, opt); \
|
||||
init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH, 8tap_sharp_smooth, opt); \
|
||||
init_##name##_fn(FILTER_2D_8TAP_SHARP, 8tap_sharp, opt)
|
||||
|
||||
#define init_8tap_fns(opt) \
|
||||
init_8tap_gen(mc, opt); \
|
||||
init_8tap_gen(mct, opt)
|
||||
|
||||
decl_8tap_fns(neon);
|
||||
decl_8tap_fns(neon_dotprod);
|
||||
decl_8tap_fns(neon_i8mm);
|
||||
decl_8tap_fns(sve2);
|
||||
|
||||
decl_mc_fn(BF(dav1d_put_bilin, neon));
|
||||
decl_mct_fn(BF(dav1d_prep_bilin, neon));
|
||||
@@ -110,17 +81,27 @@ static ALWAYS_INLINE void mc_dsp_init_arm(Dav1dMCDSPContext *const c) {
|
||||
c->warp8x8t = BF(dav1d_warp_affine_8x8t, neon);
|
||||
c->emu_edge = BF(dav1d_emu_edge, neon);
|
||||
|
||||
#if ARCH_AARCH64 && BITDEPTH == 8
|
||||
#if ARCH_AARCH64
|
||||
#if BITDEPTH == 8
|
||||
#if HAVE_DOTPROD
|
||||
if (!(flags & DAV1D_ARM_CPU_FLAG_DOTPROD)) return;
|
||||
|
||||
init_8tap_fns(neon_dotprod);
|
||||
if (flags & DAV1D_ARM_CPU_FLAG_DOTPROD) {
|
||||
init_8tap_fns(neon_dotprod);
|
||||
}
|
||||
#endif // HAVE_DOTPROD
|
||||
|
||||
#if HAVE_I8MM
|
||||
if (!(flags & DAV1D_ARM_CPU_FLAG_I8MM)) return;
|
||||
|
||||
init_8tap_fns(neon_i8mm);
|
||||
if (flags & DAV1D_ARM_CPU_FLAG_I8MM) {
|
||||
init_8tap_fns(neon_i8mm);
|
||||
}
|
||||
#endif // HAVE_I8MM
|
||||
#endif // ARCH_AARCH64 && BITDEPTH == 8
|
||||
#endif // BITDEPTH == 8
|
||||
|
||||
#if BITDEPTH == 16
|
||||
#if HAVE_SVE2
|
||||
if (flags & DAV1D_ARM_CPU_FLAG_SVE2) {
|
||||
init_8tap_fns(sve2);
|
||||
}
|
||||
#endif // HAVE_SVE2
|
||||
#endif // BITDEPTH == 16
|
||||
#endif // ARCH_AARCH64
|
||||
}
|
||||
|
||||
@@ -25,9 +25,24 @@
|
||||
* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
#include "src/arm/asm-offsets.h"
|
||||
#include "src/cpu.h"
|
||||
#include "src/refmvs.h"
|
||||
|
||||
#if ARCH_AARCH64
|
||||
CHECK_OFFSET(refmvs_frame, iw8, RMVSF_IW8);
|
||||
CHECK_OFFSET(refmvs_frame, ih8, RMVSF_IH8);
|
||||
CHECK_OFFSET(refmvs_frame, mfmv_ref, RMVSF_MFMV_REF);
|
||||
CHECK_OFFSET(refmvs_frame, mfmv_ref2cur, RMVSF_MFMV_REF2CUR);
|
||||
CHECK_OFFSET(refmvs_frame, mfmv_ref2ref, RMVSF_MFMV_REF2REF);
|
||||
CHECK_OFFSET(refmvs_frame, n_mfmvs, RMVSF_N_MFMVS);
|
||||
CHECK_OFFSET(refmvs_frame, rp_ref, RMVSF_RP_REF);
|
||||
CHECK_OFFSET(refmvs_frame, rp_proj, RMVSF_RP_PROJ);
|
||||
CHECK_OFFSET(refmvs_frame, rp_stride, RMVSF_RP_STRIDE);
|
||||
CHECK_OFFSET(refmvs_frame, n_tile_threads, RMVSF_N_TILE_THREADS);
|
||||
#endif
|
||||
|
||||
decl_load_tmvs_fn(dav1d_load_tmvs_neon);
|
||||
decl_save_tmvs_fn(dav1d_save_tmvs_neon);
|
||||
decl_splat_mv_fn(dav1d_splat_mv_neon);
|
||||
|
||||
@@ -36,6 +51,9 @@ static ALWAYS_INLINE void refmvs_dsp_init_arm(Dav1dRefmvsDSPContext *const c) {
|
||||
|
||||
if (!(flags & DAV1D_ARM_CPU_FLAG_NEON)) return;
|
||||
|
||||
#if ARCH_AARCH64
|
||||
c->load_tmvs = dav1d_load_tmvs_neon;
|
||||
#endif
|
||||
c->save_tmvs = dav1d_save_tmvs_neon;
|
||||
c->splat_mv = dav1d_splat_mv_neon;
|
||||
}
|
||||
|
||||
@@ -143,7 +143,7 @@ void bytefn(dav1d_cdef_brow)(Dav1dTaskContext *const tc,
|
||||
edges &= ~CDEF_HAVE_LEFT;
|
||||
edges |= CDEF_HAVE_RIGHT;
|
||||
enum Backup2x8Flags prev_flag = 0;
|
||||
for (int sbx = 0, last_skip = 1; sbx < sb64w; sbx++, edges |= CDEF_HAVE_LEFT) {
|
||||
for (int sbx = 0; sbx < sb64w; sbx++, edges |= CDEF_HAVE_LEFT) {
|
||||
const int sb128x = sbx >> 1;
|
||||
const int sb64_idx = ((by & sbsz) >> 3) + (sbx & 1);
|
||||
const int cdef_idx = lflvl[sb128x].cdef_idx[sb64_idx];
|
||||
@@ -151,7 +151,7 @@ void bytefn(dav1d_cdef_brow)(Dav1dTaskContext *const tc,
|
||||
(!f->frame_hdr->cdef.y_strength[cdef_idx] &&
|
||||
!f->frame_hdr->cdef.uv_strength[cdef_idx]))
|
||||
{
|
||||
last_skip = 1;
|
||||
prev_flag = 0;
|
||||
goto next_sb;
|
||||
}
|
||||
|
||||
@@ -184,10 +184,10 @@ void bytefn(dav1d_cdef_brow)(Dav1dTaskContext *const tc,
|
||||
// go to the next block
|
||||
const uint32_t bx_mask = 3U << (bx & 30);
|
||||
if (!(noskip_mask & bx_mask)) {
|
||||
last_skip = 1;
|
||||
prev_flag = 0;
|
||||
goto next_b;
|
||||
}
|
||||
const int do_left = last_skip ? flag : (prev_flag ^ flag) & flag;
|
||||
const enum Backup2x8Flags do_left = (prev_flag ^ flag) & flag;
|
||||
prev_flag = flag;
|
||||
if (do_left && edges & CDEF_HAVE_LEFT) {
|
||||
// we didn't backup the prefilter data because it wasn't
|
||||
@@ -287,7 +287,6 @@ void bytefn(dav1d_cdef_brow)(Dav1dTaskContext *const tc,
|
||||
|
||||
skip_uv:
|
||||
bit ^= 1;
|
||||
last_skip = 0;
|
||||
|
||||
next_b:
|
||||
bptrs[0] += 8;
|
||||
|
||||
+15
-6
@@ -113,6 +113,7 @@ cdef_filter_block_c(pixel *dst, const ptrdiff_t dst_stride,
|
||||
assert((w == 4 || w == 8) && (h == 4 || h == 8));
|
||||
int16_t tmp_buf[144]; // 12*12 is the maximum value of tmp_stride * (h + 4)
|
||||
int16_t *tmp = tmp_buf + 2 * tmp_stride + 2;
|
||||
const int8_t (*const cdef_dirs)[2] = &dav1d_cdef_directions[dir];
|
||||
|
||||
padding(tmp, tmp_stride, dst, dst_stride, left, top, bottom, w, h, edges);
|
||||
|
||||
@@ -129,7 +130,7 @@ cdef_filter_block_c(pixel *dst, const ptrdiff_t dst_stride,
|
||||
int max = px, min = px;
|
||||
int pri_tap_k = pri_tap;
|
||||
for (int k = 0; k < 2; k++) {
|
||||
const int off1 = dav1d_cdef_directions[dir + 2][k]; // dir
|
||||
const int off1 = cdef_dirs[2][k]; // dir
|
||||
const int p0 = tmp[x + off1];
|
||||
const int p1 = tmp[x - off1];
|
||||
sum += pri_tap_k * constrain(p0 - px, pri_strength, pri_shift);
|
||||
@@ -140,8 +141,8 @@ cdef_filter_block_c(pixel *dst, const ptrdiff_t dst_stride,
|
||||
max = imax(p0, max);
|
||||
min = umin(p1, min);
|
||||
max = imax(p1, max);
|
||||
const int off2 = dav1d_cdef_directions[dir + 4][k]; // dir + 2
|
||||
const int off3 = dav1d_cdef_directions[dir + 0][k]; // dir - 2
|
||||
const int off2 = cdef_dirs[4][k]; // dir + 2
|
||||
const int off3 = cdef_dirs[0][k]; // dir - 2
|
||||
const int s0 = tmp[x + off2];
|
||||
const int s1 = tmp[x - off2];
|
||||
const int s2 = tmp[x + off3];
|
||||
@@ -173,7 +174,7 @@ cdef_filter_block_c(pixel *dst, const ptrdiff_t dst_stride,
|
||||
int sum = 0;
|
||||
int pri_tap_k = pri_tap;
|
||||
for (int k = 0; k < 2; k++) {
|
||||
const int off = dav1d_cdef_directions[dir + 2][k]; // dir
|
||||
const int off = cdef_dirs[2][k]; // dir
|
||||
const int p0 = tmp[x + off];
|
||||
const int p1 = tmp[x - off];
|
||||
sum += pri_tap_k * constrain(p0 - px, pri_strength, pri_shift);
|
||||
@@ -194,8 +195,8 @@ cdef_filter_block_c(pixel *dst, const ptrdiff_t dst_stride,
|
||||
const int px = dst[x];
|
||||
int sum = 0;
|
||||
for (int k = 0; k < 2; k++) {
|
||||
const int off1 = dav1d_cdef_directions[dir + 4][k]; // dir + 2
|
||||
const int off2 = dav1d_cdef_directions[dir + 0][k]; // dir - 2
|
||||
const int off1 = cdef_dirs[4][k]; // dir + 2
|
||||
const int off2 = cdef_dirs[0][k]; // dir - 2
|
||||
const int s0 = tmp[x + off1];
|
||||
const int s1 = tmp[x - off1];
|
||||
const int s2 = tmp[x + off2];
|
||||
@@ -308,8 +309,12 @@ static int cdef_find_dir_c(const pixel *img, const ptrdiff_t stride,
|
||||
#include "src/arm/cdef.h"
|
||||
#elif ARCH_PPC64LE
|
||||
#include "src/ppc/cdef.h"
|
||||
#elif ARCH_RISCV
|
||||
#include "src/riscv/cdef.h"
|
||||
#elif ARCH_X86
|
||||
#include "src/x86/cdef.h"
|
||||
#elif ARCH_LOONGARCH64
|
||||
#include "src/loongarch/cdef.h"
|
||||
#endif
|
||||
#endif
|
||||
|
||||
@@ -324,8 +329,12 @@ COLD void bitfn(dav1d_cdef_dsp_init)(Dav1dCdefDSPContext *const c) {
|
||||
cdef_dsp_init_arm(c);
|
||||
#elif ARCH_PPC64LE
|
||||
cdef_dsp_init_ppc(c);
|
||||
#elif ARCH_RISCV
|
||||
cdef_dsp_init_riscv(c);
|
||||
#elif ARCH_X86
|
||||
cdef_dsp_init_x86(c);
|
||||
#elif ARCH_LOONGARCH64
|
||||
cdef_dsp_init_loongarch(c);
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -787,63 +787,53 @@ static const CdfCoefContext default_coef_cdf[4] = {
|
||||
}, .eob_hi_bit = {
|
||||
{
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16961) },
|
||||
{ CDF1(17223) }, { CDF1( 7621) }, { CDF1(16384) },
|
||||
{ CDF1(16961) }, { CDF1(17223) }, { CDF1( 7621) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(19069) },
|
||||
{ CDF1(22525) }, { CDF1(13377) }, { CDF1(16384) },
|
||||
{ CDF1(19069) }, { CDF1(22525) }, { CDF1(13377) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(20401) },
|
||||
{ CDF1(17025) }, { CDF1(12845) }, { CDF1(12873) },
|
||||
{ CDF1(14094) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(20401) }, { CDF1(17025) }, { CDF1(12845) },
|
||||
{ CDF1(12873) }, { CDF1(14094) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(20681) },
|
||||
{ CDF1(20701) }, { CDF1(15250) }, { CDF1(15017) },
|
||||
{ CDF1(14928) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(20681) }, { CDF1(20701) }, { CDF1(15250) },
|
||||
{ CDF1(15017) }, { CDF1(14928) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(23905) },
|
||||
{ CDF1(17194) }, { CDF1(16170) }, { CDF1(17695) },
|
||||
{ CDF1(13826) }, { CDF1(15810) }, { CDF1(12036) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(23905) }, { CDF1(17194) }, { CDF1(16170) },
|
||||
{ CDF1(17695) }, { CDF1(13826) }, { CDF1(15810) },
|
||||
{ CDF1(12036) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(23959) },
|
||||
{ CDF1(20799) }, { CDF1(19021) }, { CDF1(16203) },
|
||||
{ CDF1(17886) }, { CDF1(14144) }, { CDF1(12010) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(23959) }, { CDF1(20799) }, { CDF1(19021) },
|
||||
{ CDF1(16203) }, { CDF1(17886) }, { CDF1(14144) },
|
||||
{ CDF1(12010) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(27399) },
|
||||
{ CDF1(16327) }, { CDF1(18071) }, { CDF1(19584) },
|
||||
{ CDF1(20721) }, { CDF1(18432) }, { CDF1(19560) },
|
||||
{ CDF1(10150) }, { CDF1( 8805) },
|
||||
{ CDF1(27399) }, { CDF1(16327) }, { CDF1(18071) },
|
||||
{ CDF1(19584) }, { CDF1(20721) }, { CDF1(18432) },
|
||||
{ CDF1(19560) }, { CDF1(10150) }, { CDF1( 8805) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(24932) },
|
||||
{ CDF1(20833) }, { CDF1(12027) }, { CDF1(16670) },
|
||||
{ CDF1(19914) }, { CDF1(15106) }, { CDF1(17662) },
|
||||
{ CDF1(13783) }, { CDF1(28756) },
|
||||
{ CDF1(24932) }, { CDF1(20833) }, { CDF1(12027) },
|
||||
{ CDF1(16670) }, { CDF1(19914) }, { CDF1(15106) },
|
||||
{ CDF1(17662) }, { CDF1(13783) }, { CDF1(28756) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(23406) },
|
||||
{ CDF1(21845) }, { CDF1(18432) }, { CDF1(16384) },
|
||||
{ CDF1(17096) }, { CDF1(12561) }, { CDF1(17320) },
|
||||
{ CDF1(22395) }, { CDF1(21370) },
|
||||
{ CDF1(23406) }, { CDF1(21845) }, { CDF1(18432) },
|
||||
{ CDF1(16384) }, { CDF1(17096) }, { CDF1(12561) },
|
||||
{ CDF1(17320) }, { CDF1(22395) }, { CDF1(21370) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
},
|
||||
}, .eob_base_tok = {
|
||||
@@ -1600,63 +1590,53 @@ static const CdfCoefContext default_coef_cdf[4] = {
|
||||
}, .eob_hi_bit = {
|
||||
{
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(17471) },
|
||||
{ CDF1(20223) }, { CDF1(11357) }, { CDF1(16384) },
|
||||
{ CDF1(17471) }, { CDF1(20223) }, { CDF1(11357) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(20335) },
|
||||
{ CDF1(21667) }, { CDF1(14818) }, { CDF1(16384) },
|
||||
{ CDF1(20335) }, { CDF1(21667) }, { CDF1(14818) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(20430) },
|
||||
{ CDF1(20662) }, { CDF1(15367) }, { CDF1(16970) },
|
||||
{ CDF1(14657) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(20430) }, { CDF1(20662) }, { CDF1(15367) },
|
||||
{ CDF1(16970) }, { CDF1(14657) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(22117) },
|
||||
{ CDF1(22028) }, { CDF1(18650) }, { CDF1(16042) },
|
||||
{ CDF1(15885) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(22117) }, { CDF1(22028) }, { CDF1(18650) },
|
||||
{ CDF1(16042) }, { CDF1(15885) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(22409) },
|
||||
{ CDF1(21012) }, { CDF1(15650) }, { CDF1(17395) },
|
||||
{ CDF1(15469) }, { CDF1(20205) }, { CDF1(19511) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(22409) }, { CDF1(21012) }, { CDF1(15650) },
|
||||
{ CDF1(17395) }, { CDF1(15469) }, { CDF1(20205) },
|
||||
{ CDF1(19511) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(24220) },
|
||||
{ CDF1(22480) }, { CDF1(17737) }, { CDF1(18916) },
|
||||
{ CDF1(19268) }, { CDF1(18412) }, { CDF1(18844) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(24220) }, { CDF1(22480) }, { CDF1(17737) },
|
||||
{ CDF1(18916) }, { CDF1(19268) }, { CDF1(18412) },
|
||||
{ CDF1(18844) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(25991) },
|
||||
{ CDF1(20314) }, { CDF1(17731) }, { CDF1(19678) },
|
||||
{ CDF1(18649) }, { CDF1(17307) }, { CDF1(21798) },
|
||||
{ CDF1(17549) }, { CDF1(15630) },
|
||||
{ CDF1(25991) }, { CDF1(20314) }, { CDF1(17731) },
|
||||
{ CDF1(19678) }, { CDF1(18649) }, { CDF1(17307) },
|
||||
{ CDF1(21798) }, { CDF1(17549) }, { CDF1(15630) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(26585) },
|
||||
{ CDF1(21469) }, { CDF1(20432) }, { CDF1(17735) },
|
||||
{ CDF1(19280) }, { CDF1(15235) }, { CDF1(20297) },
|
||||
{ CDF1(22471) }, { CDF1(28997) },
|
||||
{ CDF1(26585) }, { CDF1(21469) }, { CDF1(20432) },
|
||||
{ CDF1(17735) }, { CDF1(19280) }, { CDF1(15235) },
|
||||
{ CDF1(20297) }, { CDF1(22471) }, { CDF1(28997) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(26605) },
|
||||
{ CDF1(11304) }, { CDF1(16726) }, { CDF1(16560) },
|
||||
{ CDF1(20866) }, { CDF1(23524) }, { CDF1(19878) },
|
||||
{ CDF1(13469) }, { CDF1(23084) },
|
||||
{ CDF1(26605) }, { CDF1(11304) }, { CDF1(16726) },
|
||||
{ CDF1(16560) }, { CDF1(20866) }, { CDF1(23524) },
|
||||
{ CDF1(19878) }, { CDF1(13469) }, { CDF1(23084) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
},
|
||||
}, .eob_base_tok = {
|
||||
@@ -2413,63 +2393,53 @@ static const CdfCoefContext default_coef_cdf[4] = {
|
||||
}, .eob_hi_bit = {
|
||||
{
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(18983) },
|
||||
{ CDF1(20512) }, { CDF1(14885) }, { CDF1(16384) },
|
||||
{ CDF1(18983) }, { CDF1(20512) }, { CDF1(14885) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(20090) },
|
||||
{ CDF1(19444) }, { CDF1(17286) }, { CDF1(16384) },
|
||||
{ CDF1(20090) }, { CDF1(19444) }, { CDF1(17286) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(19139) },
|
||||
{ CDF1(21487) }, { CDF1(18959) }, { CDF1(20910) },
|
||||
{ CDF1(19089) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(19139) }, { CDF1(21487) }, { CDF1(18959) },
|
||||
{ CDF1(20910) }, { CDF1(19089) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(20536) },
|
||||
{ CDF1(20664) }, { CDF1(20625) }, { CDF1(19123) },
|
||||
{ CDF1(14862) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(20536) }, { CDF1(20664) }, { CDF1(20625) },
|
||||
{ CDF1(19123) }, { CDF1(14862) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(19833) },
|
||||
{ CDF1(21502) }, { CDF1(17485) }, { CDF1(20267) },
|
||||
{ CDF1(18353) }, { CDF1(23329) }, { CDF1(21478) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(19833) }, { CDF1(21502) }, { CDF1(17485) },
|
||||
{ CDF1(20267) }, { CDF1(18353) }, { CDF1(23329) },
|
||||
{ CDF1(21478) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(22041) },
|
||||
{ CDF1(23434) }, { CDF1(20001) }, { CDF1(20554) },
|
||||
{ CDF1(20951) }, { CDF1(20145) }, { CDF1(15562) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(22041) }, { CDF1(23434) }, { CDF1(20001) },
|
||||
{ CDF1(20554) }, { CDF1(20951) }, { CDF1(20145) },
|
||||
{ CDF1(15562) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(23312) },
|
||||
{ CDF1(21607) }, { CDF1(16526) }, { CDF1(18957) },
|
||||
{ CDF1(18034) }, { CDF1(18934) }, { CDF1(24247) },
|
||||
{ CDF1(16921) }, { CDF1(17080) },
|
||||
{ CDF1(23312) }, { CDF1(21607) }, { CDF1(16526) },
|
||||
{ CDF1(18957) }, { CDF1(18034) }, { CDF1(18934) },
|
||||
{ CDF1(24247) }, { CDF1(16921) }, { CDF1(17080) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(26579) },
|
||||
{ CDF1(24910) }, { CDF1(18637) }, { CDF1(19800) },
|
||||
{ CDF1(20388) }, { CDF1( 9887) }, { CDF1(15642) },
|
||||
{ CDF1(30198) }, { CDF1(24721) },
|
||||
{ CDF1(26579) }, { CDF1(24910) }, { CDF1(18637) },
|
||||
{ CDF1(19800) }, { CDF1(20388) }, { CDF1( 9887) },
|
||||
{ CDF1(15642) }, { CDF1(30198) }, { CDF1(24721) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(26998) },
|
||||
{ CDF1(16737) }, { CDF1(17838) }, { CDF1(18922) },
|
||||
{ CDF1(19515) }, { CDF1(18636) }, { CDF1(17333) },
|
||||
{ CDF1(15776) }, { CDF1(22658) },
|
||||
{ CDF1(26998) }, { CDF1(16737) }, { CDF1(17838) },
|
||||
{ CDF1(18922) }, { CDF1(19515) }, { CDF1(18636) },
|
||||
{ CDF1(17333) }, { CDF1(15776) }, { CDF1(22658) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
},
|
||||
}, .eob_base_tok = {
|
||||
@@ -3226,63 +3196,53 @@ static const CdfCoefContext default_coef_cdf[4] = {
|
||||
}, .eob_hi_bit = {
|
||||
{
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(20177) },
|
||||
{ CDF1(20789) }, { CDF1(20262) }, { CDF1(16384) },
|
||||
{ CDF1(20177) }, { CDF1(20789) }, { CDF1(20262) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(21416) },
|
||||
{ CDF1(20855) }, { CDF1(23410) }, { CDF1(16384) },
|
||||
{ CDF1(21416) }, { CDF1(20855) }, { CDF1(23410) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(20238) },
|
||||
{ CDF1(21057) }, { CDF1(19159) }, { CDF1(22337) },
|
||||
{ CDF1(20159) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(20238) }, { CDF1(21057) }, { CDF1(19159) },
|
||||
{ CDF1(22337) }, { CDF1(20159) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(20125) },
|
||||
{ CDF1(20559) }, { CDF1(21707) }, { CDF1(22296) },
|
||||
{ CDF1(17333) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(20125) }, { CDF1(20559) }, { CDF1(21707) },
|
||||
{ CDF1(22296) }, { CDF1(17333) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(19941) },
|
||||
{ CDF1(20527) }, { CDF1(21470) }, { CDF1(22487) },
|
||||
{ CDF1(19558) }, { CDF1(22354) }, { CDF1(20331) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(19941) }, { CDF1(20527) }, { CDF1(21470) },
|
||||
{ CDF1(22487) }, { CDF1(19558) }, { CDF1(22354) },
|
||||
{ CDF1(20331) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(22752) },
|
||||
{ CDF1(25006) }, { CDF1(22075) }, { CDF1(21576) },
|
||||
{ CDF1(17740) }, { CDF1(21690) }, { CDF1(19211) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(22752) }, { CDF1(25006) }, { CDF1(22075) },
|
||||
{ CDF1(21576) }, { CDF1(17740) }, { CDF1(21690) },
|
||||
{ CDF1(19211) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(21442) },
|
||||
{ CDF1(22358) }, { CDF1(18503) }, { CDF1(20291) },
|
||||
{ CDF1(19945) }, { CDF1(21294) }, { CDF1(21178) },
|
||||
{ CDF1(19400) }, { CDF1(10556) },
|
||||
{ CDF1(21442) }, { CDF1(22358) }, { CDF1(18503) },
|
||||
{ CDF1(20291) }, { CDF1(19945) }, { CDF1(21294) },
|
||||
{ CDF1(21178) }, { CDF1(19400) }, { CDF1(10556) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(24648) },
|
||||
{ CDF1(24949) }, { CDF1(20708) }, { CDF1(23905) },
|
||||
{ CDF1(20501) }, { CDF1( 9558) }, { CDF1( 9423) },
|
||||
{ CDF1(30365) }, { CDF1(19253) },
|
||||
{ CDF1(24648) }, { CDF1(24949) }, { CDF1(20708) },
|
||||
{ CDF1(23905) }, { CDF1(20501) }, { CDF1( 9558) },
|
||||
{ CDF1( 9423) }, { CDF1(30365) }, { CDF1(19253) },
|
||||
},
|
||||
}, {
|
||||
{
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(26064) },
|
||||
{ CDF1(22098) }, { CDF1(19613) }, { CDF1(20525) },
|
||||
{ CDF1(17595) }, { CDF1(16618) }, { CDF1(20497) },
|
||||
{ CDF1(18989) }, { CDF1(15513) },
|
||||
{ CDF1(26064) }, { CDF1(22098) }, { CDF1(19613) },
|
||||
{ CDF1(20525) }, { CDF1(17595) }, { CDF1(16618) },
|
||||
{ CDF1(20497) }, { CDF1(18989) }, { CDF1(15513) },
|
||||
}, {
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) }, { CDF1(16384) },
|
||||
{ CDF1(16384) }, { CDF1(16384) },
|
||||
},
|
||||
},
|
||||
}, .eob_base_tok = {
|
||||
@@ -3979,7 +3939,7 @@ void dav1d_cdf_thread_update(const Dav1dFrameHeader *const hdr,
|
||||
update_cdf_4d(N_TX_SIZES, 2, 4, 2, coef.eob_base_tok);
|
||||
update_cdf_4d(N_TX_SIZES, 2, 41 /*42*/, 3, coef.base_tok);
|
||||
update_cdf_4d(4, 2, 21, 3, coef.br_tok);
|
||||
update_cdf_4d(N_TX_SIZES, 2, 11 /*22*/, 1, coef.eob_hi_bit);
|
||||
update_cdf_4d(N_TX_SIZES, 2, 9, 1, coef.eob_hi_bit);
|
||||
update_cdf_3d(N_TX_SIZES, 13, 1, coef.skip);
|
||||
update_cdf_3d(2, 3, 1, coef.dc_sign);
|
||||
|
||||
|
||||
@@ -105,7 +105,7 @@ typedef struct CdfCoefContext {
|
||||
ALIGN(uint16_t eob_base_tok[N_TX_SIZES][2][4][4], 8);
|
||||
ALIGN(uint16_t base_tok[N_TX_SIZES][2][41][4], 8);
|
||||
ALIGN(uint16_t br_tok[4 /*5*/][2][21][4], 8);
|
||||
ALIGN(uint16_t eob_hi_bit[N_TX_SIZES][2][11 /*22*/][2], 4);
|
||||
ALIGN(uint16_t eob_hi_bit[N_TX_SIZES][2][9][2], 4);
|
||||
ALIGN(uint16_t skip[N_TX_SIZES][13][2], 4);
|
||||
ALIGN(uint16_t dc_sign[2][3][2], 4);
|
||||
} CdfCoefContext;
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
*/
|
||||
#include "config.h"
|
||||
|
||||
#include <errno.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#include "src/cpu.h"
|
||||
@@ -33,20 +34,28 @@
|
||||
|
||||
#ifdef _WIN32
|
||||
#include <windows.h>
|
||||
#elif defined(__APPLE__)
|
||||
#endif
|
||||
#ifdef __APPLE__
|
||||
#include <sys/sysctl.h>
|
||||
#include <sys/types.h>
|
||||
#else
|
||||
#include <pthread.h>
|
||||
#endif
|
||||
#if HAVE_UNISTD_H
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#ifdef HAVE_PTHREAD_NP_H
|
||||
#if HAVE_PTHREAD_GETAFFINITY_NP
|
||||
#include <pthread.h>
|
||||
#if HAVE_PTHREAD_NP_H
|
||||
#include <pthread_np.h>
|
||||
#endif
|
||||
#if defined(__FreeBSD__)
|
||||
#define cpu_set_t cpuset_t
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#if HAVE_GETAUXVAL || HAVE_ELF_AUX_INFO
|
||||
#include <sys/auxv.h>
|
||||
#endif
|
||||
|
||||
unsigned dav1d_cpu_flags = 0U;
|
||||
unsigned dav1d_cpu_flags_mask = ~0U;
|
||||
@@ -87,7 +96,7 @@ COLD int dav1d_num_logical_processors(Dav1dContext *const c) {
|
||||
GetNativeSystemInfo(&system_info);
|
||||
return system_info.dwNumberOfProcessors;
|
||||
#endif
|
||||
#elif defined(HAVE_PTHREAD_GETAFFINITY_NP) && defined(CPU_COUNT)
|
||||
#elif HAVE_PTHREAD_GETAFFINITY_NP && defined(CPU_COUNT)
|
||||
cpu_set_t affinity;
|
||||
if (!pthread_getaffinity_np(pthread_self(), sizeof(affinity), &affinity))
|
||||
return CPU_COUNT(&affinity);
|
||||
@@ -103,3 +112,18 @@ COLD int dav1d_num_logical_processors(Dav1dContext *const c) {
|
||||
dav1d_log(c, "Unable to detect thread count, defaulting to single-threaded mode\n");
|
||||
return 1;
|
||||
}
|
||||
|
||||
COLD unsigned long dav1d_getauxval(unsigned long type) {
|
||||
#if HAVE_GETAUXVAL
|
||||
return getauxval(type);
|
||||
#elif HAVE_ELF_AUX_INFO
|
||||
unsigned long aux = 0;
|
||||
int ret = elf_aux_info(type, &aux, sizeof(aux));
|
||||
if (ret != 0)
|
||||
errno = ret;
|
||||
return aux;
|
||||
#else
|
||||
errno = ENOSYS;
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -53,13 +53,11 @@ EXTERN unsigned dav1d_cpu_flags_mask;
|
||||
void dav1d_init_cpu(void);
|
||||
DAV1D_API void dav1d_set_cpu_flags_mask(unsigned mask);
|
||||
int dav1d_num_logical_processors(Dav1dContext *c);
|
||||
unsigned long dav1d_getauxval(unsigned long);
|
||||
|
||||
static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
|
||||
unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
|
||||
static ALWAYS_INLINE unsigned dav1d_get_default_cpu_flags(void) {
|
||||
unsigned flags = 0;
|
||||
|
||||
#if TRIM_DSP_FUNCTIONS
|
||||
/* Since this function is inlined, unconditionally setting a flag here will
|
||||
* enable dead code elimination in the calling function. */
|
||||
#if ARCH_AARCH64 || ARCH_ARM
|
||||
#if defined(__ARM_NEON) || defined(__APPLE__) || defined(_WIN32) || ARCH_AARCH64
|
||||
flags |= DAV1D_ARM_CPU_FLAG_NEON;
|
||||
@@ -119,6 +117,17 @@ static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
|
||||
flags |= DAV1D_X86_CPU_FLAG_SSE2;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
return flags;
|
||||
}
|
||||
|
||||
static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
|
||||
unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
|
||||
|
||||
#if TRIM_DSP_FUNCTIONS
|
||||
/* Since this function is inlined, unconditionally setting a flag here will
|
||||
* enable dead code elimination in the calling function. */
|
||||
flags |= dav1d_get_default_cpu_flags();
|
||||
#endif
|
||||
|
||||
return flags;
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
/*
|
||||
* Copyright © 2024, VideoLAN and dav1d authors
|
||||
* Copyright © 2024, Two Orioles, LLC
|
||||
* All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
||||
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
* WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
|
||||
* ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
|
||||
* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
|
||||
* ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
|
||||
* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
#include "config.h"
|
||||
|
||||
#include <string.h>
|
||||
|
||||
#include "ctx.h"
|
||||
|
||||
static void memset_w1(void *const ptr, const int value) {
|
||||
set_ctx1((uint8_t *) ptr, 0, value);
|
||||
}
|
||||
|
||||
static void memset_w2(void *const ptr, const int value) {
|
||||
set_ctx2((uint8_t *) ptr, 0, value);
|
||||
}
|
||||
|
||||
static void memset_w4(void *const ptr, const int value) {
|
||||
set_ctx4((uint8_t *) ptr, 0, value);
|
||||
}
|
||||
|
||||
static void memset_w8(void *const ptr, const int value) {
|
||||
set_ctx8((uint8_t *) ptr, 0, value);
|
||||
}
|
||||
|
||||
static void memset_w16(void *const ptr, const int value) {
|
||||
set_ctx16((uint8_t *) ptr, 0, value);
|
||||
}
|
||||
|
||||
static void memset_w32(void *const ptr, const int value) {
|
||||
set_ctx32((uint8_t *) ptr, 0, value);
|
||||
}
|
||||
|
||||
const dav1d_memset_pow2_fn dav1d_memset_pow2[6] = {
|
||||
memset_w1,
|
||||
memset_w2,
|
||||
memset_w4,
|
||||
memset_w8,
|
||||
memset_w16,
|
||||
memset_w32
|
||||
};
|
||||
@@ -31,61 +31,59 @@
|
||||
#include <stdint.h>
|
||||
|
||||
#include "common/attributes.h"
|
||||
#include "common/intops.h"
|
||||
|
||||
union alias64 { uint64_t u64; uint8_t u8[8]; } ATTR_ALIAS;
|
||||
union alias32 { uint32_t u32; uint8_t u8[4]; } ATTR_ALIAS;
|
||||
union alias16 { uint16_t u16; uint8_t u8[2]; } ATTR_ALIAS;
|
||||
union alias8 { uint8_t u8; } ATTR_ALIAS;
|
||||
|
||||
#define set_ctx_rep4(type, var, off, val) do { \
|
||||
const uint64_t const_val = val; \
|
||||
((union alias64 *) &var[off + 0])->u64 = const_val; \
|
||||
((union alias64 *) &var[off + 8])->u64 = const_val; \
|
||||
((union alias64 *) &var[off + 16])->u64 = const_val; \
|
||||
((union alias64 *) &var[off + 24])->u64 = const_val; \
|
||||
typedef void (*dav1d_memset_pow2_fn)(void *ptr, int value);
|
||||
EXTERN const dav1d_memset_pow2_fn dav1d_memset_pow2[6];
|
||||
|
||||
static inline void dav1d_memset_likely_pow2(void *const ptr, const int value, const int n) {
|
||||
assert(n >= 1 && n <= 32);
|
||||
if ((n&(n-1)) == 0) {
|
||||
dav1d_memset_pow2[ulog2(n)](ptr, value);
|
||||
} else {
|
||||
memset(ptr, value, n);
|
||||
}
|
||||
}
|
||||
|
||||
// For smaller sizes use multiplication to broadcast bytes. memset misbehaves on the smaller sizes.
|
||||
// For the larger sizes, we want to use memset to get access to vector operations.
|
||||
#define set_ctx1(var, off, val) \
|
||||
((union alias8 *) &(var)[off])->u8 = (val) * 0x01
|
||||
#define set_ctx2(var, off, val) \
|
||||
((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
|
||||
#define set_ctx4(var, off, val) \
|
||||
((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
|
||||
#define set_ctx8(var, off, val) \
|
||||
((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
|
||||
#define set_ctx16(var, off, val) do { \
|
||||
memset(&(var)[off], val, 16); \
|
||||
} while (0)
|
||||
#define set_ctx_rep2(type, var, off, val) do { \
|
||||
const uint64_t const_val = val; \
|
||||
((union alias64 *) &var[off + 0])->u64 = const_val; \
|
||||
((union alias64 *) &var[off + 8])->u64 = const_val; \
|
||||
#define set_ctx32(var, off, val) do { \
|
||||
memset(&(var)[off], val, 32); \
|
||||
} while (0)
|
||||
#define set_ctx_rep1(typesz, var, off, val) \
|
||||
((union alias##typesz *) &var[off])->u##typesz = val
|
||||
#define case_set(var, dir, diridx, off) \
|
||||
#define case_set(var) \
|
||||
switch (var) { \
|
||||
case 1: set_ctx( 8, dir, diridx, off, 0x01, set_ctx_rep1); break; \
|
||||
case 2: set_ctx(16, dir, diridx, off, 0x0101, set_ctx_rep1); break; \
|
||||
case 4: set_ctx(32, dir, diridx, off, 0x01010101U, set_ctx_rep1); break; \
|
||||
case 8: set_ctx(64, dir, diridx, off, 0x0101010101010101ULL, set_ctx_rep1); break; \
|
||||
case 16: set_ctx( , dir, diridx, off, 0x0101010101010101ULL, set_ctx_rep2); break; \
|
||||
case 32: set_ctx( , dir, diridx, off, 0x0101010101010101ULL, set_ctx_rep4); break; \
|
||||
case 0: set_ctx(set_ctx1); break; \
|
||||
case 1: set_ctx(set_ctx2); break; \
|
||||
case 2: set_ctx(set_ctx4); break; \
|
||||
case 3: set_ctx(set_ctx8); break; \
|
||||
case 4: set_ctx(set_ctx16); break; \
|
||||
case 5: set_ctx(set_ctx32); break; \
|
||||
default: assert(0); \
|
||||
}
|
||||
#define case_set_upto16(var, dir, diridx, off) \
|
||||
#define case_set_upto16(var) \
|
||||
switch (var) { \
|
||||
case 1: set_ctx( 8, dir, diridx, off, 0x01, set_ctx_rep1); break; \
|
||||
case 2: set_ctx(16, dir, diridx, off, 0x0101, set_ctx_rep1); break; \
|
||||
case 4: set_ctx(32, dir, diridx, off, 0x01010101U, set_ctx_rep1); break; \
|
||||
case 8: set_ctx(64, dir, diridx, off, 0x0101010101010101ULL, set_ctx_rep1); break; \
|
||||
case 16: set_ctx( , dir, diridx, off, 0x0101010101010101ULL, set_ctx_rep2); break; \
|
||||
}
|
||||
#define case_set_upto32_with_default(var, dir, diridx, off) \
|
||||
switch (var) { \
|
||||
case 1: set_ctx( 8, dir, diridx, off, 0x01, set_ctx_rep1); break; \
|
||||
case 2: set_ctx(16, dir, diridx, off, 0x0101, set_ctx_rep1); break; \
|
||||
case 4: set_ctx(32, dir, diridx, off, 0x01010101U, set_ctx_rep1); break; \
|
||||
case 8: set_ctx(64, dir, diridx, off, 0x0101010101010101ULL, set_ctx_rep1); break; \
|
||||
case 16: set_ctx( , dir, diridx, off, 0x0101010101010101ULL, set_ctx_rep2); break; \
|
||||
case 32: set_ctx( , dir, diridx, off, 0x0101010101010101ULL, set_ctx_rep4); break; \
|
||||
default: default_memset(dir, diridx, off, var); break; \
|
||||
}
|
||||
#define case_set_upto16_with_default(var, dir, diridx, off) \
|
||||
switch (var) { \
|
||||
case 1: set_ctx( 8, dir, diridx, off, 0x01, set_ctx_rep1); break; \
|
||||
case 2: set_ctx(16, dir, diridx, off, 0x0101, set_ctx_rep1); break; \
|
||||
case 4: set_ctx(32, dir, diridx, off, 0x01010101U, set_ctx_rep1); break; \
|
||||
case 8: set_ctx(64, dir, diridx, off, 0x0101010101010101ULL, set_ctx_rep1); break; \
|
||||
case 16: set_ctx( , dir, diridx, off, 0x0101010101010101ULL, set_ctx_rep2); break; \
|
||||
default: default_memset(dir, diridx, off, var); break; \
|
||||
case 0: set_ctx(set_ctx1); break; \
|
||||
case 1: set_ctx(set_ctx2); break; \
|
||||
case 2: set_ctx(set_ctx4); break; \
|
||||
case 3: set_ctx(set_ctx8); break; \
|
||||
case 4: set_ctx(set_ctx16); break; \
|
||||
default: assert(0); \
|
||||
}
|
||||
|
||||
#endif /* DAV1D_SRC_CTX_H */
|
||||
|
||||
+107
-128
@@ -161,14 +161,8 @@ static void read_tx_tree(Dav1dTaskContext *const t,
|
||||
}
|
||||
t->by -= txsh;
|
||||
} else {
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir tx, off, is_split ? TX_4X4 : mul * txh)
|
||||
case_set_upto16(t_dim->h, l., 1, by4);
|
||||
#undef set_ctx
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir tx, off, is_split ? TX_4X4 : mul * txw)
|
||||
case_set_upto16(t_dim->w, a->, 0, bx4);
|
||||
#undef set_ctx
|
||||
dav1d_memset_pow2[t_dim->lw](&t->a->tx[bx4], is_split ? TX_4X4 : txw);
|
||||
dav1d_memset_pow2[t_dim->lh](&t->l.tx[by4], is_split ? TX_4X4 : txh);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -419,17 +413,17 @@ static void order_palette(const uint8_t *pal_idx, const ptrdiff_t stride,
|
||||
|
||||
static void read_pal_indices(Dav1dTaskContext *const t,
|
||||
uint8_t *const pal_idx,
|
||||
const Av1Block *const b, const int pl,
|
||||
const int pal_sz, const int pl,
|
||||
const int w4, const int h4,
|
||||
const int bw4, const int bh4)
|
||||
{
|
||||
Dav1dTileState *const ts = t->ts;
|
||||
const ptrdiff_t stride = bw4 * 4;
|
||||
assert(pal_idx);
|
||||
pixel *const pal_tmp = t->scratch.pal_idx_uv;
|
||||
pal_tmp[0] = dav1d_msac_decode_uniform(&ts->msac, b->pal_sz[pl]);
|
||||
uint8_t *const pal_tmp = t->scratch.pal_idx_uv;
|
||||
pal_tmp[0] = dav1d_msac_decode_uniform(&ts->msac, pal_sz);
|
||||
uint16_t (*const color_map_cdf)[8] =
|
||||
ts->cdf.m.color_map[pl][b->pal_sz[pl] - 2];
|
||||
ts->cdf.m.color_map[pl][pal_sz - 2];
|
||||
uint8_t (*const order)[8] = t->scratch.pal_order;
|
||||
uint8_t *const ctx = t->scratch.pal_ctx;
|
||||
for (int i = 1; i < 4 * (w4 + h4) - 1; i++) {
|
||||
@@ -439,7 +433,7 @@ static void read_pal_indices(Dav1dTaskContext *const t,
|
||||
order_palette(pal_tmp, stride, i, first, last, order, ctx);
|
||||
for (int j = first, m = 0; j >= last; j--, m++) {
|
||||
const int color_idx = dav1d_msac_decode_symbol_adapt8(&ts->msac,
|
||||
color_map_cdf[ctx[m]], b->pal_sz[pl] - 1);
|
||||
color_map_cdf[ctx[m]], pal_sz - 1);
|
||||
pal_tmp[(i - j) * stride + j] = order[m][color_idx];
|
||||
}
|
||||
}
|
||||
@@ -464,19 +458,13 @@ static void read_vartx_tree(Dav1dTaskContext *const t,
|
||||
{
|
||||
b->max_ytx = b->uvtx = TX_4X4;
|
||||
if (f->frame_hdr->txfm_mode == DAV1D_TX_SWITCHABLE) {
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir tx, off, TX_4X4)
|
||||
case_set(bh4, l., 1, by4);
|
||||
case_set(bw4, a->, 0, bx4);
|
||||
#undef set_ctx
|
||||
dav1d_memset_pow2[b_dim[2]](&t->a->tx[bx4], TX_4X4);
|
||||
dav1d_memset_pow2[b_dim[3]](&t->l.tx[by4], TX_4X4);
|
||||
}
|
||||
} else if (f->frame_hdr->txfm_mode != DAV1D_TX_SWITCHABLE || b->skip) {
|
||||
if (f->frame_hdr->txfm_mode == DAV1D_TX_SWITCHABLE) {
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir tx, off, mul * b_dim[2 + diridx])
|
||||
case_set(bh4, l., 1, by4);
|
||||
case_set(bw4, a->, 0, bx4);
|
||||
#undef set_ctx
|
||||
dav1d_memset_pow2[b_dim[2]](&t->a->tx[bx4], b_dim[2 + 0]);
|
||||
dav1d_memset_pow2[b_dim[3]](&t->l.tx[by4], b_dim[2 + 1]);
|
||||
}
|
||||
b->uvtx = dav1d_max_txfm_size_for_bs[bs][f->cur.p.layout];
|
||||
} else {
|
||||
@@ -696,8 +684,7 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
const enum BlockLevel bl,
|
||||
const enum BlockSize bs,
|
||||
const enum BlockPartition bp,
|
||||
const enum EdgeFlags intra_edge_flags)
|
||||
{
|
||||
const enum EdgeFlags intra_edge_flags) {
|
||||
Dav1dTileState *const ts = t->ts;
|
||||
const Dav1dFrameContext *const f = t->f;
|
||||
Av1Block b_mem, *const b = t->frame_thread.pass ?
|
||||
@@ -722,11 +709,13 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
|
||||
const enum IntraPredMode y_mode_nofilt =
|
||||
b->y_mode == FILTER_PRED ? DC_PRED : b->y_mode;
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir mode, off, mul * y_mode_nofilt); \
|
||||
rep_macro(type, t->dir intra, off, mul)
|
||||
case_set(bh4, l., 1, by4);
|
||||
case_set(bw4, a->, 0, bx4);
|
||||
#define set_ctx(rep_macro) \
|
||||
rep_macro(edge->mode, off, y_mode_nofilt); \
|
||||
rep_macro(edge->intra, off, 1)
|
||||
BlockContext *edge = t->a;
|
||||
for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
|
||||
case_set(b_dim[2 + i]);
|
||||
}
|
||||
#undef set_ctx
|
||||
if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
|
||||
refmvs_block *const r = &t->rt.r[(t->by & 31) + 5 + bh4 - 1][t->bx];
|
||||
@@ -742,17 +731,15 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
}
|
||||
|
||||
if (has_chroma) {
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir uvmode, off, mul * b->uv_mode)
|
||||
case_set(cbh4, l., 1, cby4);
|
||||
case_set(cbw4, a->, 0, cbx4);
|
||||
#undef set_ctx
|
||||
uint8_t uv_mode = b->uv_mode;
|
||||
dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], uv_mode);
|
||||
dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], uv_mode);
|
||||
}
|
||||
} else {
|
||||
if (IS_INTER_OR_SWITCH(f->frame_hdr) /* not intrabc */ &&
|
||||
b->comp_type == COMP_INTER_NONE && b->motion_mode == MM_WARP)
|
||||
{
|
||||
if (b->matrix[0] == SHRT_MIN) {
|
||||
if (b->matrix[0] == INT16_MIN) {
|
||||
t->warpmv.type = DAV1D_WM_TYPE_IDENTITY;
|
||||
} else {
|
||||
t->warpmv.type = DAV1D_WM_TYPE_AFFINE;
|
||||
@@ -784,13 +771,15 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
if (f->bd_fn.recon_b_inter(t, bs, b)) return -1;
|
||||
|
||||
const uint8_t *const filter = dav1d_filter_dir[b->filter2d];
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir filter[0], off, mul * filter[0]); \
|
||||
rep_macro(type, t->dir filter[1], off, mul * filter[1]); \
|
||||
rep_macro(type, t->dir intra, off, 0)
|
||||
case_set(bh4, l., 1, by4);
|
||||
case_set(bw4, a->, 0, bx4);
|
||||
BlockContext *edge = t->a;
|
||||
for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
|
||||
#define set_ctx(rep_macro) \
|
||||
rep_macro(edge->filter[0], off, filter[0]); \
|
||||
rep_macro(edge->filter[1], off, filter[1]); \
|
||||
rep_macro(edge->intra, off, 0)
|
||||
case_set(b_dim[2 + i]);
|
||||
#undef set_ctx
|
||||
}
|
||||
|
||||
if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
|
||||
refmvs_block *const r = &t->rt.r[(t->by & 31) + 5 + bh4 - 1][t->bx];
|
||||
@@ -808,11 +797,8 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
}
|
||||
|
||||
if (has_chroma) {
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir uvmode, off, mul * DC_PRED)
|
||||
case_set(cbh4, l., 1, cby4);
|
||||
case_set(cbw4, a->, 0, cbx4);
|
||||
#undef set_ctx
|
||||
dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], DC_PRED);
|
||||
dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], DC_PRED);
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
@@ -972,9 +958,7 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
}
|
||||
|
||||
// delta-q/lf
|
||||
if (!(t->bx & (31 >> !f->seq_hdr->sb128)) &&
|
||||
!(t->by & (31 >> !f->seq_hdr->sb128)))
|
||||
{
|
||||
if (!((t->bx | t->by) & (31 >> !f->seq_hdr->sb128))) {
|
||||
const int prev_qidx = ts->last_qidx;
|
||||
const int have_delta_q = f->frame_hdr->delta.q.present &&
|
||||
(bs != (f->seq_hdr->sb128 ? BS_128x128 : BS_64x64) || !b->skip);
|
||||
@@ -1179,7 +1163,7 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
ts->frame_thread[p].pal_idx += bw4 * bh4 * 8;
|
||||
} else
|
||||
pal_idx = t->scratch.pal_idx_y;
|
||||
read_pal_indices(t, pal_idx, b, 0, w4, h4, bw4, bh4);
|
||||
read_pal_indices(t, pal_idx, b->pal_sz[0], 0, w4, h4, bw4, bh4);
|
||||
if (DEBUG_BLOCK_INFO)
|
||||
printf("Post-y-pal-indices: r=%d\n", ts->msac.rng);
|
||||
}
|
||||
@@ -1193,7 +1177,7 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
ts->frame_thread[p].pal_idx += cbw4 * cbh4 * 8;
|
||||
} else
|
||||
pal_idx = t->scratch.pal_idx_uv;
|
||||
read_pal_indices(t, pal_idx, b, 1, cw4, ch4, cbw4, cbh4);
|
||||
read_pal_indices(t, pal_idx, b->pal_sz[1], 1, cw4, ch4, cbw4, cbh4);
|
||||
if (DEBUG_BLOCK_INFO)
|
||||
printf("Post-uv-pal-indices: r=%d\n", ts->msac.rng);
|
||||
}
|
||||
@@ -1240,39 +1224,39 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
has_chroma ? &t->a->tx_lpf_uv[cbx4] : NULL,
|
||||
has_chroma ? &t->l.tx_lpf_uv[cby4] : NULL);
|
||||
}
|
||||
|
||||
// update contexts
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir tx_intra, off, mul * (((uint8_t *) &t_dim->lw)[diridx])); \
|
||||
rep_macro(type, t->dir tx, off, mul * (((uint8_t *) &t_dim->lw)[diridx])); \
|
||||
rep_macro(type, t->dir mode, off, mul * y_mode_nofilt); \
|
||||
rep_macro(type, t->dir pal_sz, off, mul * b->pal_sz[0]); \
|
||||
rep_macro(type, t->dir seg_pred, off, mul * seg_pred); \
|
||||
rep_macro(type, t->dir skip_mode, off, 0); \
|
||||
rep_macro(type, t->dir intra, off, mul); \
|
||||
rep_macro(type, t->dir skip, off, mul * b->skip); \
|
||||
/* see aomedia bug 2183 for why we use luma coordinates here */ \
|
||||
rep_macro(type, t->pal_sz_uv[diridx], off, mul * (has_chroma ? b->pal_sz[1] : 0)); \
|
||||
if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
|
||||
rep_macro(type, t->dir comp_type, off, mul * COMP_INTER_NONE); \
|
||||
rep_macro(type, t->dir ref[0], off, mul * ((uint8_t) -1)); \
|
||||
rep_macro(type, t->dir ref[1], off, mul * ((uint8_t) -1)); \
|
||||
rep_macro(type, t->dir filter[0], off, mul * DAV1D_N_SWITCHABLE_FILTERS); \
|
||||
rep_macro(type, t->dir filter[1], off, mul * DAV1D_N_SWITCHABLE_FILTERS); \
|
||||
}
|
||||
const enum IntraPredMode y_mode_nofilt =
|
||||
b->y_mode == FILTER_PRED ? DC_PRED : b->y_mode;
|
||||
case_set(bh4, l., 1, by4);
|
||||
case_set(bw4, a->, 0, bx4);
|
||||
BlockContext *edge = t->a;
|
||||
for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
|
||||
int t_lsz = ((uint8_t *) &t_dim->lw)[i]; // lw then lh
|
||||
#define set_ctx(rep_macro) \
|
||||
rep_macro(edge->tx_intra, off, t_lsz); \
|
||||
rep_macro(edge->tx, off, t_lsz); \
|
||||
rep_macro(edge->mode, off, y_mode_nofilt); \
|
||||
rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
|
||||
rep_macro(edge->seg_pred, off, seg_pred); \
|
||||
rep_macro(edge->skip_mode, off, 0); \
|
||||
rep_macro(edge->intra, off, 1); \
|
||||
rep_macro(edge->skip, off, b->skip); \
|
||||
/* see aomedia bug 2183 for why we use luma coordinates here */ \
|
||||
rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
|
||||
if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
|
||||
rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
|
||||
rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
|
||||
rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
|
||||
rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
|
||||
rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
|
||||
}
|
||||
case_set(b_dim[2 + i]);
|
||||
#undef set_ctx
|
||||
}
|
||||
if (b->pal_sz[0])
|
||||
f->bd_fn.copy_pal_block_y(t, bx4, by4, bw4, bh4);
|
||||
if (has_chroma) {
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir uvmode, off, mul * b->uv_mode)
|
||||
case_set(cbh4, l., 1, cby4);
|
||||
case_set(cbw4, a->, 0, cbx4);
|
||||
#undef set_ctx
|
||||
uint8_t uv_mode = b->uv_mode;
|
||||
dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], uv_mode);
|
||||
dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], uv_mode);
|
||||
if (b->pal_sz[1])
|
||||
f->bd_fn.copy_pal_block_uv(t, bx4, by4, bw4, bh4);
|
||||
}
|
||||
@@ -1374,26 +1358,24 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
}
|
||||
|
||||
splat_intrabc_mv(f->c, t, bs, b, bw4, bh4);
|
||||
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir tx_intra, off, mul * b_dim[2 + diridx]); \
|
||||
rep_macro(type, t->dir mode, off, mul * DC_PRED); \
|
||||
rep_macro(type, t->dir pal_sz, off, 0); \
|
||||
/* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
|
||||
rep_macro(type, t->pal_sz_uv[diridx], off, 0); \
|
||||
rep_macro(type, t->dir seg_pred, off, mul * seg_pred); \
|
||||
rep_macro(type, t->dir skip_mode, off, 0); \
|
||||
rep_macro(type, t->dir intra, off, 0); \
|
||||
rep_macro(type, t->dir skip, off, mul * b->skip)
|
||||
case_set(bh4, l., 1, by4);
|
||||
case_set(bw4, a->, 0, bx4);
|
||||
BlockContext *edge = t->a;
|
||||
for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
|
||||
#define set_ctx(rep_macro) \
|
||||
rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
|
||||
rep_macro(edge->mode, off, DC_PRED); \
|
||||
rep_macro(edge->pal_sz, off, 0); \
|
||||
/* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
|
||||
rep_macro(t->pal_sz_uv[i], off, 0); \
|
||||
rep_macro(edge->seg_pred, off, seg_pred); \
|
||||
rep_macro(edge->skip_mode, off, 0); \
|
||||
rep_macro(edge->intra, off, 0); \
|
||||
rep_macro(edge->skip, off, b->skip)
|
||||
case_set(b_dim[2 + i]);
|
||||
#undef set_ctx
|
||||
}
|
||||
if (has_chroma) {
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir uvmode, off, mul * DC_PRED)
|
||||
case_set(cbh4, l., 1, cby4);
|
||||
case_set(cbw4, a->, 0, cbx4);
|
||||
#undef set_ctx
|
||||
dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], DC_PRED);
|
||||
dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], DC_PRED);
|
||||
}
|
||||
} else {
|
||||
// inter-specific mode/mv coding
|
||||
@@ -1837,7 +1819,7 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
b->matrix[2] = t->warpmv.matrix[4];
|
||||
b->matrix[3] = t->warpmv.matrix[5] - 0x10000;
|
||||
} else {
|
||||
b->matrix[0] = SHRT_MIN;
|
||||
b->matrix[0] = INT16_MIN;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1922,32 +1904,29 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
splat_tworef_mv(f->c, t, bs, b, bw4, bh4);
|
||||
else
|
||||
splat_oneref_mv(f->c, t, bs, b, bw4, bh4);
|
||||
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir seg_pred, off, mul * seg_pred); \
|
||||
rep_macro(type, t->dir skip_mode, off, mul * b->skip_mode); \
|
||||
rep_macro(type, t->dir intra, off, 0); \
|
||||
rep_macro(type, t->dir skip, off, mul * b->skip); \
|
||||
rep_macro(type, t->dir pal_sz, off, 0); \
|
||||
/* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
|
||||
rep_macro(type, t->pal_sz_uv[diridx], off, 0); \
|
||||
rep_macro(type, t->dir tx_intra, off, mul * b_dim[2 + diridx]); \
|
||||
rep_macro(type, t->dir comp_type, off, mul * b->comp_type); \
|
||||
rep_macro(type, t->dir filter[0], off, mul * filter[0]); \
|
||||
rep_macro(type, t->dir filter[1], off, mul * filter[1]); \
|
||||
rep_macro(type, t->dir mode, off, mul * b->inter_mode); \
|
||||
rep_macro(type, t->dir ref[0], off, mul * b->ref[0]); \
|
||||
rep_macro(type, t->dir ref[1], off, mul * ((uint8_t) b->ref[1]))
|
||||
case_set(bh4, l., 1, by4);
|
||||
case_set(bw4, a->, 0, bx4);
|
||||
BlockContext *edge = t->a;
|
||||
for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
|
||||
#define set_ctx(rep_macro) \
|
||||
rep_macro(edge->seg_pred, off, seg_pred); \
|
||||
rep_macro(edge->skip_mode, off, b->skip_mode); \
|
||||
rep_macro(edge->intra, off, 0); \
|
||||
rep_macro(edge->skip, off, b->skip); \
|
||||
rep_macro(edge->pal_sz, off, 0); \
|
||||
/* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
|
||||
rep_macro(t->pal_sz_uv[i], off, 0); \
|
||||
rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
|
||||
rep_macro(edge->comp_type, off, b->comp_type); \
|
||||
rep_macro(edge->filter[0], off, filter[0]); \
|
||||
rep_macro(edge->filter[1], off, filter[1]); \
|
||||
rep_macro(edge->mode, off, b->inter_mode); \
|
||||
rep_macro(edge->ref[0], off, b->ref[0]); \
|
||||
rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
|
||||
case_set(b_dim[2 + i]);
|
||||
#undef set_ctx
|
||||
|
||||
}
|
||||
if (has_chroma) {
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->dir uvmode, off, mul * DC_PRED)
|
||||
case_set(cbh4, l., 1, cby4);
|
||||
case_set(cbw4, a->, 0, cbx4);
|
||||
#undef set_ctx
|
||||
dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], DC_PRED);
|
||||
dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], DC_PRED);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1956,12 +1935,12 @@ static int decode_b(Dav1dTaskContext *const t,
|
||||
f->frame_hdr->segmentation.update_map)
|
||||
{
|
||||
uint8_t *seg_ptr = &f->cur_segmap[t->by * f->b4_stride + t->bx];
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
#define set_ctx(rep_macro) \
|
||||
for (int y = 0; y < bh4; y++) { \
|
||||
rep_macro(type, seg_ptr, 0, mul * b->seg_id); \
|
||||
rep_macro(seg_ptr, 0, b->seg_id); \
|
||||
seg_ptr += f->b4_stride; \
|
||||
}
|
||||
case_set(bw4, NULL, 0, 0);
|
||||
case_set(b_dim[2]);
|
||||
#undef set_ctx
|
||||
}
|
||||
if (!b->skip) {
|
||||
@@ -2398,10 +2377,10 @@ static int decode_sb(Dav1dTaskContext *const t, const enum BlockLevel bl,
|
||||
}
|
||||
|
||||
if (t->frame_thread.pass != 2 && (bp != PARTITION_SPLIT || bl == BL_8X8)) {
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, t->a->partition, bx8, mul * dav1d_al_part_ctx[0][bl][bp]); \
|
||||
rep_macro(type, t->l.partition, by8, mul * dav1d_al_part_ctx[1][bl][bp])
|
||||
case_set_upto16(hsz,,,);
|
||||
#define set_ctx(rep_macro) \
|
||||
rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
|
||||
rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
|
||||
case_set_upto16(ulog2(hsz));
|
||||
#undef set_ctx
|
||||
}
|
||||
|
||||
|
||||
@@ -243,15 +243,14 @@ static inline int get_poc_diff(const int order_hint_n_bits,
|
||||
return (diff & (mask - 1)) - (diff & mask);
|
||||
}
|
||||
|
||||
static inline int get_jnt_comp_ctx(const int order_hint_n_bits,
|
||||
const unsigned poc, const unsigned ref0poc,
|
||||
const unsigned ref1poc,
|
||||
static inline int get_jnt_comp_ctx(const int order_hint_n_bits, const int poc,
|
||||
const int ref0poc, const int ref1poc,
|
||||
const BlockContext *const a,
|
||||
const BlockContext *const l,
|
||||
const int yb4, const int xb4)
|
||||
{
|
||||
const unsigned d0 = abs(get_poc_diff(order_hint_n_bits, ref0poc, poc));
|
||||
const unsigned d1 = abs(get_poc_diff(order_hint_n_bits, poc, ref1poc));
|
||||
const int d0 = abs(get_poc_diff(order_hint_n_bits, ref0poc, poc));
|
||||
const int d1 = abs(get_poc_diff(order_hint_n_bits, poc, ref1poc));
|
||||
const int offset = d0 == d1;
|
||||
const int a_ctx = a->comp_type[xb4] >= COMP_INTER_AVG ||
|
||||
a->ref[0][xb4] == 6;
|
||||
|
||||
@@ -43,22 +43,31 @@
|
||||
%endif
|
||||
%endif
|
||||
|
||||
%define WIN32 0
|
||||
%define WIN64 0
|
||||
%define UNIX64 0
|
||||
%if ARCH_X86_64
|
||||
%ifidn __OUTPUT_FORMAT__,win32
|
||||
%define WIN32 1
|
||||
%define WIN64 1
|
||||
%elifidn __OUTPUT_FORMAT__,win64
|
||||
%define WIN32 1
|
||||
%define WIN64 1
|
||||
%elifidn __OUTPUT_FORMAT__,x64
|
||||
%define WIN32 1
|
||||
%define WIN64 1
|
||||
%else
|
||||
%define UNIX64 1
|
||||
%endif
|
||||
%else
|
||||
%ifidn __OUTPUT_FORMAT__,win32
|
||||
%define WIN32 1
|
||||
%endif
|
||||
%endif
|
||||
|
||||
%define FORMAT_ELF 0
|
||||
%define FORMAT_MACHO 0
|
||||
%define FORMAT_OBJ 0
|
||||
%ifidn __OUTPUT_FORMAT__,elf
|
||||
%define FORMAT_ELF 1
|
||||
%elifidn __OUTPUT_FORMAT__,elf32
|
||||
@@ -71,6 +80,10 @@
|
||||
%define FORMAT_MACHO 1
|
||||
%elifidn __OUTPUT_FORMAT__,macho64
|
||||
%define FORMAT_MACHO 1
|
||||
%elifidn __OUTPUT_FORMAT__,obj
|
||||
%define FORMAT_OBJ 1
|
||||
%elifidn __OUTPUT_FORMAT__,obj2
|
||||
%define FORMAT_OBJ 1
|
||||
%endif
|
||||
|
||||
%ifdef PREFIX
|
||||
@@ -89,6 +102,8 @@
|
||||
SECTION .rdata align=%1
|
||||
%elif WIN64
|
||||
SECTION .rdata align=%1
|
||||
%elifidn __OUTPUT_FORMAT__,aout
|
||||
SECTION .text
|
||||
%else
|
||||
SECTION .rodata align=%1
|
||||
%endif
|
||||
@@ -837,6 +852,13 @@ BRANCH_INSTR jz, je, jnz, jne, jl, jle, jnl, jnle, jg, jge, jng, jnge, ja, jae,
|
||||
%else
|
||||
global %2
|
||||
%endif
|
||||
%if WIN32 && !%1
|
||||
%ifdef BUILDING_DLL
|
||||
export %2
|
||||
%endif
|
||||
%elif FORMAT_OBJ && !%1
|
||||
export %2
|
||||
%endif
|
||||
align function_align
|
||||
%2:
|
||||
RESET_MM_PERMUTATION ; needed for x86-64, also makes disassembly somewhat nicer
|
||||
|
||||
@@ -417,6 +417,8 @@ fguv_ss_fn(444, 0, 0);
|
||||
#include "src/arm/filmgrain.h"
|
||||
#elif ARCH_X86
|
||||
#include "src/x86/filmgrain.h"
|
||||
#elif ARCH_RISCV
|
||||
#include "src/riscv/filmgrain.h"
|
||||
#endif
|
||||
#endif
|
||||
|
||||
@@ -436,6 +438,8 @@ COLD void bitfn(dav1d_film_grain_dsp_init)(Dav1dFilmGrainDSPContext *const c) {
|
||||
film_grain_dsp_init_arm(c);
|
||||
#elif ARCH_X86
|
||||
film_grain_dsp_init_x86(c);
|
||||
#elif ARCH_RISCV
|
||||
film_grain_dsp_init_riscv(c);
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
+2
-2
@@ -103,7 +103,7 @@ unsigned dav1d_get_uleb128(GetBits *const c) {
|
||||
i += 7;
|
||||
} while (more && i < 56);
|
||||
|
||||
if (val > UINT_MAX || more) {
|
||||
if (val > UINT32_MAX || more) {
|
||||
c->error = 1;
|
||||
return 0;
|
||||
}
|
||||
@@ -129,7 +129,7 @@ unsigned dav1d_get_vlc(GetBits *const c) {
|
||||
int n_bits = 0;
|
||||
do {
|
||||
if (++n_bits == 32)
|
||||
return 0xFFFFFFFFU;
|
||||
return UINT32_MAX;
|
||||
} while (!dav1d_get_bit(c));
|
||||
|
||||
return ((1U << n_bits) - 1) + dav1d_get_bits(c, n_bits);
|
||||
|
||||
+3
-3
@@ -169,7 +169,7 @@ struct Dav1dContext {
|
||||
Dav1dThreadPicture p;
|
||||
Dav1dRef *segmap;
|
||||
Dav1dRef *refmvs;
|
||||
unsigned refpoc[7];
|
||||
uint8_t refpoc[7];
|
||||
} refs[8];
|
||||
Dav1dMemPool *cdf_pool;
|
||||
CdfThreadContext cdf[8];
|
||||
@@ -226,7 +226,7 @@ struct Dav1dFrameContext {
|
||||
Dav1dRef *cur_segmap_ref, *prev_segmap_ref;
|
||||
uint8_t *cur_segmap;
|
||||
const uint8_t *prev_segmap;
|
||||
unsigned refpoc[7], refrefpoc[7][7];
|
||||
uint8_t refpoc[7], refrefpoc[7][7];
|
||||
uint8_t gmv_warp_allowed[7];
|
||||
CdfThreadContext in_cdf, out_cdf;
|
||||
struct Dav1dTileGroup *tile;
|
||||
@@ -302,7 +302,7 @@ struct Dav1dFrameContext {
|
||||
int cdef_buf_sbh;
|
||||
int lr_buf_plane_sz[2]; /* (stride*sbh*4) << sb128 if n_tc > 1, else stride*4 */
|
||||
int re_sz /* h */;
|
||||
ALIGN(Av1FilterLUT lim_lut, 16);
|
||||
Av1FilterLUT lim_lut;
|
||||
ALIGN(uint8_t lvl[8 /* seg_id */][4 /* dir */][8 /* ref */][2 /* is_gmv */], 16);
|
||||
int last_sharpness;
|
||||
uint8_t *tx_lpf_right_edge[2];
|
||||
|
||||
@@ -732,8 +732,12 @@ static void pal_pred_c(pixel *dst, const ptrdiff_t stride,
|
||||
#if HAVE_ASM
|
||||
#if ARCH_AARCH64 || ARCH_ARM
|
||||
#include "src/arm/ipred.h"
|
||||
#elif ARCH_RISCV
|
||||
#include "src/riscv/ipred.h"
|
||||
#elif ARCH_X86
|
||||
#include "src/x86/ipred.h"
|
||||
#elif ARCH_LOONGARCH64
|
||||
#include "src/loongarch/ipred.h"
|
||||
#endif
|
||||
#endif
|
||||
|
||||
@@ -767,8 +771,12 @@ COLD void bitfn(dav1d_intra_pred_dsp_init)(Dav1dIntraPredDSPContext *const c) {
|
||||
#if HAVE_ASM
|
||||
#if ARCH_AARCH64 || ARCH_ARM
|
||||
intra_pred_dsp_init_arm(c);
|
||||
#elif ARCH_RISCV
|
||||
intra_pred_dsp_init_riscv(c);
|
||||
#elif ARCH_X86
|
||||
intra_pred_dsp_init_x86(c);
|
||||
#elif ARCH_LOONGARCH64
|
||||
intra_pred_dsp_init_loongarch(c);
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
+65
-22
@@ -89,8 +89,8 @@ inv_dct4_1d_internal_c(int32_t *const c, const ptrdiff_t stride,
|
||||
c[3 * stride] = CLIP(t0 - t3);
|
||||
}
|
||||
|
||||
void dav1d_inv_dct4_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
static void inv_dct4_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
{
|
||||
inv_dct4_1d_internal_c(c, stride, min, max, 0);
|
||||
}
|
||||
@@ -142,8 +142,8 @@ inv_dct8_1d_internal_c(int32_t *const c, const ptrdiff_t stride,
|
||||
c[7 * stride] = CLIP(t0 - t7);
|
||||
}
|
||||
|
||||
void dav1d_inv_dct8_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
static void inv_dct8_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
{
|
||||
inv_dct8_1d_internal_c(c, stride, min, max, 0);
|
||||
}
|
||||
@@ -237,8 +237,8 @@ inv_dct16_1d_internal_c(int32_t *const c, const ptrdiff_t stride,
|
||||
c[15 * stride] = CLIP(t0 - t15a);
|
||||
}
|
||||
|
||||
void dav1d_inv_dct16_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
static void inv_dct16_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
{
|
||||
inv_dct16_1d_internal_c(c, stride, min, max, 0);
|
||||
}
|
||||
@@ -427,14 +427,14 @@ inv_dct32_1d_internal_c(int32_t *const c, const ptrdiff_t stride,
|
||||
c[31 * stride] = CLIP(t0 - t31);
|
||||
}
|
||||
|
||||
void dav1d_inv_dct32_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
static void inv_dct32_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
{
|
||||
inv_dct32_1d_internal_c(c, stride, min, max, 0);
|
||||
}
|
||||
|
||||
void dav1d_inv_dct64_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
static void inv_dct64_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
{
|
||||
assert(stride > 0);
|
||||
inv_dct32_1d_internal_c(c, stride << 1, min, max, 1);
|
||||
@@ -962,13 +962,13 @@ inv_adst16_1d_internal_c(const int32_t *const in, const ptrdiff_t in_s,
|
||||
}
|
||||
|
||||
#define inv_adst_1d(sz) \
|
||||
void dav1d_inv_adst##sz##_1d_c(int32_t *const c, const ptrdiff_t stride, \
|
||||
const int min, const int max) \
|
||||
static void inv_adst##sz##_1d_c(int32_t *const c, const ptrdiff_t stride, \
|
||||
const int min, const int max) \
|
||||
{ \
|
||||
inv_adst##sz##_1d_internal_c(c, stride, min, max, c, stride); \
|
||||
} \
|
||||
void dav1d_inv_flipadst##sz##_1d_c(int32_t *const c, const ptrdiff_t stride, \
|
||||
const int min, const int max) \
|
||||
static void inv_flipadst##sz##_1d_c(int32_t *const c, const ptrdiff_t stride, \
|
||||
const int min, const int max) \
|
||||
{ \
|
||||
inv_adst##sz##_1d_internal_c(c, stride, min, max, \
|
||||
&c[(sz - 1) * stride], -stride); \
|
||||
@@ -980,8 +980,8 @@ inv_adst_1d(16)
|
||||
|
||||
#undef inv_adst_1d
|
||||
|
||||
void dav1d_inv_identity4_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
static void inv_identity4_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
{
|
||||
assert(stride > 0);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
@@ -990,16 +990,16 @@ void dav1d_inv_identity4_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
}
|
||||
}
|
||||
|
||||
void dav1d_inv_identity8_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
static void inv_identity8_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
{
|
||||
assert(stride > 0);
|
||||
for (int i = 0; i < 8; i++)
|
||||
c[stride * i] *= 2;
|
||||
}
|
||||
|
||||
void dav1d_inv_identity16_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
static void inv_identity16_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
{
|
||||
assert(stride > 0);
|
||||
for (int i = 0; i < 16; i++) {
|
||||
@@ -1008,14 +1008,57 @@ void dav1d_inv_identity16_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
}
|
||||
}
|
||||
|
||||
void dav1d_inv_identity32_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
static void inv_identity32_1d_c(int32_t *const c, const ptrdiff_t stride,
|
||||
const int min, const int max)
|
||||
{
|
||||
assert(stride > 0);
|
||||
for (int i = 0; i < 32; i++)
|
||||
c[stride * i] *= 4;
|
||||
}
|
||||
|
||||
const itx_1d_fn dav1d_tx1d_fns[N_TX_SIZES][N_TX_1D_TYPES] = {
|
||||
[TX_4X4] = {
|
||||
[DCT] = inv_dct4_1d_c,
|
||||
[ADST] = inv_adst4_1d_c,
|
||||
[FLIPADST] = inv_flipadst4_1d_c,
|
||||
[IDENTITY] = inv_identity4_1d_c,
|
||||
}, [TX_8X8] = {
|
||||
[DCT] = inv_dct8_1d_c,
|
||||
[ADST] = inv_adst8_1d_c,
|
||||
[FLIPADST] = inv_flipadst8_1d_c,
|
||||
[IDENTITY] = inv_identity8_1d_c,
|
||||
}, [TX_16X16] = {
|
||||
[DCT] = inv_dct16_1d_c,
|
||||
[ADST] = inv_adst16_1d_c,
|
||||
[FLIPADST] = inv_flipadst16_1d_c,
|
||||
[IDENTITY] = inv_identity16_1d_c,
|
||||
}, [TX_32X32] = {
|
||||
[DCT] = inv_dct32_1d_c,
|
||||
[IDENTITY] = inv_identity32_1d_c,
|
||||
}, [TX_64X64] = {
|
||||
[DCT] = inv_dct64_1d_c,
|
||||
},
|
||||
};
|
||||
|
||||
const uint8_t /* enum Tx1dType */ dav1d_tx1d_types[N_TX_TYPES][2] = {
|
||||
[DCT_DCT] = { DCT, DCT },
|
||||
[ADST_DCT] = { ADST, DCT },
|
||||
[DCT_ADST] = { DCT, ADST },
|
||||
[ADST_ADST] = { ADST, ADST },
|
||||
[FLIPADST_DCT] = { FLIPADST, DCT },
|
||||
[DCT_FLIPADST] = { DCT, FLIPADST },
|
||||
[FLIPADST_FLIPADST] = { FLIPADST, FLIPADST },
|
||||
[ADST_FLIPADST] = { ADST, FLIPADST },
|
||||
[FLIPADST_ADST] = { FLIPADST, ADST },
|
||||
[IDTX] = { IDENTITY, IDENTITY },
|
||||
[V_DCT] = { DCT, IDENTITY },
|
||||
[H_DCT] = { IDENTITY, DCT },
|
||||
[V_ADST] = { ADST, IDENTITY },
|
||||
[H_ADST] = { IDENTITY, ADST },
|
||||
[V_FLIPADST] = { FLIPADST, IDENTITY },
|
||||
[H_FLIPADST] = { IDENTITY, FLIPADST },
|
||||
};
|
||||
|
||||
#if !(HAVE_ASM && TRIM_DSP_FUNCTIONS && ( \
|
||||
ARCH_AARCH64 || \
|
||||
(ARCH_ARM && (defined(__ARM_NEON) || defined(__APPLE__) || defined(_WIN32))) \
|
||||
|
||||
+12
-18
@@ -28,31 +28,25 @@
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#include "src/levels.h"
|
||||
|
||||
#ifndef DAV1D_SRC_ITX_1D_H
|
||||
#define DAV1D_SRC_ITX_1D_H
|
||||
|
||||
enum Tx1dType {
|
||||
DCT,
|
||||
ADST,
|
||||
IDENTITY,
|
||||
FLIPADST,
|
||||
N_TX_1D_TYPES,
|
||||
};
|
||||
|
||||
#define decl_itx_1d_fn(name) \
|
||||
void (name)(int32_t *c, ptrdiff_t stride, int min, int max)
|
||||
typedef decl_itx_1d_fn(*itx_1d_fn);
|
||||
|
||||
decl_itx_1d_fn(dav1d_inv_dct4_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_dct8_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_dct16_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_dct32_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_dct64_1d_c);
|
||||
|
||||
decl_itx_1d_fn(dav1d_inv_adst4_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_adst8_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_adst16_1d_c);
|
||||
|
||||
decl_itx_1d_fn(dav1d_inv_flipadst4_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_flipadst8_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_flipadst16_1d_c);
|
||||
|
||||
decl_itx_1d_fn(dav1d_inv_identity4_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_identity8_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_identity16_1d_c);
|
||||
decl_itx_1d_fn(dav1d_inv_identity32_1d_c);
|
||||
EXTERN const itx_1d_fn dav1d_tx1d_fns[N_TX_SIZES][N_TX_1D_TYPES];
|
||||
EXTERN const uint8_t /* enum Tx1dType */ dav1d_tx1d_types[N_TX_TYPES][2];
|
||||
|
||||
void dav1d_inv_wht4_1d_c(int32_t *c, ptrdiff_t stride);
|
||||
|
||||
|
||||
+79
-52
@@ -29,6 +29,7 @@
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#include "common/attributes.h"
|
||||
@@ -36,13 +37,17 @@
|
||||
|
||||
#include "src/itx.h"
|
||||
#include "src/itx_1d.h"
|
||||
#include "src/scan.h"
|
||||
#include "src/tables.h"
|
||||
|
||||
static NOINLINE void
|
||||
inv_txfm_add_c(pixel *dst, const ptrdiff_t stride, coef *const coeff,
|
||||
const int eob, const int w, const int h, const int shift,
|
||||
const itx_1d_fn first_1d_fn, const itx_1d_fn second_1d_fn,
|
||||
const int has_dconly HIGHBD_DECL_SUFFIX)
|
||||
const int eob, const /*enum RectTxfmSize*/ int tx, const int shift,
|
||||
const enum TxfmType txtp HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[tx];
|
||||
const int w = 4 * t_dim->w, h = 4 * t_dim->h;
|
||||
const int has_dconly = txtp == DCT_DCT;
|
||||
assert(w >= 4 && w <= 64);
|
||||
assert(h >= 4 && h <= 64);
|
||||
assert(eob >= 0);
|
||||
@@ -64,6 +69,9 @@ inv_txfm_add_c(pixel *dst, const ptrdiff_t stride, coef *const coeff,
|
||||
return;
|
||||
}
|
||||
|
||||
const uint8_t *const txtps = dav1d_tx1d_types[txtp];
|
||||
const itx_1d_fn first_1d_fn = dav1d_tx1d_fns[t_dim->lw][txtps[0]];
|
||||
const itx_1d_fn second_1d_fn = dav1d_tx1d_fns[t_dim->lh][txtps[1]];
|
||||
const int sh = imin(h, 32), sw = imin(w, 32);
|
||||
#if BITDEPTH == 8
|
||||
const int row_clip_min = INT16_MIN;
|
||||
@@ -76,7 +84,16 @@ inv_txfm_add_c(pixel *dst, const ptrdiff_t stride, coef *const coeff,
|
||||
const int col_clip_max = ~col_clip_min;
|
||||
|
||||
int32_t tmp[64 * 64], *c = tmp;
|
||||
for (int y = 0; y < sh; y++, c += w) {
|
||||
int last_nonzero_col; // in first 1d itx
|
||||
if (txtps[1] == IDENTITY && txtps[0] != IDENTITY) {
|
||||
last_nonzero_col = imin(sh - 1, eob);
|
||||
} else if (txtps[0] == IDENTITY && txtps[1] != IDENTITY) {
|
||||
last_nonzero_col = eob >> (t_dim->lw + 2);
|
||||
} else {
|
||||
last_nonzero_col = dav1d_last_nonzero_col_from_eob[tx][eob];
|
||||
}
|
||||
assert(last_nonzero_col < sh);
|
||||
for (int y = 0; y <= last_nonzero_col; y++, c += w) {
|
||||
if (is_rect2)
|
||||
for (int x = 0; x < sw; x++)
|
||||
c[x] = (coeff[y + x * sh] * 181 + 128) >> 8;
|
||||
@@ -85,6 +102,8 @@ inv_txfm_add_c(pixel *dst, const ptrdiff_t stride, coef *const coeff,
|
||||
c[x] = coeff[y + x * sh];
|
||||
first_1d_fn(c, 1, row_clip_min, row_clip_max);
|
||||
}
|
||||
if (last_nonzero_col + 1 < sh)
|
||||
memset(c, 0, sizeof(*c) * (sh - last_nonzero_col - 1) * w);
|
||||
|
||||
memset(coeff, 0, sizeof(*coeff) * sw * sh);
|
||||
for (int i = 0; i < w * sh; i++)
|
||||
@@ -99,7 +118,7 @@ inv_txfm_add_c(pixel *dst, const ptrdiff_t stride, coef *const coeff,
|
||||
dst[x] = iclip_pixel(dst[x] + ((*c++ + 8) >> 4));
|
||||
}
|
||||
|
||||
#define inv_txfm_fn(type1, type2, w, h, shift, has_dconly) \
|
||||
#define inv_txfm_fn(type1, type2, type, pfx, w, h, shift) \
|
||||
static void \
|
||||
inv_txfm_add_##type1##_##type2##_##w##x##h##_c(pixel *dst, \
|
||||
const ptrdiff_t stride, \
|
||||
@@ -107,57 +126,56 @@ inv_txfm_add_##type1##_##type2##_##w##x##h##_c(pixel *dst, \
|
||||
const int eob \
|
||||
HIGHBD_DECL_SUFFIX) \
|
||||
{ \
|
||||
inv_txfm_add_c(dst, stride, coeff, eob, w, h, shift, \
|
||||
dav1d_inv_##type1##w##_1d_c, dav1d_inv_##type2##h##_1d_c, \
|
||||
has_dconly HIGHBD_TAIL_SUFFIX); \
|
||||
inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
|
||||
HIGHBD_TAIL_SUFFIX); \
|
||||
}
|
||||
|
||||
#define inv_txfm_fn64(w, h, shift) \
|
||||
inv_txfm_fn(dct, dct, w, h, shift, 1)
|
||||
#define inv_txfm_fn64(pfx, w, h, shift) \
|
||||
inv_txfm_fn(dct, dct, DCT_DCT, pfx, w, h, shift)
|
||||
|
||||
#define inv_txfm_fn32(w, h, shift) \
|
||||
inv_txfm_fn64(w, h, shift) \
|
||||
inv_txfm_fn(identity, identity, w, h, shift, 0)
|
||||
#define inv_txfm_fn32(pfx, w, h, shift) \
|
||||
inv_txfm_fn64(pfx, w, h, shift) \
|
||||
inv_txfm_fn(identity, identity, IDTX, pfx, w, h, shift)
|
||||
|
||||
#define inv_txfm_fn16(w, h, shift) \
|
||||
inv_txfm_fn32(w, h, shift) \
|
||||
inv_txfm_fn(adst, dct, w, h, shift, 0) \
|
||||
inv_txfm_fn(dct, adst, w, h, shift, 0) \
|
||||
inv_txfm_fn(adst, adst, w, h, shift, 0) \
|
||||
inv_txfm_fn(dct, flipadst, w, h, shift, 0) \
|
||||
inv_txfm_fn(flipadst, dct, w, h, shift, 0) \
|
||||
inv_txfm_fn(adst, flipadst, w, h, shift, 0) \
|
||||
inv_txfm_fn(flipadst, adst, w, h, shift, 0) \
|
||||
inv_txfm_fn(flipadst, flipadst, w, h, shift, 0) \
|
||||
inv_txfm_fn(identity, dct, w, h, shift, 0) \
|
||||
inv_txfm_fn(dct, identity, w, h, shift, 0) \
|
||||
#define inv_txfm_fn16(pfx, w, h, shift) \
|
||||
inv_txfm_fn32(pfx, w, h, shift) \
|
||||
inv_txfm_fn(adst, dct, ADST_DCT, pfx, w, h, shift) \
|
||||
inv_txfm_fn(dct, adst, DCT_ADST, pfx, w, h, shift) \
|
||||
inv_txfm_fn(adst, adst, ADST_ADST, pfx, w, h, shift) \
|
||||
inv_txfm_fn(dct, flipadst, DCT_FLIPADST, pfx, w, h, shift) \
|
||||
inv_txfm_fn(flipadst, dct, FLIPADST_DCT, pfx, w, h, shift) \
|
||||
inv_txfm_fn(adst, flipadst, ADST_FLIPADST, pfx, w, h, shift) \
|
||||
inv_txfm_fn(flipadst, adst, FLIPADST_ADST, pfx, w, h, shift) \
|
||||
inv_txfm_fn(flipadst, flipadst, FLIPADST_FLIPADST, pfx, w, h, shift) \
|
||||
inv_txfm_fn(identity, dct, H_DCT, pfx, w, h, shift) \
|
||||
inv_txfm_fn(dct, identity, V_DCT, pfx, w, h, shift) \
|
||||
|
||||
#define inv_txfm_fn84(w, h, shift) \
|
||||
inv_txfm_fn16(w, h, shift) \
|
||||
inv_txfm_fn(identity, flipadst, w, h, shift, 0) \
|
||||
inv_txfm_fn(flipadst, identity, w, h, shift, 0) \
|
||||
inv_txfm_fn(identity, adst, w, h, shift, 0) \
|
||||
inv_txfm_fn(adst, identity, w, h, shift, 0) \
|
||||
#define inv_txfm_fn84(pfx, w, h, shift) \
|
||||
inv_txfm_fn16(pfx, w, h, shift) \
|
||||
inv_txfm_fn(identity, flipadst, H_FLIPADST, pfx, w, h, shift) \
|
||||
inv_txfm_fn(flipadst, identity, V_FLIPADST, pfx, w, h, shift) \
|
||||
inv_txfm_fn(identity, adst, H_ADST, pfx, w, h, shift) \
|
||||
inv_txfm_fn(adst, identity, V_ADST, pfx, w, h, shift) \
|
||||
|
||||
inv_txfm_fn84( 4, 4, 0)
|
||||
inv_txfm_fn84( 4, 8, 0)
|
||||
inv_txfm_fn84( 4, 16, 1)
|
||||
inv_txfm_fn84( 8, 4, 0)
|
||||
inv_txfm_fn84( 8, 8, 1)
|
||||
inv_txfm_fn84( 8, 16, 1)
|
||||
inv_txfm_fn32( 8, 32, 2)
|
||||
inv_txfm_fn84(16, 4, 1)
|
||||
inv_txfm_fn84(16, 8, 1)
|
||||
inv_txfm_fn16(16, 16, 2)
|
||||
inv_txfm_fn32(16, 32, 1)
|
||||
inv_txfm_fn64(16, 64, 2)
|
||||
inv_txfm_fn32(32, 8, 2)
|
||||
inv_txfm_fn32(32, 16, 1)
|
||||
inv_txfm_fn32(32, 32, 2)
|
||||
inv_txfm_fn64(32, 64, 1)
|
||||
inv_txfm_fn64(64, 16, 2)
|
||||
inv_txfm_fn64(64, 32, 1)
|
||||
inv_txfm_fn64(64, 64, 2)
|
||||
inv_txfm_fn84( , 4, 4, 0)
|
||||
inv_txfm_fn84(R, 4, 8, 0)
|
||||
inv_txfm_fn84(R, 4, 16, 1)
|
||||
inv_txfm_fn84(R, 8, 4, 0)
|
||||
inv_txfm_fn84( , 8, 8, 1)
|
||||
inv_txfm_fn84(R, 8, 16, 1)
|
||||
inv_txfm_fn32(R, 8, 32, 2)
|
||||
inv_txfm_fn84(R, 16, 4, 1)
|
||||
inv_txfm_fn84(R, 16, 8, 1)
|
||||
inv_txfm_fn16( , 16, 16, 2)
|
||||
inv_txfm_fn32(R, 16, 32, 1)
|
||||
inv_txfm_fn64(R, 16, 64, 2)
|
||||
inv_txfm_fn32(R, 32, 8, 2)
|
||||
inv_txfm_fn32(R, 32, 16, 1)
|
||||
inv_txfm_fn32( , 32, 32, 2)
|
||||
inv_txfm_fn64(R, 32, 64, 1)
|
||||
inv_txfm_fn64(R, 64, 16, 2)
|
||||
inv_txfm_fn64(R, 64, 32, 1)
|
||||
inv_txfm_fn64( , 64, 64, 2)
|
||||
|
||||
#if !(HAVE_ASM && TRIM_DSP_FUNCTIONS && ( \
|
||||
ARCH_AARCH64 || \
|
||||
@@ -190,6 +208,8 @@ static void inv_txfm_add_wht_wht_4x4_c(pixel *dst, const ptrdiff_t stride,
|
||||
#include "src/arm/itx.h"
|
||||
#elif ARCH_LOONGARCH64
|
||||
#include "src/loongarch/itx.h"
|
||||
#elif ARCH_PPC64LE
|
||||
#include "src/ppc/itx.h"
|
||||
#elif ARCH_RISCV
|
||||
#include "src/riscv/itx.h"
|
||||
#elif ARCH_X86
|
||||
@@ -267,18 +287,25 @@ COLD void bitfn(dav1d_itx_dsp_init)(Dav1dInvTxfmDSPContext *const c, int bpc) {
|
||||
assign_itx_all_fn64(64, 32, R);
|
||||
assign_itx_all_fn64(64, 64, );
|
||||
|
||||
int all_simd = 0;
|
||||
#if HAVE_ASM
|
||||
#if ARCH_AARCH64 || ARCH_ARM
|
||||
itx_dsp_init_arm(c, bpc);
|
||||
itx_dsp_init_arm(c, bpc, &all_simd);
|
||||
#endif
|
||||
#if ARCH_LOONGARCH64
|
||||
itx_dsp_init_loongarch(c, bpc);
|
||||
#endif
|
||||
#if ARCH_PPC64LE
|
||||
itx_dsp_init_ppc(c, bpc);
|
||||
#endif
|
||||
#if ARCH_RISCV
|
||||
itx_dsp_init_riscv(c, bpc);
|
||||
#endif
|
||||
#if ARCH_X86
|
||||
itx_dsp_init_x86(c, bpc);
|
||||
itx_dsp_init_x86(c, bpc, &all_simd);
|
||||
#endif
|
||||
#endif
|
||||
|
||||
if (!all_simd)
|
||||
dav1d_init_last_nonzero_col_from_eob_tables();
|
||||
}
|
||||
|
||||
+9
-36
@@ -64,18 +64,15 @@ static void decomp_tx(uint8_t (*const txa)[2 /* txsz, step */][32 /* y */][32 /*
|
||||
} else {
|
||||
const int lw = imin(2, t_dim->lw), lh = imin(2, t_dim->lh);
|
||||
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
#define set_ctx(rep_macro) \
|
||||
for (int y = 0; y < t_dim->h; y++) { \
|
||||
rep_macro(type, txa[0][0][y], off, mul * lw); \
|
||||
rep_macro(type, txa[1][0][y], off, mul * lh); \
|
||||
rep_macro(txa[0][0][y], 0, lw); \
|
||||
rep_macro(txa[1][0][y], 0, lh); \
|
||||
txa[0][1][y][0] = t_dim->w; \
|
||||
}
|
||||
case_set_upto16(t_dim->w,,, 0);
|
||||
#undef set_ctx
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, txa[1][1][0], off, mul * t_dim->h)
|
||||
case_set_upto16(t_dim->w,,, 0);
|
||||
case_set_upto16(t_dim->lw);
|
||||
#undef set_ctx
|
||||
dav1d_memset_pow2[t_dim->lw](txa[1][1][0], t_dim->h);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -196,20 +193,8 @@ static inline void mask_edges_intra(uint16_t (*const masks)[32][3][2],
|
||||
if (inner2) masks[1][by4 + y][thl4c][1] |= inner2;
|
||||
}
|
||||
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, a, off, mul * thl4c)
|
||||
#define default_memset(dir, diridx, off, var) \
|
||||
memset(a, thl4c, var)
|
||||
case_set_upto32_with_default(w4,,, 0);
|
||||
#undef default_memset
|
||||
#undef set_ctx
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, l, off, mul * twl4c)
|
||||
#define default_memset(dir, diridx, off, var) \
|
||||
memset(l, twl4c, var)
|
||||
case_set_upto32_with_default(h4,,, 0);
|
||||
#undef default_memset
|
||||
#undef set_ctx
|
||||
dav1d_memset_likely_pow2(a, thl4c, w4);
|
||||
dav1d_memset_likely_pow2(l, twl4c, h4);
|
||||
}
|
||||
|
||||
static void mask_edges_chroma(uint16_t (*const masks)[32][2][2],
|
||||
@@ -267,20 +252,8 @@ static void mask_edges_chroma(uint16_t (*const masks)[32][2][2],
|
||||
}
|
||||
}
|
||||
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, a, off, mul * thl4c)
|
||||
#define default_memset(dir, diridx, off, var) \
|
||||
memset(a, thl4c, var)
|
||||
case_set_upto32_with_default(cw4,,, 0);
|
||||
#undef default_memset
|
||||
#undef set_ctx
|
||||
#define set_ctx(type, dir, diridx, off, mul, rep_macro) \
|
||||
rep_macro(type, l, off, mul * twl4c)
|
||||
#define default_memset(dir, diridx, off, var) \
|
||||
memset(l, twl4c, var)
|
||||
case_set_upto32_with_default(ch4,,, 0);
|
||||
#undef default_memset
|
||||
#undef set_ctx
|
||||
dav1d_memset_likely_pow2(a, thl4c, cw4);
|
||||
dav1d_memset_likely_pow2(l, twl4c, ch4);
|
||||
}
|
||||
|
||||
void dav1d_create_lf_mask_intra(Av1Filter *const lflvl,
|
||||
|
||||
+3
-3
@@ -34,9 +34,9 @@
|
||||
#include "src/levels.h"
|
||||
|
||||
typedef struct Av1FilterLUT {
|
||||
uint8_t e[64];
|
||||
uint8_t i[64];
|
||||
uint64_t sharp[2];
|
||||
ALIGN(uint8_t e[64], 16);
|
||||
ALIGN(uint8_t i[64], 16);
|
||||
ALIGN(uint64_t sharp[2], 16);
|
||||
} Av1FilterLUT;
|
||||
|
||||
typedef struct Av1RestorationUnit {
|
||||
|
||||
@@ -31,7 +31,7 @@
|
||||
#include <errno.h>
|
||||
#include <string.h>
|
||||
|
||||
#if defined(__linux__) && defined(HAVE_DLSYM)
|
||||
#if defined(__linux__) && HAVE_DLSYM
|
||||
#include <dlfcn.h>
|
||||
#endif
|
||||
|
||||
@@ -88,9 +88,9 @@ COLD void dav1d_default_settings(Dav1dSettings *const s) {
|
||||
|
||||
static void close_internal(Dav1dContext **const c_out, int flush);
|
||||
|
||||
#if defined(__linux__) && HAVE_DLSYM && defined(__GLIBC__)
|
||||
NO_SANITIZE("cfi-icall") // CFI is broken with dlsym()
|
||||
static COLD size_t get_stack_size_internal(const pthread_attr_t *const thread_attr) {
|
||||
#if defined(__linux__) && defined(HAVE_DLSYM) && defined(__GLIBC__)
|
||||
/* glibc has an issue where the size of the TLS is subtracted from the stack
|
||||
* size instead of allocated separately. As a result the specified stack
|
||||
* size may be insufficient when used in an application with large amounts
|
||||
@@ -100,9 +100,11 @@ static COLD size_t get_stack_size_internal(const pthread_attr_t *const thread_at
|
||||
dlsym(RTLD_DEFAULT, "__pthread_get_minstack");
|
||||
if (get_minstack)
|
||||
return get_minstack(thread_attr) - PTHREAD_STACK_MIN;
|
||||
#endif
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
#define get_stack_size_internal(attr) (0)
|
||||
#endif
|
||||
|
||||
static COLD void get_num_threads(Dav1dContext *const c, const Dav1dSettings *const s,
|
||||
unsigned *n_tc, unsigned *n_fc)
|
||||
@@ -263,7 +265,6 @@ COLD int dav1d_open(Dav1dContext **const c_out, const Dav1dSettings *const s) {
|
||||
f->c = c;
|
||||
f->task_thread.ttd = &c->task_thread;
|
||||
f->lf.last_sharpness = -1;
|
||||
dav1d_refmvs_init(&f->rf);
|
||||
}
|
||||
|
||||
for (unsigned m = 0; m < c->n_tc; m++) {
|
||||
@@ -556,9 +557,9 @@ void dav1d_flush(Dav1dContext *const c) {
|
||||
if (c->n_fc == 1 && c->n_tc == 1) return;
|
||||
atomic_store(c->flush, 1);
|
||||
|
||||
// stop running tasks in worker threads
|
||||
if (c->n_tc > 1) {
|
||||
pthread_mutex_lock(&c->task_thread.lock);
|
||||
// stop running tasks in worker threads
|
||||
for (unsigned i = 0; i < c->n_tc; i++) {
|
||||
Dav1dTaskContext *const tc = &c->tc[i];
|
||||
while (!tc->task_thread.flushed) {
|
||||
@@ -580,7 +581,6 @@ void dav1d_flush(Dav1dContext *const c) {
|
||||
pthread_mutex_unlock(&c->task_thread.lock);
|
||||
}
|
||||
|
||||
// wait for threads to complete flushing
|
||||
if (c->n_fc > 1) {
|
||||
for (unsigned n = 0, next = c->frame_thread.next; n < c->n_fc; n++, next++) {
|
||||
if (next == c->n_fc) next = 0;
|
||||
@@ -588,6 +588,7 @@ void dav1d_flush(Dav1dContext *const c) {
|
||||
dav1d_decode_frame_exit(f, -1);
|
||||
f->n_tile_data = 0;
|
||||
f->task_thread.retval = 0;
|
||||
f->task_thread.error = 0;
|
||||
Dav1dThreadPicture *out_delayed = &c->frame_thread.out_delayed[next];
|
||||
if (out_delayed->p.frame_hdr) {
|
||||
dav1d_thread_picture_unref(out_delayed);
|
||||
@@ -664,7 +665,7 @@ static COLD void close_internal(Dav1dContext **const c_out, int flush) {
|
||||
dav1d_free(f->lf.lr_mask);
|
||||
dav1d_free(f->lf.tx_lpf_right_edge[0]);
|
||||
dav1d_free(f->lf.start_of_tile_row);
|
||||
dav1d_refmvs_clear(&f->rf);
|
||||
dav1d_free_aligned(f->rf.r);
|
||||
dav1d_free_aligned(f->lf.cdef_line_buf);
|
||||
dav1d_free_aligned(f->lf.lr_line_buf);
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,53 @@
|
||||
/*
|
||||
* Copyright © 2024, VideoLAN and dav1d authors
|
||||
* Copyright © 2024, Loongson Technology Corporation Limited
|
||||
* All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
||||
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
* WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
|
||||
* ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
|
||||
* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
|
||||
* ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
|
||||
* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
#ifndef DAV1D_SRC_LOONGARCH_CDEF_H
|
||||
#define DAV1D_SRC_LOONGARCH_CDEF_H
|
||||
|
||||
#include "config.h"
|
||||
#include "src/cdef.h"
|
||||
#include "src/cpu.h"
|
||||
|
||||
decl_cdef_dir_fn(BF(dav1d_cdef_find_dir, lsx));
|
||||
decl_cdef_fn(BF(dav1d_cdef_filter_block_4x4, lsx));
|
||||
decl_cdef_fn(BF(dav1d_cdef_filter_block_4x8, lsx));
|
||||
decl_cdef_fn(BF(dav1d_cdef_filter_block_8x8, lsx));
|
||||
|
||||
static ALWAYS_INLINE void cdef_dsp_init_loongarch(Dav1dCdefDSPContext *const c) {
|
||||
const unsigned flags = dav1d_get_cpu_flags();
|
||||
|
||||
if (!(flags & DAV1D_LOONGARCH_CPU_FLAG_LSX)) return;
|
||||
|
||||
#if BITDEPTH == 8
|
||||
c->dir = BF(dav1d_cdef_find_dir, lsx);
|
||||
c->fb[0] = BF(dav1d_cdef_filter_block_8x8, lsx);
|
||||
c->fb[1] = BF(dav1d_cdef_filter_block_4x8, lsx);
|
||||
c->fb[2] = BF(dav1d_cdef_filter_block_4x4, lsx);
|
||||
#endif
|
||||
}
|
||||
|
||||
#endif /* DAV1D_SRC_LOONGARCH_CDEF_H */
|
||||
+6
-4
@@ -26,9 +26,11 @@
|
||||
|
||||
#include "config.h"
|
||||
#include "common/attributes.h"
|
||||
|
||||
#include "src/cpu.h"
|
||||
#include "src/loongarch/cpu.h"
|
||||
|
||||
#if defined(HAVE_GETAUXVAL)
|
||||
#if HAVE_GETAUXVAL
|
||||
#include <sys/auxv.h>
|
||||
|
||||
#define LA_HWCAP_LSX ( 1 << 4 )
|
||||
@@ -36,9 +38,9 @@
|
||||
#endif
|
||||
|
||||
COLD unsigned dav1d_get_cpu_flags_loongarch(void) {
|
||||
unsigned flags = 0;
|
||||
#if defined(HAVE_GETAUXVAL)
|
||||
unsigned long hw_cap = getauxval(AT_HWCAP);
|
||||
unsigned flags = dav1d_get_default_cpu_flags();
|
||||
#if HAVE_GETAUXVAL
|
||||
unsigned long hw_cap = dav1d_getauxval(AT_HWCAP);
|
||||
flags |= (hw_cap & LA_HWCAP_LSX) ? DAV1D_LOONGARCH_CPU_FLAG_LSX : 0;
|
||||
flags |= (hw_cap & LA_HWCAP_LASX) ? DAV1D_LOONGARCH_CPU_FLAG_LASX : 0;
|
||||
#endif
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,96 @@
|
||||
/*
|
||||
* Copyright © 2024, VideoLAN and dav1d authors
|
||||
* Copyright © 2024, Loongson Technology Corporation Limited
|
||||
* All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
||||
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
* WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
|
||||
* ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
|
||||
* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
|
||||
* ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
|
||||
* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
#ifndef DAV1D_SRC_LOONGARCH_IPRED_H
|
||||
#define DAV1D_SRC_LOONGARCH_IPRED_H
|
||||
|
||||
#include "config.h"
|
||||
#include "src/ipred.h"
|
||||
#include "src/cpu.h"
|
||||
#include "src/tables.h"
|
||||
|
||||
#define MULTIPLIER_1x2 0x5556
|
||||
#define MULTIPLIER_1x4 0x3334
|
||||
#define BASE_SHIFT 16
|
||||
|
||||
#define init_fn(type0, type1, name, suffix) \
|
||||
c->type0[type1] = BF(dav1d_##name, suffix)
|
||||
|
||||
#define init_angular_ipred_fn(type, name, suffix) \
|
||||
init_fn(intra_pred, type, name, suffix)
|
||||
#define init_cfl_pred_fn(type, name, suffix) \
|
||||
init_fn(cfl_pred, type, name, suffix)
|
||||
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_dc, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_dc_128, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_dc_top, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_dc_left, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_h, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_v, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_paeth, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_smooth, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_smooth_v, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_smooth_h, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_filter, lsx));
|
||||
decl_angular_ipred_fn(BF(dav1d_ipred_z1, lsx));
|
||||
|
||||
decl_cfl_pred_fn(BF(dav1d_ipred_cfl, lsx));
|
||||
decl_cfl_pred_fn(BF(dav1d_ipred_cfl_128, lsx));
|
||||
decl_cfl_pred_fn(BF(dav1d_ipred_cfl_top, lsx));
|
||||
decl_cfl_pred_fn(BF(dav1d_ipred_cfl_left, lsx));
|
||||
|
||||
decl_pal_pred_fn(BF(dav1d_pal_pred, lsx));
|
||||
|
||||
static ALWAYS_INLINE void intra_pred_dsp_init_loongarch(Dav1dIntraPredDSPContext *const c) {
|
||||
const unsigned flags = dav1d_get_cpu_flags();
|
||||
|
||||
if (!(flags & DAV1D_LOONGARCH_CPU_FLAG_LSX)) return;
|
||||
|
||||
#if BITDEPTH == 8
|
||||
init_angular_ipred_fn(DC_PRED, ipred_dc, lsx);
|
||||
init_angular_ipred_fn(DC_128_PRED, ipred_dc_128, lsx);
|
||||
init_angular_ipred_fn(TOP_DC_PRED, ipred_dc_top, lsx);
|
||||
init_angular_ipred_fn(LEFT_DC_PRED, ipred_dc_left, lsx);
|
||||
init_angular_ipred_fn(HOR_PRED, ipred_h, lsx);
|
||||
init_angular_ipred_fn(VERT_PRED, ipred_v, lsx);
|
||||
init_angular_ipred_fn(PAETH_PRED, ipred_paeth, lsx);
|
||||
init_angular_ipred_fn(SMOOTH_PRED, ipred_smooth, lsx);
|
||||
init_angular_ipred_fn(SMOOTH_V_PRED, ipred_smooth_v, lsx);
|
||||
init_angular_ipred_fn(SMOOTH_H_PRED, ipred_smooth_h, lsx);
|
||||
init_angular_ipred_fn(FILTER_PRED, ipred_filter, lsx);
|
||||
init_angular_ipred_fn(Z1_PRED, ipred_z1, lsx);
|
||||
|
||||
init_cfl_pred_fn(DC_PRED, ipred_cfl, lsx);
|
||||
init_cfl_pred_fn(DC_128_PRED, ipred_cfl_128, lsx);
|
||||
init_cfl_pred_fn(TOP_DC_PRED, ipred_cfl_top, lsx);
|
||||
init_cfl_pred_fn(LEFT_DC_PRED, ipred_cfl_left, lsx);
|
||||
|
||||
c->pal_pred = BF(dav1d_pal_pred, lsx);
|
||||
#endif
|
||||
}
|
||||
|
||||
#endif /* DAV1D_SRC_LOONGARCH_IPRED_H */
|
||||
+3062
-6385
File diff suppressed because it is too large
Load Diff
+48
-124
@@ -31,67 +31,18 @@
|
||||
#include "src/cpu.h"
|
||||
#include "src/itx.h"
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_wht_wht_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_identity_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_dct_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_adst_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_adst_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_flipadst_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_adst_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_flipadst_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_dct_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_flipadst_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_identity_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_dct_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_identity_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_flipadst_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_adst_4x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_identity_4x4, lsx));
|
||||
decl_itx17_fns( 4, 4, lsx);
|
||||
decl_itx16_fns( 4, 8, lsx);
|
||||
decl_itx16_fns( 4, 16, lsx);
|
||||
decl_itx16_fns( 8, 4, lsx);
|
||||
decl_itx16_fns( 8, 8, lsx);
|
||||
decl_itx16_fns( 8, 16, lsx);
|
||||
decl_itx2_fns ( 8, 32, lsx);
|
||||
decl_itx16_fns(16, 8, lsx);
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_4x8, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_identity_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_dct_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_adst_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_adst_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_adst_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_flipadst_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_dct_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_flipadst_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_flipadst_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_identity_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_dct_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_identity_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_flipadst_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_identity_8x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_adst_8x4, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_identity_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_dct_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_adst_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_adst_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_adst_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_flipadst_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_dct_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_flipadst_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_adst_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_identity_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_identity_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_dct_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_flipadst_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_identity_8x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_flipadst_8x8, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_8x16, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_identity_8x16, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_dct_8x16, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_adst_8x16, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_16x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_dct_16x8, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_16x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_identity_identity_16x4, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_dct_16x4, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_16x16, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_adst_16x16, lsx));
|
||||
@@ -99,14 +50,23 @@ decl_itx_fn(BF(dav1d_inv_txfm_add_adst_dct_16x16, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_adst_16x16, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_dct_16x16, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_flipadst_16x16, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_flipadst_16x16, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_flipadst_adst_16x16, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_8x32, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_32x32, lsx));
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_16x32, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_32x8, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_32x16, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_32x32, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_64x32, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_dct_dct_64x64, lsx));
|
||||
|
||||
decl_itx_fn(BF(dav1d_inv_txfm_add_adst_adst_16x16, lasx));
|
||||
|
||||
static ALWAYS_INLINE void itx_dsp_init_loongarch(Dav1dInvTxfmDSPContext *const c, int bpc) {
|
||||
#if BITDEPTH == 8
|
||||
const unsigned flags = dav1d_get_cpu_flags();
|
||||
@@ -115,67 +75,20 @@ static ALWAYS_INLINE void itx_dsp_init_loongarch(Dav1dInvTxfmDSPContext *const c
|
||||
|
||||
if (BITDEPTH != 8 ) return;
|
||||
|
||||
c->itxfm_add[TX_4X4][WHT_WHT] = dav1d_inv_txfm_add_wht_wht_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][IDTX] = dav1d_inv_txfm_add_identity_identity_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][DCT_ADST] = dav1d_inv_txfm_add_adst_dct_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][ADST_DCT] = dav1d_inv_txfm_add_dct_adst_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][ADST_ADST] = dav1d_inv_txfm_add_adst_adst_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][FLIPADST_DCT] = dav1d_inv_txfm_add_dct_flipadst_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][ADST_FLIPADST] = dav1d_inv_txfm_add_flipadst_adst_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][FLIPADST_ADST] = dav1d_inv_txfm_add_adst_flipadst_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][DCT_FLIPADST] = dav1d_inv_txfm_add_flipadst_dct_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][FLIPADST_FLIPADST] = dav1d_inv_txfm_add_flipadst_flipadst_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][H_DCT] = dav1d_inv_txfm_add_dct_identity_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][V_DCT] = dav1d_inv_txfm_add_identity_dct_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][H_FLIPADST] = dav1d_inv_txfm_add_flipadst_identity_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][V_FLIPADST] = dav1d_inv_txfm_add_identity_flipadst_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][V_ADST] = dav1d_inv_txfm_add_identity_adst_4x4_8bpc_lsx;
|
||||
c->itxfm_add[TX_4X4][H_ADST] = dav1d_inv_txfm_add_adst_identity_4x4_8bpc_lsx;
|
||||
assign_itx17_fn( , 4, 4, lsx);
|
||||
assign_itx16_fn(R, 4, 8, lsx);
|
||||
assign_itx16_fn(R, 4, 16, lsx);
|
||||
assign_itx16_fn(R, 8, 4, lsx);
|
||||
assign_itx16_fn( , 8, 8, lsx);
|
||||
assign_itx16_fn(R, 8, 16, lsx);
|
||||
assign_itx2_fn (R, 8, 32, lsx);
|
||||
assign_itx16_fn(R, 16, 8, lsx);
|
||||
assign_itx1_fn (R, 64, 32, lsx);
|
||||
assign_itx1_fn ( , 64, 64, lsx);
|
||||
|
||||
c->itxfm_add[RTX_4X8][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_4x8_8bpc_lsx;
|
||||
|
||||
c->itxfm_add[RTX_8X4][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][IDTX] = dav1d_inv_txfm_add_identity_identity_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][DCT_ADST] = dav1d_inv_txfm_add_adst_dct_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][ADST_DCT] = dav1d_inv_txfm_add_dct_adst_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][ADST_ADST] = dav1d_inv_txfm_add_adst_adst_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][ADST_FLIPADST] = dav1d_inv_txfm_add_flipadst_adst_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][FLIPADST_ADST] = dav1d_inv_txfm_add_adst_flipadst_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][DCT_FLIPADST] = dav1d_inv_txfm_add_flipadst_dct_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][FLIPADST_DCT] = dav1d_inv_txfm_add_dct_flipadst_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][FLIPADST_FLIPADST] = dav1d_inv_txfm_add_flipadst_flipadst_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][H_DCT] = dav1d_inv_txfm_add_dct_identity_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][V_DCT] = dav1d_inv_txfm_add_identity_dct_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][H_FLIPADST] = dav1d_inv_txfm_add_flipadst_identity_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][V_FLIPADST] = dav1d_inv_txfm_add_identity_flipadst_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][H_ADST] = dav1d_inv_txfm_add_adst_identity_8x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X4][V_ADST] = dav1d_inv_txfm_add_identity_adst_8x4_8bpc_lsx;
|
||||
|
||||
c->itxfm_add[TX_8X8][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][IDTX] = dav1d_inv_txfm_add_identity_identity_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][DCT_ADST] = dav1d_inv_txfm_add_adst_dct_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][ADST_DCT] = dav1d_inv_txfm_add_dct_adst_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][ADST_ADST] = dav1d_inv_txfm_add_adst_adst_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][ADST_FLIPADST] = dav1d_inv_txfm_add_flipadst_adst_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][FLIPADST_ADST] = dav1d_inv_txfm_add_adst_flipadst_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][DCT_FLIPADST] = dav1d_inv_txfm_add_flipadst_dct_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][FLIPADST_DCT] = dav1d_inv_txfm_add_dct_flipadst_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][FLIPADST_FLIPADST] = dav1d_inv_txfm_add_flipadst_flipadst_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][H_DCT] = dav1d_inv_txfm_add_dct_identity_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][V_DCT] = dav1d_inv_txfm_add_identity_dct_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][H_FLIPADST] = dav1d_inv_txfm_add_flipadst_identity_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][V_FLIPADST] = dav1d_inv_txfm_add_identity_flipadst_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][H_ADST] = dav1d_inv_txfm_add_adst_identity_8x8_8bpc_lsx;
|
||||
c->itxfm_add[TX_8X8][V_ADST] = dav1d_inv_txfm_add_identity_adst_8x8_8bpc_lsx;
|
||||
|
||||
c->itxfm_add[RTX_8X16][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_8x16_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X16][IDTX] = dav1d_inv_txfm_add_identity_identity_8x16_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X16][DCT_ADST] = dav1d_inv_txfm_add_adst_dct_8x16_8bpc_lsx;
|
||||
c->itxfm_add[RTX_8X16][ADST_DCT] = dav1d_inv_txfm_add_dct_adst_8x16_8bpc_lsx;
|
||||
|
||||
c->itxfm_add[RTX_16X8][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_16x8_8bpc_lsx;
|
||||
c->itxfm_add[RTX_16X8][DCT_ADST] = dav1d_inv_txfm_add_adst_dct_16x8_8bpc_lsx;
|
||||
c->itxfm_add[RTX_16X4][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_16x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_16X4][IDTX] = dav1d_inv_txfm_add_identity_identity_16x4_8bpc_lsx;
|
||||
c->itxfm_add[RTX_16X4][DCT_ADST] = dav1d_inv_txfm_add_adst_dct_16x4_8bpc_lsx;
|
||||
|
||||
c->itxfm_add[TX_16X16][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_16x16_8bpc_lsx;
|
||||
c->itxfm_add[TX_16X16][ADST_ADST] = dav1d_inv_txfm_add_adst_adst_16x16_8bpc_lsx;
|
||||
@@ -183,12 +96,23 @@ static ALWAYS_INLINE void itx_dsp_init_loongarch(Dav1dInvTxfmDSPContext *const c
|
||||
c->itxfm_add[TX_16X16][ADST_DCT] = dav1d_inv_txfm_add_dct_adst_16x16_8bpc_lsx;
|
||||
c->itxfm_add[TX_16X16][DCT_FLIPADST] = dav1d_inv_txfm_add_flipadst_dct_16x16_8bpc_lsx;
|
||||
c->itxfm_add[TX_16X16][FLIPADST_DCT] = dav1d_inv_txfm_add_dct_flipadst_16x16_8bpc_lsx;
|
||||
c->itxfm_add[TX_16X16][FLIPADST_ADST] = dav1d_inv_txfm_add_adst_flipadst_16x16_8bpc_lsx;
|
||||
c->itxfm_add[TX_16X16][ADST_FLIPADST] = dav1d_inv_txfm_add_flipadst_adst_16x16_8bpc_lsx;
|
||||
|
||||
c->itxfm_add[RTX_8X32][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_8x32_8bpc_lsx;
|
||||
c->itxfm_add[RTX_16X32][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_16x32_8bpc_lsx;
|
||||
|
||||
c->itxfm_add[RTX_32X8][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_32x8_8bpc_lsx;
|
||||
|
||||
c->itxfm_add[RTX_32X16][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_32x16_8bpc_lsx;
|
||||
|
||||
c->itxfm_add[TX_32X32][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_32x32_8bpc_lsx;
|
||||
|
||||
c->itxfm_add[TX_64X64][DCT_DCT] = dav1d_inv_txfm_add_dct_dct_64x64_8bpc_lsx;
|
||||
if (!(flags & DAV1D_LOONGARCH_CPU_FLAG_LASX)) return;
|
||||
|
||||
if (BITDEPTH != 8 ) return;
|
||||
|
||||
c->itxfm_add[TX_16X16][ADST_ADST] = dav1d_inv_txfm_add_adst_adst_16x16_8bpc_lasx;
|
||||
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -59,6 +59,7 @@
|
||||
.text ;
|
||||
.align \align ;
|
||||
.globl ASM_PREF\name ;
|
||||
.hidden ASM_PREF\name ;
|
||||
.type ASM_PREF\name, @function ;
|
||||
ASM_PREF\name: ;
|
||||
.endm
|
||||
|
||||
@@ -0,0 +1,192 @@
|
||||
/******************************************************************************
|
||||
* Copyright © 2024, VideoLAN and dav1d authors
|
||||
* Copyright © 2024, Loongson Technology Corporation Limited
|
||||
* All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
||||
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
* WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
|
||||
* ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
|
||||
* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
|
||||
* ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
|
||||
* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*****************************************************************************/
|
||||
|
||||
#ifndef DAV1D_SRC_LOONGSON_UTIL_S
|
||||
#define DAV1D_SRC_LOONGSON_UTIL_S
|
||||
|
||||
#ifndef DEFAULT_ALIGN
|
||||
#define DEFAULT_ALIGN 5
|
||||
#endif
|
||||
|
||||
//That l means local defines local functions
|
||||
.macro functionl name, align=DEFAULT_ALIGN
|
||||
.macro endfuncl
|
||||
jirl $r0, $r1, 0x0
|
||||
.size \name, . - \name
|
||||
.purgem endfuncl
|
||||
.endm
|
||||
.text ;
|
||||
.align \align ;
|
||||
.hidden \name ;
|
||||
.type \name, @function ;
|
||||
\name: ;
|
||||
.endm
|
||||
|
||||
.macro TRANSPOSE_4x16B in0, in1 ,in2, in3, in4, in5, in6, in7
|
||||
vpackev.b \in4, \in1, \in0
|
||||
vpackod.b \in5, \in1, \in0
|
||||
vpackev.b \in6, \in3, \in2
|
||||
vpackod.b \in7, \in3, \in2
|
||||
|
||||
vpackev.h \in0, \in6, \in4
|
||||
vpackod.h \in2, \in6, \in4
|
||||
vpackev.h \in1, \in7, \in5
|
||||
vpackod.h \in3, \in7, \in5
|
||||
.endm
|
||||
|
||||
.macro TRANSPOSE_8x16B in0, in1, in2, in3, in4, in5, in6, in7, in8, in9
|
||||
vpackev.b \in8, \in1, \in0
|
||||
vpackod.b \in9, \in1, \in0
|
||||
vpackev.b \in1, \in3, \in2
|
||||
vpackod.b \in3, \in3, \in2
|
||||
vpackev.b \in0, \in5, \in4
|
||||
vpackod.b \in5, \in5, \in4
|
||||
vpackev.b \in2, \in7, \in6
|
||||
vpackod.b \in7, \in7, \in6
|
||||
|
||||
vpackev.h \in4, \in2, \in0
|
||||
vpackod.h \in2, \in2, \in0
|
||||
vpackev.h \in6, \in7, \in5
|
||||
vpackod.h \in7, \in7, \in5
|
||||
vpackev.h \in5, \in3, \in9
|
||||
vpackod.h \in9, \in3, \in9
|
||||
vpackev.h \in3, \in1, \in8
|
||||
vpackod.h \in8, \in1, \in8
|
||||
|
||||
vpackev.w \in0, \in4, \in3
|
||||
vpackod.w \in4, \in4, \in3
|
||||
vpackev.w \in1, \in6, \in5
|
||||
vpackod.w \in5, \in6, \in5
|
||||
vpackod.w \in6, \in2, \in8
|
||||
vpackev.w \in2, \in2, \in8
|
||||
vpackev.w \in3, \in7, \in9
|
||||
vpackod.w \in7, \in7, \in9
|
||||
.endm
|
||||
|
||||
.macro vld_x8 src, start, stride, in0, in1, in2, in3, in4, in5, in6, in7
|
||||
vld \in0, \src, \start
|
||||
vld \in1, \src, \start+(\stride*1)
|
||||
vld \in2, \src, \start+(\stride*2)
|
||||
vld \in3, \src, \start+(\stride*3)
|
||||
vld \in4, \src, \start+(\stride*4)
|
||||
vld \in5, \src, \start+(\stride*5)
|
||||
vld \in6, \src, \start+(\stride*6)
|
||||
vld \in7, \src, \start+(\stride*7)
|
||||
.endm
|
||||
|
||||
.macro vst_x8 src, start, stride, in0, in1, in2, in3, in4, in5, in6, in7
|
||||
vst \in0, \src, \start
|
||||
vst \in1, \src, \start+(\stride*1)
|
||||
vst \in2, \src, \start+(\stride*2)
|
||||
vst \in3, \src, \start+(\stride*3)
|
||||
vst \in4, \src, \start+(\stride*4)
|
||||
vst \in5, \src, \start+(\stride*5)
|
||||
vst \in6, \src, \start+(\stride*6)
|
||||
vst \in7, \src, \start+(\stride*7)
|
||||
.endm
|
||||
|
||||
.macro vld_x16 src, start, stride, in0, in1, in2, in3, in4, in5, in6, in7, \
|
||||
in8, in9, in10, in11, in12, in13, in14, in15
|
||||
|
||||
vld_x8 \src, \start, \stride, \in0, \in1, \in2, \in3, \in4, \in5, \in6, \in7
|
||||
|
||||
vld \in8, \src, \start+(\stride*8)
|
||||
vld \in9, \src, \start+(\stride*9)
|
||||
vld \in10, \src, \start+(\stride*10)
|
||||
vld \in11, \src, \start+(\stride*11)
|
||||
vld \in12, \src, \start+(\stride*12)
|
||||
vld \in13, \src, \start+(\stride*13)
|
||||
vld \in14, \src, \start+(\stride*14)
|
||||
vld \in15, \src, \start+(\stride*15)
|
||||
.endm
|
||||
|
||||
.macro vst_x16 src, start, stride, in0, in1, in2, in3, in4, in5, in6, in7, \
|
||||
in8, in9, in10, in11, in12, in13, in14, in15
|
||||
|
||||
vst_x8 \src, \start, \stride, \in0, \in1, \in2, \in3, \in4, \in5, \in6, \in7
|
||||
|
||||
vst \in8, \src, \start+(\stride*8)
|
||||
vst \in9, \src, \start+(\stride*9)
|
||||
vst \in10, \src, \start+(\stride*10)
|
||||
vst \in11, \src, \start+(\stride*11)
|
||||
vst \in12, \src, \start+(\stride*12)
|
||||
vst \in13, \src, \start+(\stride*13)
|
||||
vst \in14, \src, \start+(\stride*14)
|
||||
vst \in15, \src, \start+(\stride*15)
|
||||
.endm
|
||||
|
||||
.macro xvld_x8 src, start, stride, in0, in1, in2, in3, in4, in5, in6, in7
|
||||
xvld \in0, \src, \start
|
||||
xvld \in1, \src, \start+(\stride)
|
||||
xvld \in2, \src, \start+(\stride<<1)
|
||||
xvld \in3, \src, \start+(\stride<<1)+(\stride)
|
||||
xvld \in4, \src, \start+(\stride<<2)
|
||||
xvld \in5, \src, \start+(\stride<<2)+(\stride)
|
||||
xvld \in6, \src, \start+(\stride*6)
|
||||
xvld \in7, \src, \start+(\stride<<3)-(\stride)
|
||||
.endm
|
||||
|
||||
.macro xvst_x8 src, start, stride, in0, in1, in2, in3, in4, in5, in6, in7
|
||||
xvst \in0, \src, \start
|
||||
xvst \in1, \src, \start+(\stride)
|
||||
xvst \in2, \src, \start+(\stride<<1)
|
||||
xvst \in3, \src, \start+(\stride<<1)+(\stride)
|
||||
xvst \in4, \src, \start+(\stride<<2)
|
||||
xvst \in5, \src, \start+(\stride<<2)+(\stride)
|
||||
xvst \in6, \src, \start+(\stride*6)
|
||||
xvst \in7, \src, \start+(\stride<<3)-(\stride)
|
||||
.endm
|
||||
|
||||
.macro xvld_x16 src, start, stride, in0, in1, in2, in3, in4, in5, in6, in7, \
|
||||
in8, in9, in10, in11, in12, in13, in14, in15
|
||||
xvld_x8 \src, \start, \stride, \in0, \in1, \in2, \in3, \in4, \in5, \in6, \in7
|
||||
|
||||
xvld \in8, \src, \start+(\stride<<3)
|
||||
xvld \in9, \src, \start+(\stride<<3)+(\stride)
|
||||
xvld \in10, \src, \start+(\stride*10)
|
||||
xvld \in11, \src, \start+(\stride*11)
|
||||
xvld \in12, \src, \start+(\stride*12)
|
||||
xvld \in13, \src, \start+(\stride*13)
|
||||
xvld \in14, \src, \start+(\stride*14)
|
||||
xvld \in15, \src, \start+(\stride<<4)-(\stride)
|
||||
.endm
|
||||
|
||||
.macro xvst_x16 src, start, stride, in0, in1, in2, in3, in4, in5, in6, in7, \
|
||||
in8, in9, in10, in11, in12, in13, in14, in15
|
||||
xvst_x8 \src, \start, \stride, \in0, \in1, \in2, \in3, \in4, \in5, \in6, \in7
|
||||
|
||||
xvst \in8, \src, \start+(\stride<<3)
|
||||
xvst \in9, \src, \start+(\stride<<3)+(\stride)
|
||||
xvst \in10, \src, \start+(\stride*10)
|
||||
xvst \in11, \src, \start+(\stride*11)
|
||||
xvst \in12, \src, \start+(\stride*12)
|
||||
xvst \in13, \src, \start+(\stride*13)
|
||||
xvst \in14, \src, \start+(\stride*14)
|
||||
xvst \in15, \src, \start+(\stride<<4)-(\stride)
|
||||
.endm
|
||||
|
||||
#endif /* DAV1D_SRC_LOONGSON_UTIL_S */
|
||||
+1148
-1041
File diff suppressed because it is too large
Load Diff
+595
-46
@@ -92,13 +92,11 @@ function wiener_filter_h_8bpc_lsx
|
||||
|
||||
vsllwil.hu.bu vr15, vr8, 0 // 3 4 5 6 7 8 9 10
|
||||
vexth.hu.bu vr16, vr8 // 11 12 13 14 15 16 17 18
|
||||
vsllwil.wu.hu vr17, vr15, 0 // 3 4 5 6
|
||||
vsllwil.wu.hu vr17, vr15, 7 // 3 4 5 6
|
||||
vexth.wu.hu vr18, vr15 // 7 8 9 10
|
||||
vsllwil.wu.hu vr19, vr16, 0 // 11 12 13 14
|
||||
vsllwil.wu.hu vr19, vr16, 7 // 11 12 13 14
|
||||
vexth.wu.hu vr20, vr16 // 15 16 17 18
|
||||
vslli.w vr17, vr17, 7
|
||||
vslli.w vr18, vr18, 7
|
||||
vslli.w vr19, vr19, 7
|
||||
vslli.w vr20, vr20, 7
|
||||
vxor.v vr15, vr15, vr15
|
||||
vxor.v vr14, vr14, vr14
|
||||
@@ -315,30 +313,24 @@ function boxsum3_h_8bpc_lsx
|
||||
vmulwod.h.bu vr10, vr3, vr3
|
||||
vmulwev.h.bu vr11, vr5, vr5
|
||||
vmulwod.h.bu vr12, vr5, vr5
|
||||
vmul.h vr7, vr7, vr7
|
||||
vmul.h vr8, vr8, vr8
|
||||
vaddwev.w.hu vr13, vr10, vr9
|
||||
vaddwod.w.hu vr14, vr10, vr9
|
||||
vilvl.w vr3, vr14, vr13
|
||||
vilvh.w vr4, vr14, vr13
|
||||
vaddwev.w.hu vr13, vr12, vr11
|
||||
vaddwod.w.hu vr14, vr12, vr11
|
||||
vilvl.w vr15, vr14, vr13
|
||||
vilvh.w vr16, vr14, vr13
|
||||
vsllwil.wu.hu vr9, vr7, 0
|
||||
vexth.wu.hu vr10, vr7
|
||||
vsllwil.wu.hu vr11, vr8, 0
|
||||
vexth.wu.hu vr12, vr8
|
||||
vadd.w vr9, vr9, vr3
|
||||
vadd.w vr10, vr10, vr4
|
||||
vadd.w vr11, vr11, vr15
|
||||
vadd.w vr12, vr12, vr16
|
||||
vaddwev.w.hu vr15, vr12, vr11
|
||||
vaddwod.w.hu vr16, vr12, vr11
|
||||
vmaddwev.w.hu vr13, vr7, vr7
|
||||
vmaddwod.w.hu vr14, vr7, vr7
|
||||
vmaddwev.w.hu vr15, vr8, vr8
|
||||
vmaddwod.w.hu vr16, vr8, vr8
|
||||
vilvl.w vr9, vr14, vr13
|
||||
vilvh.w vr10, vr14, vr13
|
||||
vilvl.w vr11, vr16, vr15
|
||||
vilvh.w vr12, vr16, vr15
|
||||
vst vr9, t2, REST_UNIT_STRIDE<<2
|
||||
vst vr10, t2, (REST_UNIT_STRIDE<<2)+16
|
||||
vst vr11, t2, (REST_UNIT_STRIDE<<2)+32
|
||||
vst vr12, t2, (REST_UNIT_STRIDE<<2)+48
|
||||
addi.d t2, t2, 64
|
||||
|
||||
addi.d t2, t2, 64
|
||||
addi.w t5, t5, -16
|
||||
addi.d t3, t3, 16
|
||||
blt zero, t5, .LBS3_H_W
|
||||
@@ -348,8 +340,6 @@ function boxsum3_h_8bpc_lsx
|
||||
addi.d a2, a2, REST_UNIT_STRIDE
|
||||
addi.d a4, a4, -1
|
||||
blt zero, a4, .LBS3_H_H
|
||||
|
||||
.LBS3_H_END:
|
||||
endfunc
|
||||
|
||||
/*
|
||||
@@ -379,10 +369,10 @@ function boxsum3_v_8bpc_lsx
|
||||
vld vr7, t0, 20 // 4 5 6 7
|
||||
vld vr8, t0, 24 // 5 6 7 8
|
||||
vadd.h vr9, vr0, vr1
|
||||
vadd.h vr9, vr9, vr2
|
||||
vadd.w vr10, vr3, vr4
|
||||
vadd.w vr10, vr10, vr5
|
||||
vadd.w vr11, vr6, vr7
|
||||
vadd.h vr9, vr9, vr2
|
||||
vadd.w vr10, vr10, vr5
|
||||
vadd.w vr11, vr11, vr8
|
||||
vpickve2gr.h t7, vr2, 6
|
||||
vpickve2gr.w t8, vr8, 2
|
||||
@@ -395,7 +385,7 @@ function boxsum3_v_8bpc_lsx
|
||||
addi.d t5, t5, 32
|
||||
addi.d t6, t6, 16
|
||||
addi.d t3, t3, -8
|
||||
ble t3, zero, .LBS3_V_H0
|
||||
bge zero, t3, .LBS3_V_H0
|
||||
|
||||
.LBS3_V_W8:
|
||||
vld vr0, t1, 0 // a 0 1 2 3 4 5 6 7
|
||||
@@ -425,15 +415,13 @@ function boxsum3_v_8bpc_lsx
|
||||
addi.d t0, t0, 32
|
||||
addi.d t5, t5, 32
|
||||
addi.d t6, t6, 16
|
||||
blt zero, t3, .LBS3_V_W8
|
||||
blt zero, t3, .LBS3_V_W8
|
||||
|
||||
.LBS3_V_H0:
|
||||
addi.d a1, a1, REST_UNIT_STRIDE<<1
|
||||
addi.d a0, a0, REST_UNIT_STRIDE<<2
|
||||
addi.w a3, a3, -1
|
||||
bnez a3, .LBS3_V_H
|
||||
|
||||
.LBS3_V_END:
|
||||
endfunc
|
||||
|
||||
/*
|
||||
@@ -559,14 +547,14 @@ function boxsum3_sgf_v_8bpc_lsx
|
||||
.LBS3SGF_V_W:
|
||||
vld vr0, t0, 0 // P[i - REST_UNIT_STRIDE]
|
||||
vld vr1, t0, 16
|
||||
vld vr2, t1, -4 // P[i-1]
|
||||
vld vr3, t1, 12
|
||||
vld vr2, t1, -4 // P[i-1] -1 0 1 2
|
||||
vld vr3, t1, 12 // 3 4 5 6
|
||||
vld vr4, t2, 0 // P[i + REST_UNIT_STRIDE]
|
||||
vld vr5, t2, 16
|
||||
vld vr6, t1, 0 // p[i]
|
||||
vld vr7, t1, 16
|
||||
vld vr8, t1, 4 // p[i+1]
|
||||
vld vr9, t1, 20
|
||||
vld vr6, t1, 0 // p[i] 0 1 2 3
|
||||
vld vr7, t1, 16 // 4 5 6 7
|
||||
vld vr8, t1, 4 // p[i+1] 1 2 3 4
|
||||
vld vr9, t1, 20 // 5 6 7 8
|
||||
|
||||
vld vr10, t0, -4 // P[i - 1 - REST_UNIT_STRIDE]
|
||||
vld vr11, t0, 12
|
||||
@@ -666,6 +654,144 @@ function boxsum3_sgf_v_8bpc_lsx
|
||||
bnez a5, .LBS3SGF_V_H
|
||||
endfunc
|
||||
|
||||
function boxsum3_sgf_v_8bpc_lasx
|
||||
addi.d a1, a1, (3*REST_UNIT_STRIDE+3) // src
|
||||
addi.d a2, a2, REST_UNIT_STRIDE<<2
|
||||
addi.d a2, a2, (REST_UNIT_STRIDE<<2)+12
|
||||
addi.d a3, a3, REST_UNIT_STRIDE<<2
|
||||
addi.d a3, a3, 6
|
||||
.LBS3SGF_V_H_LASX:
|
||||
// A int32_t *sumsq
|
||||
addi.d t0, a2, -(REST_UNIT_STRIDE<<2) // -stride
|
||||
addi.d t1, a2, 0 // sumsq
|
||||
addi.d t2, a2, REST_UNIT_STRIDE<<2 // +stride
|
||||
addi.d t6, a1, 0
|
||||
addi.w t7, a4, 0
|
||||
addi.d t8, a0, 0
|
||||
// B coef *sum
|
||||
addi.d t3, a3, -(REST_UNIT_STRIDE<<1) // -stride
|
||||
addi.d t4, a3, 0
|
||||
addi.d t5, a3, REST_UNIT_STRIDE<<1
|
||||
|
||||
.LBS3SGF_V_W_LASX:
|
||||
xvld xr0, t0, 0 // P[i - REST_UNIT_STRIDE]
|
||||
xvld xr1, t0, 32
|
||||
xvld xr2, t1, -4 // P[i-1] -1 0 1 2
|
||||
xvld xr3, t1, 28 // 3 4 5 6
|
||||
xvld xr4, t2, 0 // P[i + REST_UNIT_STRIDE]
|
||||
xvld xr5, t2, 32
|
||||
xvld xr6, t1, 0 // p[i] 0 1 2 3
|
||||
xvld xr7, t1, 32 // 4 5 6 7
|
||||
xvld xr8, t1, 4 // p[i+1] 1 2 3 4
|
||||
xvld xr9, t1, 36 // 5 6 7 8
|
||||
|
||||
xvld xr10, t0, -4 // P[i - 1 - REST_UNIT_STRIDE]
|
||||
xvld xr11, t0, 28
|
||||
xvld xr12, t2, -4 // P[i - 1 + REST_UNIT_STRIDE]
|
||||
xvld xr13, t2, 28
|
||||
xvld xr14, t0, 4 // P[i + 1 - REST_UNIT_STRIDE]
|
||||
xvld xr15, t0, 36
|
||||
xvld xr16, t2, 4 // P[i + 1 + REST_UNIT_STRIDE]
|
||||
xvld xr17, t2, 36
|
||||
|
||||
xvadd.w xr0, xr2, xr0
|
||||
xvadd.w xr4, xr6, xr4
|
||||
xvadd.w xr0, xr0, xr8
|
||||
xvadd.w xr20, xr0, xr4
|
||||
xvslli.w xr20, xr20, 2 // 0 1 2 3
|
||||
xvadd.w xr0, xr1, xr3
|
||||
xvadd.w xr4, xr5, xr7
|
||||
xvadd.w xr0, xr0, xr9
|
||||
xvadd.w xr21, xr0, xr4
|
||||
xvslli.w xr21, xr21, 2 // 4 5 6 7
|
||||
xvadd.w xr12, xr10, xr12
|
||||
xvadd.w xr16, xr14, xr16
|
||||
xvadd.w xr22, xr12, xr16
|
||||
xvslli.w xr23, xr22, 1
|
||||
xvadd.w xr22, xr23, xr22
|
||||
xvadd.w xr11, xr11, xr13
|
||||
xvadd.w xr15, xr15, xr17
|
||||
xvadd.w xr0, xr11, xr15
|
||||
xvslli.w xr23, xr0, 1
|
||||
xvadd.w xr23, xr23, xr0
|
||||
xvadd.w xr20, xr20, xr22 // b
|
||||
xvadd.w xr21, xr21, xr23
|
||||
|
||||
// B coef *sum
|
||||
xvld xr0, t3, 0 // P[i - REST_UNIT_STRIDE]
|
||||
xvld xr1, t4, -2 // p[i - 1]
|
||||
xvld xr2, t4, 0 // p[i]
|
||||
xvld xr3, t4, 2 // p[i + 1]
|
||||
xvld xr4, t5, 0 // P[i + REST_UNIT_STRIDE]
|
||||
xvld xr5, t3, -2 // P[i - 1 - REST_UNIT_STRIDE]
|
||||
xvld xr6, t5, -2 // P[i - 1 + REST_UNIT_STRIDE]
|
||||
xvld xr7, t3, 2 // P[i + 1 - REST_UNIT_STRIDE]
|
||||
xvld xr8, t5, 2 // P[i + 1 + REST_UNIT_STRIDE]
|
||||
|
||||
xvaddwev.w.h xr9, xr0, xr1
|
||||
xvaddwod.w.h xr10, xr0, xr1
|
||||
xvaddwev.w.h xr11, xr2, xr3
|
||||
xvaddwod.w.h xr12, xr2, xr3
|
||||
xvadd.w xr9, xr11, xr9 // 0 2 4 6 8 10 12 14
|
||||
xvadd.w xr10, xr12, xr10 // 1 3 5 7 9 11 13 15
|
||||
xvilvl.w xr11, xr10, xr9 // 0 1 2 3 8 9 10 11
|
||||
xvilvh.w xr12, xr10, xr9 // 4 5 6 7 12 13 14 15
|
||||
xvsllwil.w.h xr0, xr4, 0 // 0 1 2 3 8 9 10 11
|
||||
xvexth.w.h xr1, xr4 // 4 5 6 7 12 13 14 15
|
||||
|
||||
xvadd.w xr0, xr11, xr0
|
||||
xvadd.w xr1, xr12, xr1
|
||||
xvslli.w xr0, xr0, 2
|
||||
xvslli.w xr1, xr1, 2
|
||||
|
||||
xvaddwev.w.h xr9, xr5, xr6
|
||||
xvaddwod.w.h xr10, xr5, xr6
|
||||
xvaddwev.w.h xr11, xr7, xr8
|
||||
xvaddwod.w.h xr12, xr7, xr8
|
||||
xvadd.w xr9, xr11, xr9
|
||||
xvadd.w xr10, xr12, xr10
|
||||
xvilvl.w xr13, xr10, xr9 // 0 1 2 3 8 9 10 11
|
||||
xvilvh.w xr14, xr10, xr9 // 4 5 6 7 12 13 14 15
|
||||
|
||||
xvslli.w xr15, xr13, 1
|
||||
xvslli.w xr16, xr14, 1
|
||||
xvadd.w xr15, xr13, xr15 // a
|
||||
xvadd.w xr16, xr14, xr16
|
||||
xvadd.w xr22, xr0, xr15 // A B
|
||||
xvadd.w xr23, xr1, xr16 // C D
|
||||
|
||||
vld vr0, t6, 0 // src
|
||||
vilvh.d vr2, vr0, vr0
|
||||
vext2xv.wu.bu xr1, xr0
|
||||
vext2xv.wu.bu xr2, xr2
|
||||
xvor.v xr15, xr22, xr22 // A B
|
||||
xvpermi.q xr22, xr23, 0b00000010 // A C
|
||||
xvpermi.q xr23, xr15, 0b00110001
|
||||
xvmadd.w xr20, xr22, xr1
|
||||
xvmadd.w xr21, xr23, xr2
|
||||
xvssrlrni.h.w xr21, xr20, 9
|
||||
xvpermi.d xr22, xr21, 0b11011000
|
||||
xvst xr22, t8, 0
|
||||
addi.d t8, t8, 32
|
||||
|
||||
addi.d t0, t0, 64
|
||||
addi.d t1, t1, 64
|
||||
addi.d t2, t2, 64
|
||||
addi.d t3, t3, 32
|
||||
addi.d t4, t4, 32
|
||||
addi.d t5, t5, 32
|
||||
addi.d t6, t6, 16
|
||||
addi.w t7, t7, -16
|
||||
blt zero, t7, .LBS3SGF_V_W_LASX
|
||||
|
||||
addi.w a5, a5, -1
|
||||
addi.d a0, a0, 384*2
|
||||
addi.d a1, a1, REST_UNIT_STRIDE
|
||||
addi.d a3, a3, REST_UNIT_STRIDE<<1
|
||||
addi.d a2, a2, REST_UNIT_STRIDE<<2
|
||||
bnez a5, .LBS3SGF_V_H_LASX
|
||||
endfunc
|
||||
|
||||
#define FILTER_OUT_STRIDE (384)
|
||||
|
||||
/*
|
||||
@@ -835,20 +961,15 @@ function boxsum5_h_8bpc_lsx
|
||||
vadd.w vr6, vr6, vr20
|
||||
vadd.w vr7, vr7, vr21
|
||||
vadd.w vr8, vr8, vr22
|
||||
vmaddwev.w.hu vr5, vr11, vr11
|
||||
vmaddwod.w.hu vr6, vr11, vr11
|
||||
vmaddwev.w.hu vr7, vr12, vr12
|
||||
vmaddwod.w.hu vr8, vr12, vr12
|
||||
vilvl.w vr19, vr6, vr5
|
||||
vilvh.w vr20, vr6, vr5
|
||||
vilvl.w vr21, vr8, vr7
|
||||
vilvh.w vr22, vr8, vr7
|
||||
vmul.h vr11, vr11, vr11
|
||||
vmul.h vr12, vr12, vr12
|
||||
vsllwil.wu.hu vr0, vr11, 0
|
||||
vexth.wu.hu vr1, vr11
|
||||
vsllwil.wu.hu vr2, vr12, 0
|
||||
vexth.wu.hu vr3, vr12
|
||||
vadd.w vr19, vr19, vr0
|
||||
vadd.w vr20, vr20, vr1
|
||||
vadd.w vr21, vr21, vr2
|
||||
vadd.w vr22, vr22, vr3
|
||||
|
||||
vst vr19, t0, 0
|
||||
vst vr20, t0, 16
|
||||
vst vr21, t0, 32
|
||||
@@ -921,7 +1042,7 @@ function boxsum5_v_8bpc_lsx
|
||||
addi.d t0, t0, 32
|
||||
addi.d t2, t2, 32
|
||||
addi.w t4, t4, -8
|
||||
ble t4, zero, .LBOXSUM5_V_H1
|
||||
bge zero, t4, .LBOXSUM5_V_H1
|
||||
|
||||
.LBOXSUM5_V_W:
|
||||
vld vr0, t1, 0 // a 0 1 2 3 4 5 6 7
|
||||
@@ -1405,3 +1526,431 @@ function sgr_mix_finish_8bpc_lsx
|
||||
|
||||
.LSGR_MIX_END:
|
||||
endfunc
|
||||
|
||||
.macro MADD_HU_BU_LASX in0, in1, out0, out1
|
||||
xvsllwil.hu.bu xr12, \in0, 0
|
||||
xvexth.hu.bu xr13, \in0
|
||||
xvmadd.h \out0, xr12, \in1
|
||||
xvmadd.h \out1, xr13, \in1
|
||||
.endm
|
||||
|
||||
const wiener_shuf_lasx
|
||||
.byte 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18
|
||||
.byte 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18
|
||||
endconst
|
||||
|
||||
function wiener_filter_h_8bpc_lasx
|
||||
addi.d sp, sp, -40
|
||||
fst.d f24, sp, 0
|
||||
fst.d f25, sp, 8
|
||||
fst.d f26, sp, 16
|
||||
fst.d f27, sp, 24
|
||||
fst.d f28, sp, 32
|
||||
li.w t7, 1<<14 // clip_limit
|
||||
|
||||
la.local t1, wiener_shuf_lasx
|
||||
xvld xr4, t1, 0
|
||||
vld vr27, a2, 0 // filter[0][k]
|
||||
xvpermi.q xr14, xr27, 0b00000000
|
||||
xvrepl128vei.h xr21, xr14, 0
|
||||
xvrepl128vei.h xr22, xr14, 1
|
||||
xvrepl128vei.h xr23, xr14, 2
|
||||
xvrepl128vei.h xr24, xr14, 3
|
||||
xvrepl128vei.h xr25, xr14, 4
|
||||
xvrepl128vei.h xr26, xr14, 5
|
||||
xvrepl128vei.h xr27, xr14, 6
|
||||
xvreplgr2vr.w xr0, t7
|
||||
|
||||
.WIENER_FILTER_H_H_LASX:
|
||||
addi.w a4, a4, -1 // h
|
||||
addi.w t0, a3, 0 // w
|
||||
addi.d t1, a1, 0 // tmp_ptr
|
||||
addi.d t2, a0, 0 // hor_ptr
|
||||
|
||||
.WIENER_FILTER_H_W_LASX:
|
||||
addi.w t0, t0, -32
|
||||
xvld xr5, t1, 0
|
||||
xvld xr13, t1, 16
|
||||
|
||||
xvsubi.bu xr14, xr4, 2
|
||||
xvsubi.bu xr15, xr4, 1
|
||||
xvshuf.b xr6, xr13, xr5, xr14 // 1 ... 8, 9 ... 16
|
||||
xvshuf.b xr7, xr13, xr5, xr15 // 2 ... 9, 10 ... 17
|
||||
xvshuf.b xr8, xr13, xr5, xr4 // 3 ... 10, 11 ... 18
|
||||
xvaddi.bu xr14, xr4, 1
|
||||
xvaddi.bu xr15, xr4, 2
|
||||
xvshuf.b xr9, xr13, xr5, xr14 // 4 ... 11, 12 ... 19
|
||||
xvshuf.b xr10, xr13, xr5, xr15 // 5 ... 12, 13 ... 20
|
||||
xvaddi.bu xr14, xr4, 3
|
||||
xvshuf.b xr11, xr13, xr5, xr14 // 6 ... 13, 14 ... 21
|
||||
|
||||
xvsllwil.hu.bu xr15, xr8, 0 // 3 4 5 6 7 8 9 10
|
||||
xvexth.hu.bu xr16, xr8 // 11 12 13 14 15 16 17 18
|
||||
xvsllwil.wu.hu xr17, xr15, 7 // 3 4 5 6
|
||||
xvexth.wu.hu xr18, xr15 // 7 8 9 10
|
||||
xvsllwil.wu.hu xr19, xr16, 7 // 11 12 13 14
|
||||
xvexth.wu.hu xr20, xr16 // 15 16 17 18
|
||||
xvslli.w xr18, xr18, 7
|
||||
xvslli.w xr20, xr20, 7
|
||||
xvxor.v xr15, xr15, xr15
|
||||
xvxor.v xr14, xr14, xr14
|
||||
|
||||
MADD_HU_BU_LASX xr5, xr21, xr14, xr15
|
||||
MADD_HU_BU_LASX xr6, xr22, xr14, xr15
|
||||
MADD_HU_BU_LASX xr7, xr23, xr14, xr15
|
||||
MADD_HU_BU_LASX xr8, xr24, xr14, xr15
|
||||
MADD_HU_BU_LASX xr9, xr25, xr14, xr15
|
||||
MADD_HU_BU_LASX xr10, xr26, xr14, xr15
|
||||
MADD_HU_BU_LASX xr11, xr27, xr14, xr15
|
||||
|
||||
xvsllwil.w.h xr5, xr14, 0 // 0 1 2 3
|
||||
xvexth.w.h xr6, xr14 // 4 5 6 7
|
||||
xvsllwil.w.h xr7, xr15, 0 // 8 9 10 11
|
||||
xvexth.w.h xr8, xr15 // 12 13 14 15
|
||||
xvadd.w xr17, xr17, xr5
|
||||
xvadd.w xr18, xr18, xr6
|
||||
xvadd.w xr19, xr19, xr7
|
||||
xvadd.w xr20, xr20, xr8
|
||||
xvadd.w xr17, xr17, xr0
|
||||
xvadd.w xr18, xr18, xr0
|
||||
xvadd.w xr19, xr19, xr0
|
||||
xvadd.w xr20, xr20, xr0
|
||||
|
||||
xvsrli.w xr1, xr0, 1
|
||||
xvsubi.wu xr1, xr1, 1
|
||||
xvxor.v xr3, xr3, xr3
|
||||
xvsrari.w xr17, xr17, 3
|
||||
xvsrari.w xr18, xr18, 3
|
||||
xvsrari.w xr19, xr19, 3
|
||||
xvsrari.w xr20, xr20, 3
|
||||
xvclip.w xr17, xr17, xr3, xr1
|
||||
xvclip.w xr18, xr18, xr3, xr1
|
||||
xvclip.w xr19, xr19, xr3, xr1
|
||||
xvclip.w xr20, xr20, xr3, xr1
|
||||
|
||||
xvor.v xr5, xr17, xr17
|
||||
xvor.v xr6, xr19, xr19
|
||||
xvpermi.q xr17, xr18, 0b00000010
|
||||
xvpermi.q xr19, xr20, 0b00000010
|
||||
|
||||
xvst xr17, t2, 0
|
||||
xvst xr19, t2, 32
|
||||
xvpermi.q xr18, xr5, 0b00110001
|
||||
xvpermi.q xr20, xr6, 0b00110001
|
||||
xvst xr18, t2, 64
|
||||
xvst xr20, t2, 96
|
||||
addi.d t1, t1, 32
|
||||
addi.d t2, t2, 128
|
||||
blt zero, t0, .WIENER_FILTER_H_W_LASX
|
||||
|
||||
addi.d a1, a1, REST_UNIT_STRIDE
|
||||
addi.d a0, a0, (REST_UNIT_STRIDE << 2)
|
||||
bnez a4, .WIENER_FILTER_H_H_LASX
|
||||
|
||||
fld.d f24, sp, 0
|
||||
fld.d f25, sp, 8
|
||||
fld.d f26, sp, 16
|
||||
fld.d f27, sp, 24
|
||||
fld.d f28, sp, 32
|
||||
addi.d sp, sp, 40
|
||||
endfunc
|
||||
|
||||
.macro APPLY_FILTER_LASX in0, in1, in2
|
||||
alsl.d t7, \in0, \in1, 2
|
||||
xvld xr10, t7, 0
|
||||
xvld xr12, t7, 32
|
||||
xvmadd.w xr14, xr10, \in2
|
||||
xvmadd.w xr16, xr12, \in2
|
||||
.endm
|
||||
|
||||
.macro wiener_filter_v_8bpc_core_lasx
|
||||
xvreplgr2vr.w xr14, t6
|
||||
xvreplgr2vr.w xr16, t6
|
||||
|
||||
addi.w t7, t2, 0 // j + index k
|
||||
mul.w t7, t7, t8 // (j + index) * REST_UNIT_STRIDE
|
||||
add.w t7, t7, t4 // (j + index) * REST_UNIT_STRIDE + i
|
||||
|
||||
APPLY_FILTER_LASX t7, a2, xr2
|
||||
APPLY_FILTER_LASX t8, t7, xr3
|
||||
APPLY_FILTER_LASX t8, t7, xr4
|
||||
APPLY_FILTER_LASX t8, t7, xr5
|
||||
APPLY_FILTER_LASX t8, t7, xr6
|
||||
APPLY_FILTER_LASX t8, t7, xr7
|
||||
APPLY_FILTER_LASX t8, t7, xr8
|
||||
xvssrarni.hu.w xr16, xr14, 11
|
||||
xvpermi.d xr17, xr16, 0b11011000
|
||||
xvssrlni.bu.h xr17, xr17, 0
|
||||
xvpermi.d xr17, xr17, 0b00001000
|
||||
.endm
|
||||
|
||||
function wiener_filter_v_8bpc_lasx
|
||||
li.w t6, -(1 << 18)
|
||||
|
||||
li.w t8, REST_UNIT_STRIDE
|
||||
ld.h t0, a3, 0
|
||||
ld.h t1, a3, 2
|
||||
xvreplgr2vr.w xr2, t0
|
||||
xvreplgr2vr.w xr3, t1
|
||||
ld.h t0, a3, 4
|
||||
ld.h t1, a3, 6
|
||||
xvreplgr2vr.w xr4, t0
|
||||
xvreplgr2vr.w xr5, t1
|
||||
ld.h t0, a3, 8
|
||||
ld.h t1, a3, 10
|
||||
xvreplgr2vr.w xr6, t0
|
||||
xvreplgr2vr.w xr7, t1
|
||||
ld.h t0, a3, 12
|
||||
xvreplgr2vr.w xr8, t0
|
||||
|
||||
andi t1, a4, 0xf
|
||||
sub.w t0, a4, t1 // w-w%16
|
||||
or t2, zero, zero // j
|
||||
or t4, zero, zero
|
||||
beqz t0, .WIENER_FILTER_V_W_LT16_LASX
|
||||
|
||||
.WIENER_FILTER_V_H_LASX:
|
||||
andi t1, a4, 0xf
|
||||
add.d t3, zero, a0 // p
|
||||
or t4, zero, zero // i
|
||||
|
||||
.WIENER_FILTER_V_W_LASX:
|
||||
|
||||
wiener_filter_v_8bpc_core_lasx
|
||||
|
||||
mul.w t5, t2, a1 // j * stride
|
||||
add.w t5, t5, t4 // j * stride + i
|
||||
add.d t3, a0, t5
|
||||
addi.w t4, t4, 16
|
||||
vst vr17, t3, 0
|
||||
bne t0, t4, .WIENER_FILTER_V_W_LASX
|
||||
|
||||
beqz t1, .WIENER_FILTER_V_W_EQ16_LASX
|
||||
|
||||
wiener_filter_v_8bpc_core_lsx
|
||||
|
||||
addi.d t3, t3, 16
|
||||
andi t1, a4, 0xf
|
||||
|
||||
.WIENER_FILTER_V_ST_REM_LASX:
|
||||
vstelm.b vr17, t3, 0, 0
|
||||
vbsrl.v vr17, vr17, 1
|
||||
addi.d t3, t3, 1
|
||||
addi.w t1, t1, -1
|
||||
bnez t1, .WIENER_FILTER_V_ST_REM_LASX
|
||||
.WIENER_FILTER_V_W_EQ16_LASX:
|
||||
addi.w t2, t2, 1
|
||||
blt t2, a5, .WIENER_FILTER_V_H_LASX
|
||||
b .WIENER_FILTER_V_LASX_END
|
||||
|
||||
.WIENER_FILTER_V_W_LT16_LASX:
|
||||
andi t1, a4, 0xf
|
||||
add.d t3, zero, a0
|
||||
|
||||
wiener_filter_v_8bpc_core_lsx
|
||||
|
||||
mul.w t5, t2, a1 // j * stride
|
||||
add.d t3, a0, t5
|
||||
|
||||
.WIENER_FILTER_V_ST_REM_1_LASX:
|
||||
vstelm.b vr17, t3, 0, 0
|
||||
vbsrl.v vr17, vr17, 1
|
||||
addi.d t3, t3, 1
|
||||
addi.w t1, t1, -1
|
||||
bnez t1, .WIENER_FILTER_V_ST_REM_1_LASX
|
||||
|
||||
addi.w t2, t2, 1
|
||||
blt t2, a5, .WIENER_FILTER_V_W_LT16_LASX
|
||||
|
||||
.WIENER_FILTER_V_LASX_END:
|
||||
endfunc
|
||||
|
||||
function boxsum3_sgf_h_8bpc_lasx
|
||||
addi.d a0, a0, (REST_UNIT_STRIDE<<2)+12 // AA
|
||||
//addi.d a0, a0, 12 // AA
|
||||
addi.d a1, a1, (REST_UNIT_STRIDE<<1)+6 // BB
|
||||
//addi.d a1, a1, 6 // BB
|
||||
la.local t8, dav1d_sgr_x_by_x
|
||||
li.w t6, 455
|
||||
xvreplgr2vr.w xr20, t6
|
||||
li.w t6, 255
|
||||
xvreplgr2vr.w xr22, t6
|
||||
xvaddi.wu xr21, xr22, 1 // 256
|
||||
xvreplgr2vr.w xr6, a4
|
||||
xvldi xr19, 0x809
|
||||
addi.w a2, a2, 2 // w + 2
|
||||
addi.w a3, a3, 2 // h + 2
|
||||
|
||||
.LBS3SGF_H_H_LASX:
|
||||
addi.w t2, a2, 0
|
||||
addi.d t0, a0, -4
|
||||
addi.d t1, a1, -2
|
||||
|
||||
.LBS3SGF_H_W_LASX:
|
||||
addi.w t2, t2, -16
|
||||
xvld xr0, t0, 0 // AA[i]
|
||||
xvld xr1, t0, 32
|
||||
xvld xr2, t1, 0 // BB[i]
|
||||
|
||||
xvmul.w xr4, xr0, xr19 // a * n
|
||||
xvmul.w xr5, xr1, xr19
|
||||
vext2xv.w.h xr9, xr2
|
||||
xvpermi.q xr10, xr2, 0b00000001
|
||||
vext2xv.w.h xr10, xr10
|
||||
xvmsub.w xr4, xr9, xr9 // p
|
||||
xvmsub.w xr5, xr10, xr10
|
||||
xvmaxi.w xr4, xr4, 0
|
||||
xvmaxi.w xr5, xr5, 0
|
||||
xvmul.w xr4, xr4, xr6 // p * s
|
||||
xvmul.w xr5, xr5, xr6
|
||||
xvsrlri.w xr4, xr4, 20
|
||||
xvsrlri.w xr5, xr5, 20
|
||||
xvmin.w xr4, xr4, xr22
|
||||
xvmin.w xr5, xr5, xr22
|
||||
|
||||
vpickve2gr.w t6, vr4, 0
|
||||
ldx.bu t7, t8, t6
|
||||
vinsgr2vr.w vr7, t7, 0
|
||||
vpickve2gr.w t6, vr4, 1
|
||||
ldx.bu t7, t8, t6
|
||||
vinsgr2vr.w vr7, t7, 1
|
||||
vpickve2gr.w t6, vr4, 2
|
||||
ldx.bu t7, t8, t6
|
||||
vinsgr2vr.w vr7, t7, 2
|
||||
vpickve2gr.w t6, vr4, 3
|
||||
ldx.bu t7, t8, t6
|
||||
vinsgr2vr.w vr7, t7, 3
|
||||
|
||||
xvpickve2gr.w t6, xr4, 4
|
||||
ldx.bu t7, t8, t6
|
||||
xvinsgr2vr.w xr7, t7, 4
|
||||
xvpickve2gr.w t6, xr4, 5
|
||||
ldx.bu t7, t8, t6
|
||||
xvinsgr2vr.w xr7, t7, 5
|
||||
xvpickve2gr.w t6, xr4, 6
|
||||
ldx.bu t7, t8, t6
|
||||
xvinsgr2vr.w xr7, t7, 6
|
||||
xvpickve2gr.w t6, xr4, 7
|
||||
ldx.bu t7, t8, t6
|
||||
xvinsgr2vr.w xr7, t7, 7 // x
|
||||
|
||||
vpickve2gr.w t6, vr5, 0
|
||||
ldx.bu t7, t8, t6
|
||||
vinsgr2vr.w vr8, t7, 0
|
||||
vpickve2gr.w t6, vr5, 1
|
||||
ldx.bu t7, t8, t6
|
||||
vinsgr2vr.w vr8, t7, 1
|
||||
vpickve2gr.w t6, vr5, 2
|
||||
ldx.bu t7, t8, t6
|
||||
vinsgr2vr.w vr8, t7, 2
|
||||
vpickve2gr.w t6, vr5, 3
|
||||
ldx.bu t7, t8, t6
|
||||
vinsgr2vr.w vr8, t7, 3
|
||||
|
||||
xvpickve2gr.w t6, xr5, 4
|
||||
ldx.bu t7, t8, t6
|
||||
xvinsgr2vr.w xr8, t7, 4
|
||||
xvpickve2gr.w t6, xr5, 5
|
||||
ldx.bu t7, t8, t6
|
||||
xvinsgr2vr.w xr8, t7, 5
|
||||
xvpickve2gr.w t6, xr5, 6
|
||||
ldx.bu t7, t8, t6
|
||||
xvinsgr2vr.w xr8, t7, 6
|
||||
xvpickve2gr.w t6, xr5, 7
|
||||
ldx.bu t7, t8, t6
|
||||
xvinsgr2vr.w xr8, t7, 7 // x
|
||||
|
||||
xvmul.w xr9, xr7, xr9 // x * BB[i]
|
||||
xvmul.w xr10, xr8, xr10
|
||||
xvmul.w xr9, xr9, xr20 // x * BB[i] * sgr_one_by_x
|
||||
xvmul.w xr10, xr10, xr20
|
||||
xvsrlri.w xr9, xr9, 12
|
||||
xvsrlri.w xr10, xr10, 12
|
||||
xvsub.w xr7, xr21, xr7
|
||||
xvsub.w xr8, xr21, xr8
|
||||
xvpickev.h xr12, xr8, xr7
|
||||
xvpermi.d xr11, xr12, 0b11011000
|
||||
|
||||
xvst xr9, t0, 0
|
||||
xvst xr10, t0, 32
|
||||
xvst xr11, t1, 0
|
||||
addi.d t0, t0, 64
|
||||
addi.d t1, t1, 32
|
||||
blt zero, t2, .LBS3SGF_H_W_LASX
|
||||
|
||||
addi.d a0, a0, REST_UNIT_STRIDE<<2
|
||||
addi.d a1, a1, REST_UNIT_STRIDE<<1
|
||||
addi.w a3, a3, -1
|
||||
bnez a3, .LBS3SGF_H_H_LASX
|
||||
endfunc
|
||||
|
||||
function boxsum3_h_8bpc_lasx
|
||||
addi.d a2, a2, REST_UNIT_STRIDE
|
||||
li.w t0, 1
|
||||
addi.w a3, a3, -2
|
||||
addi.w a4, a4, -4
|
||||
.LBS3_H_H_LASX:
|
||||
alsl.d t1, t0, a1, 1 // sum_v *sum_v = sum + x
|
||||
alsl.d t2, t0, a0, 2 // sumsq_v *sumsq_v = sumsq + x
|
||||
add.d t3, t0, a2 // s
|
||||
addi.w t5, a3, 0
|
||||
|
||||
.LBS3_H_W_LASX:
|
||||
xvld xr0, t3, 0
|
||||
xvld xr1, t3, REST_UNIT_STRIDE
|
||||
xvld xr2, t3, (REST_UNIT_STRIDE<<1)
|
||||
|
||||
xvilvl.b xr3, xr1, xr0
|
||||
xvhaddw.hu.bu xr4, xr3, xr3
|
||||
xvilvh.b xr5, xr1, xr0
|
||||
xvhaddw.hu.bu xr6, xr5, xr5
|
||||
xvsllwil.hu.bu xr7, xr2, 0
|
||||
xvexth.hu.bu xr8, xr2
|
||||
// sum_v
|
||||
xvadd.h xr4, xr4, xr7 // 0 2
|
||||
xvadd.h xr6, xr6, xr8 // 1 3
|
||||
xvor.v xr9, xr4, xr4
|
||||
xvpermi.q xr4, xr6, 0b00000010
|
||||
xvpermi.q xr6, xr9, 0b00110001
|
||||
xvst xr4, t1, REST_UNIT_STRIDE<<1
|
||||
xvst xr6, t1, (REST_UNIT_STRIDE<<1)+32
|
||||
addi.d t1, t1, 64
|
||||
// sumsq
|
||||
xvmulwev.h.bu xr9, xr3, xr3
|
||||
xvmulwod.h.bu xr10, xr3, xr3
|
||||
xvmulwev.h.bu xr11, xr5, xr5
|
||||
xvmulwod.h.bu xr12, xr5, xr5
|
||||
xvaddwev.w.hu xr13, xr10, xr9
|
||||
xvaddwod.w.hu xr14, xr10, xr9
|
||||
xvaddwev.w.hu xr15, xr12, xr11
|
||||
xvaddwod.w.hu xr16, xr12, xr11
|
||||
xvmaddwev.w.hu xr13, xr7, xr7
|
||||
xvmaddwod.w.hu xr14, xr7, xr7
|
||||
xvmaddwev.w.hu xr15, xr8, xr8
|
||||
xvmaddwod.w.hu xr16, xr8, xr8
|
||||
xvilvl.w xr9, xr14, xr13
|
||||
xvilvh.w xr10, xr14, xr13
|
||||
xvilvl.w xr11, xr16, xr15
|
||||
xvilvh.w xr12, xr16, xr15
|
||||
xvor.v xr7, xr9, xr9
|
||||
xvor.v xr8, xr11, xr11
|
||||
xvpermi.q xr9, xr10, 0b00000010
|
||||
xvpermi.q xr10, xr7, 0b00110001
|
||||
xvpermi.q xr11, xr12, 0b00000010
|
||||
xvpermi.q xr12, xr8, 0b00110001
|
||||
xvst xr9, t2, REST_UNIT_STRIDE<<2
|
||||
xvst xr11, t2, (REST_UNIT_STRIDE<<2)+32
|
||||
xvst xr10, t2, (REST_UNIT_STRIDE<<2)+64
|
||||
xvst xr12, t2, (REST_UNIT_STRIDE<<2)+96
|
||||
|
||||
addi.d t2, t2, 128
|
||||
addi.w t5, t5, -32
|
||||
addi.d t3, t3, 32
|
||||
blt zero, t5, .LBS3_H_W_LASX
|
||||
|
||||
addi.d a0, a0, REST_UNIT_STRIDE<<2
|
||||
addi.d a1, a1, REST_UNIT_STRIDE<<1
|
||||
addi.d a2, a2, REST_UNIT_STRIDE
|
||||
addi.d a4, a4, -1
|
||||
blt zero, a4, .LBS3_H_H_LASX
|
||||
endfunc
|
||||
|
||||
@@ -39,6 +39,13 @@ void dav1d_wiener_filter_lsx(uint8_t *p, const ptrdiff_t stride,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX);
|
||||
|
||||
void dav1d_wiener_filter_lasx(uint8_t *p, const ptrdiff_t stride,
|
||||
const uint8_t (*const left)[4],
|
||||
const uint8_t *lpf,
|
||||
const int w, const int h,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX);
|
||||
|
||||
void dav1d_sgr_filter_3x3_lsx(pixel *p, const ptrdiff_t p_stride,
|
||||
const pixel (*const left)[4],
|
||||
const pixel *lpf,
|
||||
@@ -46,6 +53,13 @@ void dav1d_sgr_filter_3x3_lsx(pixel *p, const ptrdiff_t p_stride,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX);
|
||||
|
||||
void dav1d_sgr_filter_3x3_lasx(pixel *p, const ptrdiff_t p_stride,
|
||||
const pixel (*const left)[4],
|
||||
const pixel *lpf,
|
||||
const int w, const int h,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX);
|
||||
|
||||
void dav1d_sgr_filter_5x5_lsx(pixel *p, const ptrdiff_t p_stride,
|
||||
const pixel (*const left)[4],
|
||||
const pixel *lpf,
|
||||
@@ -60,6 +74,13 @@ void dav1d_sgr_filter_mix_lsx(pixel *p, const ptrdiff_t p_stride,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX);
|
||||
|
||||
void dav1d_sgr_filter_mix_lasx(pixel *p, const ptrdiff_t p_stride,
|
||||
const pixel (*const left)[4],
|
||||
const pixel *lpf,
|
||||
const int w, const int h,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX);
|
||||
|
||||
static ALWAYS_INLINE void loop_restoration_dsp_init_loongarch(Dav1dLoopRestorationDSPContext *const c, int bpc)
|
||||
{
|
||||
const unsigned flags = dav1d_get_cpu_flags();
|
||||
@@ -73,6 +94,14 @@ static ALWAYS_INLINE void loop_restoration_dsp_init_loongarch(Dav1dLoopRestorati
|
||||
c->sgr[1] = dav1d_sgr_filter_3x3_lsx;
|
||||
c->sgr[2] = dav1d_sgr_filter_mix_lsx;
|
||||
#endif
|
||||
|
||||
if (!(flags & DAV1D_LOONGARCH_CPU_FLAG_LASX)) return;
|
||||
|
||||
#if BITDEPTH == 8
|
||||
c->wiener[0] = c->wiener[1] = dav1d_wiener_filter_lasx;
|
||||
|
||||
c->sgr[1] = dav1d_sgr_filter_3x3_lasx;
|
||||
#endif
|
||||
}
|
||||
|
||||
#endif /* DAV1D_SRC_LOONGARCH_LOOPRESTORATION_H */
|
||||
|
||||
@@ -36,12 +36,23 @@ void BF(dav1d_wiener_filter_h, lsx)(int32_t *hor_ptr,
|
||||
const int16_t filterh[8],
|
||||
const int w, const int h);
|
||||
|
||||
void BF(dav1d_wiener_filter_h, lasx)(int32_t *hor_ptr,
|
||||
uint8_t *tmp_ptr,
|
||||
const int16_t filterh[8],
|
||||
const int w, const int h);
|
||||
|
||||
void BF(dav1d_wiener_filter_v, lsx)(uint8_t *p,
|
||||
const ptrdiff_t p_stride,
|
||||
const int32_t *hor,
|
||||
const int16_t filterv[8],
|
||||
const int w, const int h);
|
||||
|
||||
void BF(dav1d_wiener_filter_v, lasx)(uint8_t *p,
|
||||
const ptrdiff_t p_stride,
|
||||
const int32_t *hor,
|
||||
const int16_t filterv[8],
|
||||
const int w, const int h);
|
||||
|
||||
// This function refers to the function in the ppc/looprestoration_init_tmpl.c.
|
||||
static inline void padding(uint8_t *dst, const uint8_t *p,
|
||||
const ptrdiff_t stride, const uint8_t (*left)[4],
|
||||
@@ -156,20 +167,46 @@ void dav1d_wiener_filter_lsx(uint8_t *p, const ptrdiff_t p_stride,
|
||||
BF(dav1d_wiener_filter_v, lsx)(p, p_stride, hor, filter[1], w, h);
|
||||
}
|
||||
|
||||
void dav1d_wiener_filter_lasx(uint8_t *p, const ptrdiff_t p_stride,
|
||||
const uint8_t (*const left)[4],
|
||||
const uint8_t *lpf,
|
||||
const int w, const int h,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
const int16_t (*const filter)[8] = params->filter;
|
||||
|
||||
// Wiener filtering is applied to a maximum stripe height of 64 + 3 pixels
|
||||
// of padding above and below
|
||||
ALIGN_STK_16(uint8_t, tmp, 70 /*(64 + 3 + 3)*/ * REST_UNIT_STRIDE,);
|
||||
padding(tmp, p, p_stride, left, lpf, w, h, edges);
|
||||
ALIGN_STK_16(int32_t, hor, 70 /*(64 + 3 + 3)*/ * REST_UNIT_STRIDE + 64,);
|
||||
|
||||
BF(dav1d_wiener_filter_h, lasx)(hor, tmp, filter[0], w, h + 6);
|
||||
BF(dav1d_wiener_filter_v, lasx)(p, p_stride, hor, filter[1], w, h);
|
||||
}
|
||||
|
||||
void BF(dav1d_boxsum3_h, lsx)(int32_t *sumsq, int16_t *sum, pixel *src,
|
||||
const int w, const int h);
|
||||
void BF(dav1d_boxsum3_v, lsx)(int32_t *sumsq, int16_t *sum,
|
||||
const int w, const int h);
|
||||
|
||||
void BF(dav1d_boxsum3_sgf_h, lsx)(int32_t *sumsq, int16_t *sum,
|
||||
const int w, const int h, const int w1);
|
||||
const int w, const int h, const int w1);
|
||||
void BF(dav1d_boxsum3_sgf_v, lsx)(int16_t *dst, uint8_t *tmp,
|
||||
int32_t *sumsq, int16_t *sum,
|
||||
const int w, const int h);
|
||||
int32_t *sumsq, int16_t *sum,
|
||||
const int w, const int h);
|
||||
void BF(dav1d_sgr_3x3_finish, lsx)(pixel *p, const ptrdiff_t p_stride,
|
||||
int16_t *dst, int w1,
|
||||
const int w, const int h);
|
||||
int16_t *dst, int w1,
|
||||
const int w, const int h);
|
||||
|
||||
void BF(dav1d_boxsum3_h, lasx)(int32_t *sumsq, int16_t *sum, pixel *src,
|
||||
const int w, const int h);
|
||||
void BF(dav1d_boxsum3_sgf_h, lasx)(int32_t *sumsq, int16_t *sum,
|
||||
const int w, const int h, const int w1);
|
||||
void BF(dav1d_boxsum3_sgf_v, lasx)(int16_t *dst, uint8_t *tmp,
|
||||
int32_t *sumsq, int16_t *sum,
|
||||
const int w, const int h);
|
||||
|
||||
static inline void boxsum3_lsx(int32_t *sumsq, coef *sum, pixel *src,
|
||||
const int w, const int h)
|
||||
@@ -178,6 +215,13 @@ static inline void boxsum3_lsx(int32_t *sumsq, coef *sum, pixel *src,
|
||||
BF(dav1d_boxsum3_v, lsx)(sumsq, sum, w + 6, h + 6);
|
||||
}
|
||||
|
||||
static inline void boxsum3_lasx(int32_t *sumsq, coef *sum, pixel *src,
|
||||
const int w, const int h)
|
||||
{
|
||||
BF(dav1d_boxsum3_h, lasx)(sumsq, sum, src, w + 6, h + 6);
|
||||
BF(dav1d_boxsum3_v, lsx)(sumsq, sum, w + 6, h + 6);
|
||||
}
|
||||
|
||||
void dav1d_sgr_filter_3x3_lsx(pixel *p, const ptrdiff_t p_stride,
|
||||
const pixel (*const left)[4],
|
||||
const pixel *lpf,
|
||||
@@ -198,6 +242,26 @@ void dav1d_sgr_filter_3x3_lsx(pixel *p, const ptrdiff_t p_stride,
|
||||
BF(dav1d_sgr_3x3_finish, lsx)(p, p_stride, dst, params->sgr.w1, w, h);
|
||||
}
|
||||
|
||||
void dav1d_sgr_filter_3x3_lasx(pixel *p, const ptrdiff_t p_stride,
|
||||
const pixel (*const left)[4],
|
||||
const pixel *lpf,
|
||||
const int w, const int h,
|
||||
const LooprestorationParams *const params,
|
||||
const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
|
||||
{
|
||||
ALIGN_STK_16(uint8_t, tmp, 70 /*(64 + 3 + 3)*/ * REST_UNIT_STRIDE,);
|
||||
padding(tmp, p, p_stride, left, lpf, w, h, edges);
|
||||
coef dst[64 * 384];
|
||||
|
||||
ALIGN_STK_16(int32_t, sumsq, 68 * REST_UNIT_STRIDE + 8, );
|
||||
ALIGN_STK_16(int16_t, sum, 68 * REST_UNIT_STRIDE + 16, );
|
||||
|
||||
boxsum3_lasx(sumsq, sum, tmp, w, h);
|
||||
BF(dav1d_boxsum3_sgf_h, lasx)(sumsq, sum, w, h, params->sgr.s1);
|
||||
BF(dav1d_boxsum3_sgf_v, lasx)(dst, tmp, sumsq, sum, w, h);
|
||||
BF(dav1d_sgr_3x3_finish, lsx)(p, p_stride, dst, params->sgr.w1, w, h);
|
||||
}
|
||||
|
||||
void BF(dav1d_boxsum5_h, lsx)(int32_t *sumsq, int16_t *sum,
|
||||
const uint8_t *const src,
|
||||
const int w, const int h);
|
||||
+3164
-1777
File diff suppressed because it is too large
Load Diff
+14
-36
@@ -43,16 +43,12 @@ decl_mask_fn(BF(dav1d_mask, lsx));
|
||||
decl_warp8x8_fn(BF(dav1d_warp_affine_8x8, lsx));
|
||||
decl_warp8x8t_fn(BF(dav1d_warp_affine_8x8t, lsx));
|
||||
decl_w_mask_fn(BF(dav1d_w_mask_420, lsx));
|
||||
decl_blend_fn(BF(dav1d_blend, lsx));
|
||||
decl_blend_dir_fn(BF(dav1d_blend_v, lsx));
|
||||
decl_blend_dir_fn(BF(dav1d_blend_h, lsx));
|
||||
decl_emu_edge_fn(BF(dav1d_emu_edge, lsx));
|
||||
|
||||
decl_mc_fn(BF(dav1d_put_8tap_regular, lsx));
|
||||
decl_mc_fn(BF(dav1d_put_8tap_regular_smooth, lsx));
|
||||
decl_mc_fn(BF(dav1d_put_8tap_regular_sharp, lsx));
|
||||
decl_mc_fn(BF(dav1d_put_8tap_smooth, lsx));
|
||||
decl_mc_fn(BF(dav1d_put_8tap_smooth_regular, lsx));
|
||||
decl_mc_fn(BF(dav1d_put_8tap_smooth_sharp, lsx));
|
||||
decl_mc_fn(BF(dav1d_put_8tap_sharp, lsx));
|
||||
decl_mc_fn(BF(dav1d_put_8tap_sharp_regular, lsx));
|
||||
decl_mc_fn(BF(dav1d_put_8tap_sharp_smooth, lsx));
|
||||
decl_8tap_fns(lsx);
|
||||
|
||||
decl_avg_fn(BF(dav1d_avg, lasx));
|
||||
decl_w_avg_fn(BF(dav1d_w_avg, lasx));
|
||||
@@ -60,16 +56,9 @@ decl_mask_fn(BF(dav1d_mask, lasx));
|
||||
decl_warp8x8_fn(BF(dav1d_warp_affine_8x8, lasx));
|
||||
decl_warp8x8t_fn(BF(dav1d_warp_affine_8x8t, lasx));
|
||||
decl_w_mask_fn(BF(dav1d_w_mask_420, lasx));
|
||||
decl_blend_dir_fn(BF(dav1d_blend_h, lasx));
|
||||
|
||||
decl_mct_fn(BF(dav1d_prep_8tap_regular, lasx));
|
||||
decl_mct_fn(BF(dav1d_prep_8tap_regular_smooth, lasx));
|
||||
decl_mct_fn(BF(dav1d_prep_8tap_regular_sharp, lasx));
|
||||
decl_mct_fn(BF(dav1d_prep_8tap_smooth, lasx));
|
||||
decl_mct_fn(BF(dav1d_prep_8tap_smooth_regular, lasx));
|
||||
decl_mct_fn(BF(dav1d_prep_8tap_smooth_sharp, lasx));
|
||||
decl_mct_fn(BF(dav1d_prep_8tap_sharp, lasx));
|
||||
decl_mct_fn(BF(dav1d_prep_8tap_sharp_regular, lasx));
|
||||
decl_mct_fn(BF(dav1d_prep_8tap_sharp_smooth, lasx));
|
||||
decl_8tap_gen(mct, prep, lasx);
|
||||
|
||||
static ALWAYS_INLINE void mc_dsp_init_loongarch(Dav1dMCDSPContext *const c) {
|
||||
#if BITDEPTH == 8
|
||||
@@ -83,16 +72,12 @@ static ALWAYS_INLINE void mc_dsp_init_loongarch(Dav1dMCDSPContext *const c) {
|
||||
c->warp8x8 = BF(dav1d_warp_affine_8x8, lsx);
|
||||
c->warp8x8t = BF(dav1d_warp_affine_8x8t, lsx);
|
||||
c->w_mask[2] = BF(dav1d_w_mask_420, lsx);
|
||||
c->blend = BF(dav1d_blend, lsx);
|
||||
c->blend_v = BF(dav1d_blend_v, lsx);
|
||||
c->blend_h = BF(dav1d_blend_h, lsx);
|
||||
c->emu_edge = BF(dav1d_emu_edge, lsx);
|
||||
|
||||
init_mc_fn(FILTER_2D_8TAP_REGULAR, 8tap_regular, lsx);
|
||||
init_mc_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, lsx);
|
||||
init_mc_fn(FILTER_2D_8TAP_REGULAR_SHARP, 8tap_regular_sharp, lsx);
|
||||
init_mc_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, lsx);
|
||||
init_mc_fn(FILTER_2D_8TAP_SMOOTH, 8tap_smooth, lsx);
|
||||
init_mc_fn(FILTER_2D_8TAP_SMOOTH_SHARP, 8tap_smooth_sharp, lsx);
|
||||
init_mc_fn(FILTER_2D_8TAP_SHARP_REGULAR, 8tap_sharp_regular, lsx);
|
||||
init_mc_fn(FILTER_2D_8TAP_SHARP_SMOOTH, 8tap_sharp_smooth, lsx);
|
||||
init_mc_fn(FILTER_2D_8TAP_SHARP, 8tap_sharp, lsx);
|
||||
init_8tap_fns(lsx);
|
||||
|
||||
if (!(flags & DAV1D_LOONGARCH_CPU_FLAG_LASX)) return;
|
||||
|
||||
@@ -102,16 +87,9 @@ static ALWAYS_INLINE void mc_dsp_init_loongarch(Dav1dMCDSPContext *const c) {
|
||||
c->warp8x8 = BF(dav1d_warp_affine_8x8, lasx);
|
||||
c->warp8x8t = BF(dav1d_warp_affine_8x8t, lasx);
|
||||
c->w_mask[2] = BF(dav1d_w_mask_420, lasx);
|
||||
c->blend_h = BF(dav1d_blend_h, lasx);
|
||||
|
||||
init_mct_fn(FILTER_2D_8TAP_REGULAR, 8tap_regular, lasx);
|
||||
init_mct_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, lasx);
|
||||
init_mct_fn(FILTER_2D_8TAP_REGULAR_SHARP, 8tap_regular_sharp, lasx);
|
||||
init_mct_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, lasx);
|
||||
init_mct_fn(FILTER_2D_8TAP_SMOOTH, 8tap_smooth, lasx);
|
||||
init_mct_fn(FILTER_2D_8TAP_SMOOTH_SHARP, 8tap_smooth_sharp, lasx);
|
||||
init_mct_fn(FILTER_2D_8TAP_SHARP_REGULAR, 8tap_sharp_regular, lasx);
|
||||
init_mct_fn(FILTER_2D_8TAP_SHARP_SMOOTH, 8tap_sharp_smooth, lasx);
|
||||
init_mct_fn(FILTER_2D_8TAP_SHARP, 8tap_sharp, lasx);
|
||||
init_8tap_gen(mct, lasx);
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
+262
-26
@@ -31,28 +31,29 @@ const min_prob
|
||||
.short 60, 56, 52, 48, 44, 40, 36, 32, 28, 24, 20, 16, 12, 8, 4, 0
|
||||
endconst
|
||||
|
||||
const ph_0xff00
|
||||
.rept 8
|
||||
.short 0xff00
|
||||
.endr
|
||||
endconst
|
||||
|
||||
.macro decode_symbol_adapt w
|
||||
addi.d sp, sp, -48
|
||||
addi.d a4, a0, 24
|
||||
vldrepl.h vr0, a4, 0 //rng
|
||||
vldrepl.h vr0, a0, 24 //rng
|
||||
fst.s f0, sp, 0 //val==0
|
||||
vld vr1, a1, 0 //cdf
|
||||
.if \w == 16
|
||||
li.w t4, 16
|
||||
vldx vr11, a1, t4
|
||||
vld vr11, a1, 16
|
||||
.endif
|
||||
addi.d a6, a0, 16
|
||||
vldrepl.d vr2, a6, 0 //dif
|
||||
addi.d t0, a0, 32
|
||||
ld.w t1, t0, 0 //allow_update_cdf
|
||||
vldrepl.d vr2, a0, 16 //dif
|
||||
ld.w t1, a0, 32 //allow_update_cdf
|
||||
la.local t2, min_prob
|
||||
addi.d t2, t2, 32
|
||||
addi.w t3, a2, 1
|
||||
slli.w t3, t3, 1
|
||||
addi.d t2, t2, 30
|
||||
slli.w t3, a2, 1
|
||||
sub.d t2, t2, t3
|
||||
vld vr3, t2, 0 //min_prob
|
||||
.if \w == 16
|
||||
vldx vr13, t2, t4
|
||||
vld vr13, t2, 16
|
||||
.endif
|
||||
vsrli.h vr4, vr0, 8 //r = s->rng >> 8
|
||||
vslli.h vr4, vr4, 8 //r << 8
|
||||
@@ -68,17 +69,15 @@ endconst
|
||||
vmuh.hu vr15, vr4, vr15
|
||||
vadd.h vr15, vr15, vr13
|
||||
.endif
|
||||
addi.d t8, sp, 4
|
||||
addi.d t8, sp, 2
|
||||
vst vr5, t8, 0 //store v
|
||||
.if \w == 16
|
||||
vstx vr15, t8, t4
|
||||
vst vr15, t8, 16
|
||||
.endif
|
||||
vreplvei.h vr20, vr2, 3 //c
|
||||
vssub.hu vr6, vr5, vr20 //c >=v
|
||||
vseqi.h vr6, vr6, 0
|
||||
vsle.hu vr6, vr5, vr20
|
||||
.if \w == 16
|
||||
vssub.hu vr16, vr15, vr20 //c >=v
|
||||
vseqi.h vr16, vr16, 0
|
||||
vsle.hu vr16, vr15, vr20
|
||||
vpickev.b vr21, vr16, vr6
|
||||
.endif
|
||||
.if \w <= 8
|
||||
@@ -92,10 +91,14 @@ endconst
|
||||
alsl.d t1, a2, a1, 1
|
||||
ld.h t2, t1, 0 //count
|
||||
srli.w t3, t2, 4 //count >> 4
|
||||
.if \w == 16
|
||||
addi.w t3, t3, 5 //rate
|
||||
.else
|
||||
addi.w t3, t3, 4
|
||||
li.w t5, 2
|
||||
sltu t5, t5, a2
|
||||
add.w t3, t3, t5 //rate
|
||||
.endif
|
||||
sltui t5, t2, 32
|
||||
add.w t2, t2, t5 //count + (count < 32)
|
||||
vreplgr2vr.h vr9, t3
|
||||
@@ -118,7 +121,7 @@ endconst
|
||||
.if \w == 16
|
||||
vsra.h vr15, vr15, vr9
|
||||
vadd.h vr18, vr18, vr15
|
||||
vstx vr18, a1, t4
|
||||
vst vr18, a1, 16
|
||||
.endif
|
||||
st.h t2, t1, 0
|
||||
|
||||
@@ -127,8 +130,7 @@ endconst
|
||||
ctz.w a7, t3 // ret
|
||||
alsl.d t3, a7, t8, 1
|
||||
ld.hu t4, t3, 0 // v
|
||||
addi.d t3, t3, -2
|
||||
ld.hu t5, t3, 0 // u
|
||||
ld.hu t5, t3, -2 // u
|
||||
sub.w t5, t5, t4 // rng
|
||||
slli.d t4, t4, 48
|
||||
vpickve2gr.d t6, vr2, 0
|
||||
@@ -136,11 +138,10 @@ endconst
|
||||
clz.w t4, t5 // d
|
||||
xori t4, t4, 16 // d
|
||||
sll.d t6, t6, t4
|
||||
addi.d a5, a0, 28 // cnt
|
||||
ld.w t0, a5, 0
|
||||
ld.w t0, a0, 28 //cnt
|
||||
sll.w t5, t5, t4
|
||||
sub.w t7, t0, t4 // cnt-d
|
||||
st.w t5, a4, 0 // store rng
|
||||
st.w t5, a0, 24 // store rng
|
||||
bgeu t0, t4, 9f
|
||||
|
||||
// refill
|
||||
@@ -186,8 +187,8 @@ endconst
|
||||
4:
|
||||
or t6, t6, t3 // dif |= next_bits
|
||||
9:
|
||||
st.w t7, a5, 0 // store cnt
|
||||
st.d t6, a6, 0 // store dif
|
||||
st.w t7, a0, 28 // store cnt
|
||||
st.d t6, a0, 16 // store dif
|
||||
move a0, a7
|
||||
addi.d sp, sp, 48
|
||||
.endm
|
||||
@@ -281,6 +282,82 @@ function msac_decode_bool_lsx
|
||||
move a0, t8
|
||||
endfunc
|
||||
|
||||
function msac_decode_bool_equi_lsx
|
||||
ld.w t0, a0, 24 // rng
|
||||
ld.d t1, a0, 16 // dif
|
||||
ld.w a5, a0, 28 // cnt
|
||||
srli.w t2, t0, 8 // r >> 8
|
||||
slli.w t2, t2, 7
|
||||
addi.w t2, t2, 4 // v
|
||||
|
||||
slli.d t3, t2, 48 // vw
|
||||
sltu t4, t1, t3
|
||||
move t8, t4 // ret
|
||||
xori t4, t4, 1
|
||||
maskeqz t6, t3, t4 // if (ret) vw
|
||||
sub.d t6, t1, t6 // dif
|
||||
slli.w t5, t2, 1
|
||||
sub.w t5, t0, t5 // r - 2v
|
||||
maskeqz t7, t5, t4 // if (ret) r - 2v
|
||||
add.w t5, t2, t7 // v(rng)
|
||||
|
||||
// renorm
|
||||
clz.w t4, t5 // d
|
||||
xori t4, t4, 16 // d
|
||||
sll.d t6, t6, t4
|
||||
sll.w t5, t5, t4
|
||||
sub.w t7, a5, t4 // cnt-d
|
||||
st.w t5, a0, 24 // store rng
|
||||
bgeu a5, t4, 9f
|
||||
|
||||
// refill
|
||||
ld.d t0, a0, 0 // buf_pos
|
||||
ld.d t1, a0, 8 // buf_end
|
||||
addi.d t2, t0, 8
|
||||
bltu t1, t2, 2f
|
||||
|
||||
ld.d t3, t0, 0 // next_bits
|
||||
addi.w t1, t7, -48 // shift_bits = cnt + 16 (- 64)
|
||||
nor t3, t3, t3
|
||||
sub.w t2, zero, t1
|
||||
revb.d t3, t3 // next_bits = bswap(next_bits)
|
||||
srli.w t2, t2, 3 // num_bytes_read
|
||||
srl.d t3, t3, t1 // next_bits >>= (shift_bits & 63)
|
||||
b 3f
|
||||
1:
|
||||
addi.w t3, t7, -48
|
||||
srl.d t3, t3, t3 // pad with ones
|
||||
b 4f
|
||||
2:
|
||||
bgeu t0, t1, 1b
|
||||
ld.d t3, t1, -8 // next_bits
|
||||
sub.w t2, t2, t1
|
||||
sub.w t1, t1, t0 // num_bytes_left
|
||||
slli.w t2, t2, 3
|
||||
srl.d t3, t3, t2
|
||||
addi.w t2, t7, -48
|
||||
nor t3, t3, t3
|
||||
sub.w t4, zero, t2
|
||||
revb.d t3, t3
|
||||
srli.w t4, t4, 3
|
||||
srl.d t3, t3, t2
|
||||
sltu t2, t1, t4
|
||||
maskeqz t1, t1, t2
|
||||
masknez t2, t4, t2
|
||||
or t2, t2, t1 // num_bytes_read
|
||||
3:
|
||||
slli.w t1, t2, 3
|
||||
add.d t0, t0, t2
|
||||
add.w t7, t7, t1 // cnt += num_bits_read
|
||||
st.d t0, a0, 0
|
||||
4:
|
||||
or t6, t6, t3 // dif |= next_bits
|
||||
9:
|
||||
st.w t7, a0, 28 // store cnt
|
||||
st.d t6, a0, 16 // store dif
|
||||
move a0, t8
|
||||
endfunc
|
||||
|
||||
function msac_decode_bool_adapt_lsx
|
||||
ld.hu a3, a1, 0 // cdf[0] /f
|
||||
ld.w t0, a0, 24 // rng
|
||||
@@ -374,3 +451,162 @@ function msac_decode_bool_adapt_lsx
|
||||
st.d t6, a0, 16 // store dif
|
||||
move a0, t8
|
||||
endfunc
|
||||
|
||||
.macro HI_TOK allow_update_cdf
|
||||
.\allow_update_cdf\()_hi_tok_lsx_start:
|
||||
.if \allow_update_cdf == 1
|
||||
ld.hu a4, a1, 0x06 // cdf[3]
|
||||
.endif
|
||||
vor.v vr1, vr0, vr0
|
||||
vsrli.h vr1, vr1, 0x06 // cdf[val] >> EC_PROB_SHIFT
|
||||
vstelm.h vr2, sp, 0, 0 // -0x1a
|
||||
vand.v vr2, vr2, vr4 // (8 x rng) & 0xff00
|
||||
vslli.h vr1, vr1, 0x07
|
||||
vmuh.hu vr1, vr1, vr2
|
||||
vadd.h vr1, vr1, vr5 // v += EC_MIN_PROB/* 4 */ * ((unsigned)n_symbols/* 3 */ - val);
|
||||
vst vr1, sp, 0x02 // -0x18
|
||||
vssub.hu vr1, vr1, vr3 // v - c
|
||||
vseqi.h vr1, vr1, 0
|
||||
.if \allow_update_cdf == 1
|
||||
addi.d t4, a4, 0x50
|
||||
srli.d t4, t4, 0x04
|
||||
sltui t7, a4, 32
|
||||
add.w a4, a4, t7
|
||||
|
||||
vreplgr2vr.h vr7, t4
|
||||
vavgr.hu vr9, vr8, vr1
|
||||
vsub.h vr9, vr9, vr0
|
||||
vsub.h vr0, vr0, vr1
|
||||
vsra.h vr9, vr9, vr7
|
||||
vadd.h vr0, vr0, vr9
|
||||
vstelm.d vr0, a1, 0, 0
|
||||
st.h a4, a1, 0x06
|
||||
.endif
|
||||
vmsknz.b vr7, vr1
|
||||
movfr2gr.s t4, f7
|
||||
ctz.w t4, t4 // loop_times * 2
|
||||
addi.d t7, t4, 2
|
||||
ldx.hu t6, sp, t4 // u
|
||||
ldx.hu t5, sp, t7 // v
|
||||
addi.w t3, t3, 0x05
|
||||
addi.w t4, t4, -0x05 // if t4 == 3, continue
|
||||
sub.w t6, t6, t5 // u - v , rng for ctx_norm
|
||||
slli.d t5, t5, 0x30 // (ec_win)v << (EC_WIN_SIZE - 16)
|
||||
sub.d t1, t1, t5 // s->dif - ((ec_win)v << (EC_WIN_SIZE - 16))
|
||||
// Init ctx_norm param
|
||||
clz.w t7, t6
|
||||
xori t7, t7, 0x1f
|
||||
xori t7, t7, 0x0f // d = 15 ^ (31 ^ clz(rng));
|
||||
sll.d t1, t1, t7 // dif << d
|
||||
sll.d t6, t6, t7 // rng << d
|
||||
// update vr2 8 x rng
|
||||
vreplgr2vr.h vr2, t6
|
||||
vreplvei.h vr2, vr2, 0
|
||||
st.w t6, a0, 0x18 // store rng
|
||||
move t0, t2
|
||||
sub.w t2, t2, t7 // cnt - d
|
||||
bgeu t0, t7, .\allow_update_cdf\()_hi_tok_lsx_ctx_norm_end // if ((unsigned)cnt < (unsigned)d) goto ctx_norm_end
|
||||
// Step into ctx_fill
|
||||
ld.d t5, a0, 0x00 // buf_pos
|
||||
ld.d t6, a0, 0x08 // end_pos
|
||||
addi.d t7, t5, 0x08 // buf_pos + 8
|
||||
sub.d t7, t7, t6 // (buf_pos + 8) - end_pos
|
||||
blt zero, t7, .\allow_update_cdf\()_hi_tok_lsx_ctx_refill_eob
|
||||
// (end_pos - buf_pos) >= 8
|
||||
ld.d t6, t5, 0x00 // load buf_pos[0]~buf_pos[7]
|
||||
addi.w t7, t2, -0x30 // cnt - 0x30
|
||||
nor t6, t6, t6 // not buf data
|
||||
revb.d t6, t6 // Byte reversal
|
||||
srl.d t6, t6, t7 // Replace left shift with right shift
|
||||
sub.w t7, zero, t7 // neg
|
||||
srli.w t7, t7, 0x03 // Loop times
|
||||
or t1, t1, t6 // dif |= (ec_win)(*buf_pos++ ^ 0xff) << c
|
||||
b .\allow_update_cdf\()_hi_tok_lsx_ctx_refill_end
|
||||
.\allow_update_cdf\()_hi_tok_lsx_ctx_refill_eob:
|
||||
bge t5, t6, .\allow_update_cdf\()_hi_tok_lsx_ctx_refill_one
|
||||
// end_pos - buf_pos < 8 && buf_pos < end_pos
|
||||
ld.d t0, t6, -0x08
|
||||
slli.d t7, t7, 0x03
|
||||
srl.d t6, t0, t7 // Retrieve the buf data and remove the excess data
|
||||
addi.w t7, t2, -0x30 // cnt - 0x30
|
||||
nor t6, t6, t6 // not
|
||||
revb.d t6, t6 // Byte reversal
|
||||
srl.d t6, t6, t7 // Replace left shift with right shift
|
||||
sub.w t7, zero, t7 // neg
|
||||
or t1, t1, t6 // dif |= (ec_win)(*buf_pos++ ^ 0xff) << c
|
||||
ld.d t6, a0, 0x08 // end_pos
|
||||
srli.w t7, t7, 0x03 // Loop times
|
||||
sub.d t6, t6, t5 // end_pos - buf_pos
|
||||
slt t0, t6, t7
|
||||
maskeqz a3, t6, t0 // min(loop_times, end_pos - buf_pos)
|
||||
masknez t0, t7, t0
|
||||
or t7, a3, t0
|
||||
b .\allow_update_cdf\()_hi_tok_lsx_ctx_refill_end
|
||||
.\allow_update_cdf\()_hi_tok_lsx_ctx_refill_one:
|
||||
// buf_pos >= end_pos
|
||||
addi.w t7, t2, -0x10
|
||||
andi t7, t7, 0xf
|
||||
nor t0, zero, zero
|
||||
srl.d t0, t0, t7
|
||||
or t1, t1, t0 // dif |= ~(~(ec_win)0xff << c);
|
||||
b .\allow_update_cdf\()_hi_tok_lsx_ctx_norm_end
|
||||
.\allow_update_cdf\()_hi_tok_lsx_ctx_refill_end:
|
||||
add.d t5, t5, t7 // buf_pos + Loop_times
|
||||
st.d t5, a0, 0x00 // Store buf_pos
|
||||
alsl.w t2, t7, t2, 0x03 // update cnt
|
||||
.\allow_update_cdf\()_hi_tok_lsx_ctx_norm_end:
|
||||
srli.d t7, t1, 0x30
|
||||
vreplgr2vr.h vr3, t7 // broadcast the high 16 bits of dif
|
||||
add.w t3, t4, t3 // update control parameter
|
||||
beqz t3, .\allow_update_cdf\()_hi_tok_lsx_end // control loop for at most 4 times.
|
||||
blt zero, t4, .\allow_update_cdf\()_hi_tok_lsx_start // tok_br == 3
|
||||
.\allow_update_cdf\()_hi_tok_lsx_end:
|
||||
addi.d t3, t3, 0x1e
|
||||
st.d t1, a0, 0x10 // store dif
|
||||
st.w t2, a0, 0x1c // store cnt
|
||||
srli.w a0, t3, 0x01 // tok
|
||||
addi.d sp, sp, 0x1a
|
||||
.endm
|
||||
|
||||
/**
|
||||
* @param unsigned dav1d_msac_decode_hi_tok_c(MsacContext *const s, uint16_t *const cdf)
|
||||
* * Reg Alloction
|
||||
* * vr0: cdf;
|
||||
* * vr1: temp;
|
||||
* * vr2: rng;
|
||||
* * vr3: dif;
|
||||
* * vr4: const 0xff00ff00...ff00ff00;
|
||||
* * vr5: const 0x0004080c;
|
||||
* * vr6: const 0;
|
||||
* * t0: allow_update_cdf, tmp;
|
||||
* * t1: dif;
|
||||
* * t2: cnt;
|
||||
* * t3: 0xffffffe8, outermost control parameter;
|
||||
* * t4: loop time
|
||||
* * t5: v, buf_pos, temp;
|
||||
* * t6: u, rng, end_pos, buf, temp;
|
||||
* * t7: temp;
|
||||
*/
|
||||
function msac_decode_hi_tok_lsx
|
||||
fld.d f0, a1, 0 // Load cdf[0]~cdf[3]
|
||||
vldrepl.h vr2, a0, 0x18 // 8 x rng, assert(rng <= 65535U), only the lower 16 bits are valid
|
||||
vldrepl.h vr3, a0, 0x16 // broadcast the high 16 bits of dif, c = s->dif >> (EC_WIN_SIZE - 16)
|
||||
ld.w t0, a0, 0x20 // allow_update_cdf
|
||||
la.local t7, ph_0xff00
|
||||
vld vr4, t7, 0x00 // 0xff00ff00...ff00ff00
|
||||
la.local t7, min_prob
|
||||
vld vr5, t7, 12 * 2 // 0x0004080c
|
||||
vxor.v vr6, vr6, vr6 // const 0
|
||||
ld.d t1, a0, 0x10 // dif
|
||||
ld.w t2, a0, 0x1c // cnt
|
||||
orn t3, t3, t3
|
||||
srli.d t3, t3, 32
|
||||
addi.d t3, t3, -0x17 // 0xffffffe8
|
||||
vseq.h vr8, vr8, vr8
|
||||
addi.d sp, sp, -0x1a // alloc stack
|
||||
beqz t0, .hi_tok_lsx_no_update_cdf
|
||||
HI_TOK 1
|
||||
jirl zero, ra, 0x0
|
||||
.hi_tok_lsx_no_update_cdf:
|
||||
HI_TOK 0
|
||||
endfunc
|
||||
|
||||
@@ -36,11 +36,15 @@ unsigned dav1d_msac_decode_symbol_adapt16_lsx(MsacContext *s, uint16_t *cdf,
|
||||
size_t n_symbols);
|
||||
unsigned dav1d_msac_decode_bool_adapt_lsx(MsacContext *s, uint16_t *cdf);
|
||||
unsigned dav1d_msac_decode_bool_lsx(MsacContext *s, unsigned f);
|
||||
unsigned dav1d_msac_decode_bool_equi_lsx(MsacContext *s);
|
||||
unsigned dav1d_msac_decode_hi_tok_lsx(MsacContext *s, uint16_t *cdf);
|
||||
|
||||
#define dav1d_msac_decode_symbol_adapt4 dav1d_msac_decode_symbol_adapt4_lsx
|
||||
#define dav1d_msac_decode_symbol_adapt8 dav1d_msac_decode_symbol_adapt8_lsx
|
||||
#define dav1d_msac_decode_symbol_adapt16 dav1d_msac_decode_symbol_adapt16_lsx
|
||||
#define dav1d_msac_decode_bool_adapt dav1d_msac_decode_bool_adapt_lsx
|
||||
#define dav1d_msac_decode_bool dav1d_msac_decode_bool_lsx
|
||||
#define dav1d_msac_decode_bool_equi dav1d_msac_decode_bool_equi_lsx
|
||||
#define dav1d_msac_decode_hi_tok dav1d_msac_decode_hi_tok_lsx
|
||||
|
||||
#endif /* DAV1D_SRC_LOONGARCH_MSAC_H */
|
||||
|
||||
@@ -150,3 +150,552 @@ function splat_mv_lsx
|
||||
|
||||
.splat_end:
|
||||
endfunc
|
||||
|
||||
const la_div_mult
|
||||
.short 0, 16384, 8192, 5461, 4096, 3276, 2730, 2340
|
||||
.short 2048, 1820, 1638, 1489, 1365, 1260, 1170, 1092
|
||||
.short 1024, 963, 910, 862, 819, 780, 744, 712
|
||||
.short 682, 655, 630, 606, 585, 564, 546, 528
|
||||
endconst
|
||||
|
||||
/*
|
||||
* temp reg: a6 a7
|
||||
*/
|
||||
.macro LOAD_SET_LOOP is_odd
|
||||
slli.d a6, t6, 2
|
||||
add.d a6, a6, t6 // col_w * 5
|
||||
0:
|
||||
addi.d a7, zero, 0 // x
|
||||
.if \is_odd
|
||||
stx.w t7, t3, a7
|
||||
addi.d a7, a7, 5
|
||||
bge a7, a6, 2f
|
||||
.endif
|
||||
|
||||
1:
|
||||
stx.w t7, t3, a7
|
||||
addi.d a7, a7, 5
|
||||
stx.w t7, t3, a7
|
||||
addi.d a7, a7, 5
|
||||
blt a7, a6, 1b
|
||||
2:
|
||||
add.d t3, t3, t2
|
||||
addi.d t5, t5, 1
|
||||
blt t5, a5, 0b
|
||||
.endm
|
||||
|
||||
/*
|
||||
* static void load_tmvs_c(const refmvs_frame *const rf, int tile_row_idx,
|
||||
* const int col_start8, const int col_end8,
|
||||
* const int row_start8, int row_end8)
|
||||
*/
|
||||
function load_tmvs_lsx
|
||||
addi.d sp, sp, -80
|
||||
st.d s0, sp, 0
|
||||
st.d s1, sp, 8
|
||||
st.d s2, sp, 16
|
||||
st.d s3, sp, 24
|
||||
st.d s4, sp, 32
|
||||
st.d s5, sp, 40
|
||||
st.d s6, sp, 48
|
||||
st.d s7, sp, 56
|
||||
st.d s8, sp, 64
|
||||
|
||||
vld vr16, a0, 16
|
||||
vld vr0, a0, 48 // rf->mfmv_ref, rf->mfmv_ref2cur
|
||||
ld.w s8, a0, 80 // [0] - rf->n_mfmvs
|
||||
vld vr17, a0, 96 // [0] - rp_ref| [1]- rp_proj
|
||||
ld.d t1, a0, 112 // stride
|
||||
ld.w t0, a0, 128
|
||||
addi.w t0, t0, -1
|
||||
bnez t0, 1f
|
||||
addi.w a1, zero, 0
|
||||
1:
|
||||
addi.d t0, a3, 8
|
||||
vinsgr2vr.w vr1, t0, 0
|
||||
vinsgr2vr.w vr1, a5, 1
|
||||
vmin.w vr1, vr1, vr16 // [0] col_end8i [1] row_end8
|
||||
addi.d t0, a2, -8
|
||||
bge t0, zero, 2f
|
||||
addi.w t0, zero, 0 // t0 col_start8i
|
||||
2:
|
||||
vpickve2gr.d t4, vr17, 1 // rf->rp_proj
|
||||
slli.d t2, t1, 2
|
||||
add.d t2, t2, t1 // stride * 5
|
||||
slli.d a1, a1, 4 // tile_row_idx * 16
|
||||
andi t3, a4, 0xf
|
||||
add.d t3, t3, a1 // tile_row_idx * 16 + row_start8 & 15
|
||||
mul.w t3, t3, t2
|
||||
mul.w t8, a1, t2
|
||||
vpickve2gr.w a5, vr1, 1
|
||||
addi.d t5, a4, 0
|
||||
sub.d t6, a3, a2 // col_end8 - col_start8
|
||||
li.w t7, 0x80008000
|
||||
slli.d a7, a2, 2
|
||||
add.d t3, t3, a2
|
||||
add.d t3, t3, a7
|
||||
add.d t3, t3, t4 // rp_proj
|
||||
andi a6, t6, 1
|
||||
bnez a6, 3f
|
||||
LOAD_SET_LOOP 0
|
||||
b 4f
|
||||
3:
|
||||
LOAD_SET_LOOP 1
|
||||
4:
|
||||
addi.d a6, zero, 0 // n
|
||||
bge a6, s8, .end_load
|
||||
add.d t3, t8, t4 // rp_proj
|
||||
mul.w t6, a4, t2
|
||||
addi.d s7, zero, 40
|
||||
vpickve2gr.w t1, vr1, 0 // col_end8i
|
||||
addi.d t5, a0, 58 // rf->mfmv_ref2ref - 1
|
||||
la.local t8, la_div_mult
|
||||
vld vr6, t8, 0
|
||||
vld vr7, t8, 16
|
||||
vld vr8, t8, 32
|
||||
vld vr9, t8, 48
|
||||
li.w t8, 0x3fff
|
||||
vreplgr2vr.h vr21, t8
|
||||
vxor.v vr18, vr18, vr18 // zero
|
||||
vsub.h vr20, vr18, vr21
|
||||
vpickev.b vr12, vr7, vr6
|
||||
vpickod.b vr13, vr7, vr6
|
||||
vpickev.b vr14, vr9, vr8
|
||||
vpickod.b vr15, vr9, vr8
|
||||
vpickve2gr.d s6, vr17, 0 // rf->rp_ref
|
||||
5:
|
||||
vld vr10, t5, 0 // ref2ref [1...7]
|
||||
vpickve2gr.b t8, vr0, 8 // ref2cur
|
||||
vbsrl.v vr0, vr0, 1
|
||||
addi.w t4, t8, 32
|
||||
beqz t4, 8f // INVALID_REF2CUR
|
||||
|
||||
vreplgr2vr.h vr23, t8
|
||||
vshuf.b vr6, vr14, vr12, vr10
|
||||
vshuf.b vr7, vr15, vr13, vr10
|
||||
vilvl.b vr8, vr7, vr6
|
||||
vmulwev.w.h vr6, vr8, vr23
|
||||
vmulwod.w.h vr7, vr8, vr23
|
||||
|
||||
vpickve2gr.b s0, vr0, 4 // ref
|
||||
slli.d t8, s0, 3
|
||||
ldx.d s1, s6, t8 // rf->rp_ref[ref]
|
||||
addi.d s0, s0, -4 // ref_sign
|
||||
vreplgr2vr.h vr19, s0
|
||||
add.d s1, s1, t6 // &rf->rp_ref[ref][row_start8 * stride]
|
||||
addi.d s2, a4, 0 // y
|
||||
vilvl.w vr8, vr7, vr6
|
||||
vilvh.w vr9, vr7, vr6
|
||||
6: // for (int y = row_start8;
|
||||
andi s3, s2, 0xff8
|
||||
|
||||
addi.d s4, s3, 8
|
||||
blt a4, s3, 0f
|
||||
addi.d s3, a4, 0 // y_proj_start
|
||||
0:
|
||||
blt s4, a5, 0f
|
||||
addi.d s4, a5, 0 // y_proj_end
|
||||
0:
|
||||
addi.d s5, t0, 0 // x
|
||||
7: // for (int x = col_start8i;
|
||||
slli.d a7, s5, 2
|
||||
add.d a7, a7, s5
|
||||
add.d a7, s1, a7 // rb
|
||||
vld vr3, a7, 0 // [rb]
|
||||
vpickve2gr.b t4, vr3, 4 // b_ref
|
||||
beqz t4, .end_x
|
||||
vreplve.b vr11, vr10, t4
|
||||
vpickve2gr.b t7, vr11, 4 // ref2ref
|
||||
beqz t7, .end_x
|
||||
vsllwil.w.h vr4, vr3, 0
|
||||
vreplgr2vr.w vr6, t4
|
||||
vshuf.w vr6, vr9, vr8 // frac
|
||||
vmul.w vr5, vr6, vr4
|
||||
vsrai.w vr4, vr5, 31
|
||||
vadd.w vr4, vr4, vr5
|
||||
vssrarni.h.w vr4, vr4, 14
|
||||
vclip.h vr4, vr4, vr20, vr21 // offset
|
||||
vxor.v vr5, vr4, vr19 // offset.x ^ ref_sign
|
||||
vori.b vr5, vr5, 0x1 // offset.x ^ ref_sign
|
||||
vabsd.h vr4, vr4, vr18
|
||||
vsrli.h vr4, vr4, 6 // abs(offset.x) >> 6
|
||||
vsigncov.h vr4, vr5, vr4 // apply_sign
|
||||
vpickve2gr.h s0, vr4, 0
|
||||
add.d s0, s2, s0 // pos_y
|
||||
blt s0, s3, .n_posy
|
||||
bge s0, s4, .n_posy
|
||||
andi s0, s0, 0xf
|
||||
mul.w s0, s0, t2 // pos
|
||||
vpickve2gr.h t7, vr4, 1
|
||||
add.d t7, t7, s5 // pos_x
|
||||
add.d s0, t3, s0 // rp_proj + pos
|
||||
|
||||
.loop_posx:
|
||||
andi t4, s5, 0xff8 // x_sb_align
|
||||
|
||||
blt t7, a2, .n_posx
|
||||
addi.d t8, t4, -8
|
||||
blt t7, t8, .n_posx
|
||||
|
||||
bge t7, a3, .n_posx
|
||||
addi.d t4, t4, 16
|
||||
bge t7, t4, .n_posx
|
||||
|
||||
slli.d t4, t7, 2
|
||||
add.d t4, t4, t7 // pos_x * 5
|
||||
add.d t4, s0, t4 // rp_proj[pos + pos_x]
|
||||
vstelm.w vr3, t4, 0, 0
|
||||
vstelm.b vr11, t4, 4, 4
|
||||
|
||||
.n_posx:
|
||||
addi.d s5, s5, 1 // x + 1
|
||||
bge s5, t1, .ret_posx
|
||||
addi.d a7, a7, 5 // rb + 1
|
||||
vld vr4, a7, 0 // [rb]
|
||||
vseq.b vr5, vr4, vr3
|
||||
|
||||
vpickve2gr.d t8, vr5, 0
|
||||
cto.d t8, t8
|
||||
blt t8, s7, 7b
|
||||
|
||||
addi.d t7, t7, 1 // pos_x + 1
|
||||
|
||||
/* Core computing loop expansion(sencond) */
|
||||
andi t4, s5, 0xff8 // x_sb_align
|
||||
|
||||
blt t7, a2, .n_posx
|
||||
addi.d t8, t4, -8
|
||||
blt t7, t8, .n_posx
|
||||
|
||||
bge t7, a3, .n_posx
|
||||
addi.d t4, t4, 16
|
||||
bge t7, t4, .n_posx
|
||||
|
||||
slli.d t4, t7, 2
|
||||
add.d t4, t4, t7 // pos_x * 5
|
||||
add.d t4, s0, t4 // rp_proj[pos + pos_x]
|
||||
vstelm.w vr3, t4, 0, 0
|
||||
vstelm.b vr11, t4, 4, 4
|
||||
|
||||
addi.d s5, s5, 1 // x + 1
|
||||
bge s5, t1, .ret_posx
|
||||
addi.d a7, a7, 5 // rb + 1
|
||||
vld vr4, a7, 0 // [rb]
|
||||
vseq.b vr5, vr4, vr3
|
||||
|
||||
vpickve2gr.d t8, vr5, 0
|
||||
cto.d t8, t8
|
||||
blt t8, s7, 7b
|
||||
|
||||
addi.d t7, t7, 1 // pos_x + 1
|
||||
|
||||
/* Core computing loop expansion(third) */
|
||||
andi t4, s5, 0xff8 // x_sb_align
|
||||
|
||||
blt t7, a2, .n_posx
|
||||
addi.d t8, t4, -8
|
||||
blt t7, t8, .n_posx
|
||||
|
||||
bge t7, a3, .n_posx
|
||||
addi.d t4, t4, 16
|
||||
bge t7, t4, .n_posx
|
||||
|
||||
slli.d t4, t7, 2
|
||||
add.d t4, t4, t7 // pos_x * 5
|
||||
add.d t4, s0, t4 // rp_proj[pos + pos_x]
|
||||
vstelm.w vr3, t4, 0, 0
|
||||
vstelm.b vr11, t4, 4, 4
|
||||
|
||||
addi.d s5, s5, 1 // x + 1
|
||||
bge s5, t1, .ret_posx
|
||||
addi.d a7, a7, 5 // rb + 1
|
||||
vld vr4, a7, 0 // [rb]
|
||||
vseq.b vr5, vr4, vr3
|
||||
|
||||
vpickve2gr.d t8, vr5, 0
|
||||
cto.d t8, t8
|
||||
blt t8, s7, 7b
|
||||
|
||||
addi.d t7, t7, 1 // pos_x + 1
|
||||
|
||||
b .loop_posx
|
||||
|
||||
.n_posy:
|
||||
addi.d s5, s5, 1 // x + 1
|
||||
bge s5, t1, .ret_posx
|
||||
addi.d a7, a7, 5 // rb + 1
|
||||
vld vr4, a7, 0 // [rb]
|
||||
vseq.b vr5, vr4, vr3
|
||||
|
||||
vpickve2gr.d t8, vr5, 0
|
||||
cto.d t8, t8
|
||||
blt t8, s7, 7b
|
||||
|
||||
addi.d s5, s5, 1 // x + 1
|
||||
bge s5, t1, .ret_posx
|
||||
addi.d a7, a7, 5 // rb + 1
|
||||
vld vr4, a7, 0 // [rb]
|
||||
vseq.b vr5, vr4, vr3
|
||||
|
||||
vpickve2gr.d t8, vr5, 0
|
||||
cto.d t8, t8
|
||||
blt t8, s7, 7b
|
||||
|
||||
b .n_posy
|
||||
|
||||
.end_x:
|
||||
addi.d s5, s5, 1 // x + 1
|
||||
blt s5, t1, 7b
|
||||
|
||||
.ret_posx:
|
||||
add.d s1, s1, t2 // r + stride
|
||||
addi.d s2, s2, 1 // y + 1
|
||||
blt s2, a5, 6b
|
||||
8:
|
||||
addi.d a6, a6, 1 // n + 1
|
||||
addi.d t5, t5, 7 // mfmv_ref2ref(offset) + 7
|
||||
blt a6, s8, 5b
|
||||
|
||||
.end_load:
|
||||
ld.d s0, sp, 0
|
||||
ld.d s1, sp, 8
|
||||
ld.d s2, sp, 16
|
||||
ld.d s3, sp, 24
|
||||
ld.d s4, sp, 32
|
||||
ld.d s5, sp, 40
|
||||
ld.d s6, sp, 48
|
||||
ld.d s7, sp, 56
|
||||
ld.d s8, sp, 64
|
||||
addi.d sp, sp, 80
|
||||
endfunc
|
||||
|
||||
const mv_tbls
|
||||
.byte 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255
|
||||
.byte 0, 1, 2, 3, 8, 0, 1, 2, 3, 8, 0, 1, 2, 3, 8, 0
|
||||
.byte 4, 5, 6, 7, 9, 4, 5, 6, 7, 9, 4, 5, 6, 7, 9, 4
|
||||
.byte 4, 5, 6, 7, 9, 4, 5, 6, 7, 9, 4, 5, 6, 7, 9, 4
|
||||
endconst
|
||||
|
||||
const mask_mult
|
||||
.byte 1, 0, 2, 0, 1, 0, 2, 0, 0, 0, 0, 0, 0, 0, 0, 0
|
||||
endconst
|
||||
|
||||
const mask_mv0
|
||||
.byte 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16
|
||||
endconst
|
||||
|
||||
const mask_mv1
|
||||
.byte 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19
|
||||
endconst
|
||||
|
||||
// void dav1d_save_tmvs_lsx(refmvs_temporal_block *rp, ptrdiff_t stride,
|
||||
// refmvs_block **rr, const uint8_t *ref_sign,
|
||||
// int col_end8, int row_end8,
|
||||
// int col_start8, int row_start8)
|
||||
function save_tmvs_lsx
|
||||
addi.d sp, sp, -0x28
|
||||
st.d s0, sp, 0x00
|
||||
st.d s1, sp, 0x08
|
||||
st.d s2, sp, 0x10
|
||||
st.d s3, sp, 0x18
|
||||
st.d s4, sp, 0x20
|
||||
move t0, ra
|
||||
|
||||
vxor.v vr10, vr10, vr10
|
||||
vld vr11, a3, 0 // Load ref_sign[0] ~ Load ref_sign[7]
|
||||
la.local t2, .save_tevs_tbl
|
||||
la.local s1, mask_mult
|
||||
la.local t7, mv_tbls
|
||||
vld vr9, s1, 0 // Load mask_mult
|
||||
vslli.d vr11, vr11, 8 // 0, ref_sign[0], ... ,ref_sign[6]
|
||||
la.local s3, mask_mv0
|
||||
vld vr8, s3, 0 // Load mask_mv0
|
||||
la.local s4, mask_mv1
|
||||
vld vr7, s4, 0 // Load mask_mv1
|
||||
li.d s0, 5
|
||||
li.d t8, 12 * 2
|
||||
mul.d a1, a1, s0 // stride *= 5
|
||||
sub.d a5, a5, a7 // h = row_end8 - row_start8
|
||||
slli.d a7, a7, 1 // row_start8 <<= 1
|
||||
1:
|
||||
li.d s0, 5
|
||||
andi t3, a7, 30 // (y & 15) * 2
|
||||
slli.d s4, t3, 3
|
||||
ldx.d t3, a2, s4 // b = rr[(y & 15) * 2]
|
||||
addi.d t3, t3, 12 // &b[... + 1]
|
||||
mul.d s4, a4, t8
|
||||
add.d t4, s4, t3 // end_cand_b = &b[col_end8*2 + 1]
|
||||
mul.d s3, a6, t8
|
||||
add.d t3, s3, t3 // cand_b = &b[x*2 + 1]
|
||||
mul.d s4, a6, s0
|
||||
add.d a3, s4, a0 // &rp[x]
|
||||
2:
|
||||
/* First cand_b */
|
||||
ld.b t5, t3, 10 // cand_b->bs
|
||||
vld vr0, t3, 0 // cand_b->mv and ref
|
||||
alsl.d t5, t5, t2, 2 // bt2 index
|
||||
ld.h s3, t3, 8 // cand_b->ref
|
||||
ld.h t6, t5, 0 // bt2
|
||||
move s0, t2
|
||||
alsl.d t3, t6, t3, 1 // Next cand_b += bt2 * 2
|
||||
vor.v vr2, vr0, vr0
|
||||
vinsgr2vr.h vr1, s3, 0
|
||||
move t1 , t3
|
||||
bge t3, t4, 3f
|
||||
|
||||
/* Next cand_b */
|
||||
ld.b s0, t3, 10 // cand_b->bs
|
||||
vld vr4, t3, 0 // cand_b->mv and ref
|
||||
alsl.d s0, s0, t2, 2 // bt2 index
|
||||
ld.h s4, t3, 8 // cand_b->ref
|
||||
ld.h t6, s0, 0 // bt2
|
||||
alsl.d t3, t6, t3, 1 // Next cand_b += bt2*2
|
||||
vpackev.d vr2, vr4, vr0 // a0.mv[0] a0.mv[1] a1.mv[0], a1.mv[1]
|
||||
vinsgr2vr.h vr1, s4, 1 // a0.ref[0] a0.ref[1], a1.ref[0], a1.ref[1]
|
||||
3:
|
||||
vabsd.h vr2, vr2, vr10 // abs(mv[].xy)
|
||||
vsle.b vr16, vr10, vr1
|
||||
vand.v vr1, vr16, vr1
|
||||
vshuf.b vr1, vr11, vr11, vr1 // ref_sign[ref]
|
||||
vsrli.h vr2, vr2, 12 // abs(mv[].xy) >> 12
|
||||
vilvl.b vr1, vr1, vr1
|
||||
vmulwev.h.bu vr1, vr1, vr9 // ef_sign[ref] * {1, 2}
|
||||
|
||||
vseqi.w vr2, vr2, 0 // abs(mv[].xy) <= 4096
|
||||
vpickev.h vr2, vr2, vr2 // abs() condition to 16 bit
|
||||
|
||||
vand.v vr1, vr2, vr1 // h[0-3] contains conditions for mv[0-1]
|
||||
vhaddw.wu.hu vr1, vr1, vr1 // Combine condition for [1] and [0]
|
||||
vpickve2gr.wu s1, vr1, 0 // Extract case for first block
|
||||
vpickve2gr.wu s2, vr1, 1
|
||||
|
||||
ld.hu t5, t5, 2 // Fetch jump table entry
|
||||
ld.hu s0, s0, 2
|
||||
alsl.d s3, s1, t7, 4 // Load permutation table base on case
|
||||
vld vr1, s3, 0
|
||||
alsl.d s4, s2, t7, 4
|
||||
vld vr5, s4, 0
|
||||
sub.d t5, t2, t5 // Find jump table target
|
||||
sub.d s0, t2, s0
|
||||
|
||||
vshuf.b vr0, vr0, vr0, vr1 // Permute cand_b to output refmvs_temporal_block
|
||||
vshuf.b vr4, vr4, vr4, vr5
|
||||
vsle.b vr16, vr10, vr1
|
||||
vand.v vr0, vr16, vr0
|
||||
|
||||
vsle.b vr17, vr10, vr5
|
||||
vand.v vr4, vr17, vr4
|
||||
// v1 follows on v0, with another 3 full repetitions of the pattern.
|
||||
vshuf.b vr1, vr0, vr0, vr8 // 1, 2, 3, ... , 15, 16
|
||||
vshuf.b vr5, vr4, vr4, vr8 // 1, 2, 3, ... , 15, 16
|
||||
// v2 ends with 3 complete repetitions of the pattern.
|
||||
vshuf.b vr2, vr1, vr0, vr7
|
||||
vshuf.b vr6, vr5, vr4, vr7 // 4, 5, 6, 7, ... , 12, 13, 14, 15, 16, 17, 18, 19
|
||||
|
||||
jirl ra, t5, 0
|
||||
bge t1 , t4, 4f // if (cand_b >= end)
|
||||
vor.v vr0, vr4, vr4
|
||||
vor.v vr1, vr5, vr5
|
||||
vor.v vr2, vr6, vr6
|
||||
jirl ra, s0, 0
|
||||
blt t3, t4, 2b // if (cand_b < end)
|
||||
|
||||
4:
|
||||
addi.d a5, a5, -1 // h--
|
||||
addi.d a7, a7, 2 // y += 2
|
||||
add.d a0, a0, a1 // rp += stride
|
||||
blt zero, a5, 1b
|
||||
|
||||
ld.d s0, sp, 0x00
|
||||
ld.d s1, sp, 0x08
|
||||
ld.d s2, sp, 0x10
|
||||
ld.d s3, sp, 0x18
|
||||
ld.d s4, sp, 0x20
|
||||
addi.d sp, sp, 0x28
|
||||
|
||||
move ra, t0
|
||||
jirl zero, ra, 0x00
|
||||
|
||||
10:
|
||||
addi.d s1, a3, 4
|
||||
vstelm.w vr0, a3, 0, 0 // .mv
|
||||
vstelm.b vr0, s1, 0, 4 // .ref
|
||||
addi.d a3, a3, 5
|
||||
jirl zero, ra, 0x00
|
||||
20:
|
||||
addi.d s1, a3, 8
|
||||
vstelm.d vr0, a3, 0, 0 // .mv
|
||||
vstelm.h vr0, s1, 0, 4 // .ref
|
||||
addi.d a3, a3, 2 * 5
|
||||
jirl zero, ra, 0x00
|
||||
40:
|
||||
vst vr0, a3, 0
|
||||
vstelm.w vr1, a3, 0x10, 0
|
||||
addi.d a3, a3, 4 * 5
|
||||
jirl zero, ra, 0x00
|
||||
|
||||
80:
|
||||
vst vr0, a3, 0
|
||||
vst vr1, a3, 0x10 // This writes 6 full entries plus 2 extra bytes
|
||||
vst vr2, a3, 5 * 8 - 16 // Write the last few, overlapping with the first write.
|
||||
addi.d a3, a3, 8 * 5
|
||||
jirl zero, ra, 0x00
|
||||
160:
|
||||
addi.d s1, a3, 6 * 5
|
||||
addi.d s2, a3, 12 * 5
|
||||
vst vr0, a3, 0
|
||||
vst vr1, a3, 0x10 // This writes 6 full entries plus 2 extra bytes
|
||||
vst vr0, a3, 6 * 5
|
||||
vst vr1, a3, 6 * 5 + 16 // Write another 6 full entries, slightly overlapping with the first set
|
||||
vstelm.d vr0, s2, 0, 0 // Write 8 bytes (one full entry) after the first 12
|
||||
vst vr2, a3, 5 * 16 - 16 // Write the last 3 entries
|
||||
addi.d a3, a3, 16 * 5
|
||||
jirl zero, ra, 0x00
|
||||
|
||||
.save_tevs_tbl:
|
||||
.hword 16 * 12 // bt2 * 12, 12 is sizeof(refmvs_block)
|
||||
.hword .save_tevs_tbl - 160b
|
||||
.hword 16 * 12
|
||||
.hword .save_tevs_tbl - 160b
|
||||
.hword 8 * 12
|
||||
.hword .save_tevs_tbl - 80b
|
||||
.hword 8 * 12
|
||||
.hword .save_tevs_tbl - 80b
|
||||
.hword 8 * 12
|
||||
.hword .save_tevs_tbl - 80b
|
||||
.hword 8 * 12
|
||||
.hword .save_tevs_tbl - 80b
|
||||
.hword 4 * 12
|
||||
.hword .save_tevs_tbl - 40b
|
||||
.hword 4 * 12
|
||||
.hword .save_tevs_tbl - 40b
|
||||
.hword 4 * 12
|
||||
.hword .save_tevs_tbl - 40b
|
||||
.hword 4 * 12
|
||||
.hword .save_tevs_tbl - 40b
|
||||
.hword 2 * 12
|
||||
.hword .save_tevs_tbl - 20b
|
||||
.hword 2 * 12
|
||||
.hword .save_tevs_tbl - 20b
|
||||
.hword 2 * 12
|
||||
.hword .save_tevs_tbl - 20b
|
||||
.hword 2 * 12
|
||||
.hword .save_tevs_tbl - 20b
|
||||
.hword 2 * 12
|
||||
.hword .save_tevs_tbl - 20b
|
||||
.hword 1 * 12
|
||||
.hword .save_tevs_tbl - 10b
|
||||
.hword 1 * 12
|
||||
.hword .save_tevs_tbl - 10b
|
||||
.hword 1 * 12
|
||||
.hword .save_tevs_tbl - 10b
|
||||
.hword 1 * 12
|
||||
.hword .save_tevs_tbl - 10b
|
||||
.hword 1 * 12
|
||||
.hword .save_tevs_tbl - 10b
|
||||
.hword 1 * 12
|
||||
.hword .save_tevs_tbl - 10b
|
||||
.hword 1 * 12
|
||||
.hword .save_tevs_tbl - 10b
|
||||
endfunc
|
||||
|
||||
|
||||
@@ -32,6 +32,8 @@
|
||||
#include "src/refmvs.h"
|
||||
|
||||
decl_splat_mv_fn(dav1d_splat_mv_lsx);
|
||||
decl_load_tmvs_fn(dav1d_load_tmvs_lsx);
|
||||
decl_save_tmvs_fn(dav1d_save_tmvs_lsx);
|
||||
|
||||
static ALWAYS_INLINE void refmvs_dsp_init_loongarch(Dav1dRefmvsDSPContext *const c) {
|
||||
const unsigned flags = dav1d_get_cpu_flags();
|
||||
@@ -39,6 +41,8 @@ static ALWAYS_INLINE void refmvs_dsp_init_loongarch(Dav1dRefmvsDSPContext *const
|
||||
if (!(flags & DAV1D_LOONGARCH_CPU_FLAG_LSX)) return;
|
||||
|
||||
c->splat_mv = dav1d_splat_mv_lsx;
|
||||
c->load_tmvs = dav1d_load_tmvs_lsx;
|
||||
c->save_tmvs = dav1d_save_tmvs_lsx;
|
||||
}
|
||||
|
||||
#endif /* DAV1D_SRC_LOONGARCH_REFMVS_H */
|
||||
|
||||
+1252
-426
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user