Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2480e5046e | ||
|
|
488911486a | ||
|
|
54a0c0bce3 | ||
|
|
6025daca63 | ||
|
|
e545614cd0 | ||
|
|
b6001a2ee3 | ||
|
|
9c8d1e013f | ||
|
|
ed430cd963 | ||
|
|
b3f4b8c95a | ||
|
|
6ed52576f8 | ||
|
|
126ad48991 | ||
|
|
f67a0620a3 | ||
|
|
449fb7d849 | ||
|
|
7a7fbb11c3 | ||
|
|
b31349c22a | ||
|
|
4d61e453cc | ||
|
|
f3b51ec608 | ||
|
|
92b7b949dd | ||
|
|
a0cc119f26 | ||
|
|
c8d05aa7a5 | ||
|
|
b0a590f4fe | ||
|
|
f4d1f0333b | ||
|
|
b610d2de37 | ||
|
|
697e2752d7 | ||
|
|
a8f62a347b | ||
|
|
774267fdac | ||
|
|
d38110a5ce | ||
|
|
23a7561353 | ||
|
|
214fbcee15 | ||
|
|
f7f7fea0dc | ||
|
|
2241068c26 | ||
|
|
3e9a52869c | ||
|
|
eee3381cbe | ||
|
|
5378046abd | ||
|
|
dd1f645371 | ||
|
|
a1fea1fe2a | ||
|
|
2ae73a2b34 | ||
|
|
8d11278e28 | ||
|
|
abe1ce3434 | ||
|
|
ea09355eae | ||
|
|
54d321d742 | ||
|
|
c248442df4 | ||
|
|
1470b7e4de | ||
|
|
0882db30a2 | ||
|
|
9a45b5123f | ||
|
|
84125e4035 | ||
|
|
7b5b93037d | ||
|
|
0de36f7b5c | ||
|
|
c78fdcc80d | ||
|
|
86ae89bf33 | ||
|
|
454edd741c | ||
|
|
bcfbdc81b2 | ||
|
|
7c6370cbfd | ||
|
|
fbfc8b1b83 | ||
|
|
ca65a4e91d | ||
|
|
1af73ce38e | ||
|
|
e7fca060db | ||
|
|
bc4c98de26 | ||
|
|
c3b1e55bdc | ||
|
|
5c1cd5e0c2 | ||
|
|
d5c9353f1b | ||
|
|
fb891f33da | ||
|
|
9f59b19fcd | ||
|
|
f4da23dcb6 | ||
|
|
531a28b6a0 | ||
|
|
9b9cb90bb1 | ||
|
|
9388f05a3c | ||
|
|
b58d4f31ab | ||
|
|
52a3f004a0 | ||
|
|
a3cd36acff | ||
|
|
b7df500106 | ||
|
|
19ccef5fb1 | ||
|
|
e6ed4be02e | ||
|
|
feeb8283a5 | ||
|
|
ec4daf420f | ||
|
|
302f22693a | ||
|
|
7b825531a6 | ||
|
|
de2ed66596 | ||
|
|
3c7eed0e53 | ||
|
|
8f6c8d1a9e | ||
|
|
46947efb83 | ||
|
|
a569fa1540 | ||
|
|
d6194d6a0c | ||
|
|
7d996b1c36 | ||
|
|
5b7a3c0e1b | ||
|
|
9cc0098ce2 | ||
|
|
ab7917910d | ||
|
|
2d7ca63e21 | ||
|
|
4f057bffd6 | ||
|
|
2c32e462ac | ||
|
|
7bcd64357d | ||
|
|
08f8bb66c0 | ||
|
|
faae86fba2 | ||
|
|
4ea9a14567 | ||
|
|
3737766bdd | ||
|
|
efb16fafb0 | ||
|
|
f119e26354 | ||
|
|
7093372e32 | ||
|
|
abf45f7235 | ||
|
|
2847a24498 | ||
|
|
25f99fa9f8 | ||
|
|
a8fbdbac34 | ||
|
|
a6fd497820 | ||
|
|
746b4f0f17 | ||
|
|
9874cd11cb | ||
|
|
bb01e26cfe | ||
|
|
77747bc536 | ||
|
|
aa231b5875 | ||
|
|
22a616bd8f | ||
|
|
1a10d3e09d | ||
|
|
f573ccd107 | ||
|
|
dd103eac93 | ||
|
|
ead476025d | ||
|
|
44950ca173 | ||
|
|
03f1354336 | ||
|
|
4b3769823a | ||
|
|
059d3a04c1 | ||
|
|
2845f54eb8 | ||
|
|
c6208bbb45 | ||
|
|
6975cbe1f0 | ||
|
|
22bf5c27ba | ||
|
|
8cbf61792d | ||
|
|
63a103ba6e | ||
|
|
82194ea9d2 | ||
|
|
8632380a96 | ||
|
|
6bc8204ce5 | ||
|
|
f018aa342a | ||
|
|
a52456b168 | ||
|
|
f2485352a6 | ||
|
|
9ab33228bb | ||
|
|
7b2f5cb3b7 | ||
|
|
10d52646e2 | ||
|
|
88154ed02d | ||
|
|
0abbcd19c1 | ||
|
|
a70bfb52d5 | ||
|
|
6051c86741 | ||
|
|
d0b253ac6e | ||
|
|
1d48b7cb16 | ||
|
|
b57acdf2d3 | ||
|
|
02ea3db8e7 | ||
|
|
4e4f78442e | ||
|
|
556788281d | ||
|
|
f348506463 | ||
|
|
28a77a8698 | ||
|
|
481b3dc4b4 | ||
|
|
efd7ac241d | ||
|
|
1eca91f315 | ||
|
|
4280dff103 | ||
|
|
3dc6052c7e | ||
|
|
8a87e80c74 | ||
|
|
b54b50fe3a | ||
|
|
24233b7c49 | ||
|
|
8c20ca345a | ||
|
|
8e4c209002 | ||
|
|
e819341ec1 | ||
|
|
2b28b88cab | ||
|
|
d7351deccf | ||
|
|
1cce778585 | ||
|
|
04f3ecd026 | ||
|
|
dcb005351e | ||
|
|
32b4d01d16 | ||
|
|
de459ed806 | ||
|
|
efe42481e2 | ||
|
|
c75759876c | ||
|
|
9549167357 | ||
|
|
686e1f0052 | ||
|
|
5a468ae87a | ||
|
|
f0e8560660 | ||
|
|
ad87d62748 | ||
|
|
4f86650979 | ||
|
|
9cc95e5657 | ||
|
|
337b65133d | ||
|
|
ddb0ff5353 | ||
|
|
fe497efa05 | ||
|
|
2be5ee3cca | ||
|
|
c34e63ff2f | ||
|
|
fe3c778c51 | ||
|
|
2d33e12a11 | ||
|
|
240555033b | ||
|
|
ee5ca8a328 | ||
|
|
9f52abf12c | ||
|
|
b7bb2e36b8 | ||
|
|
e76ff6a44e | ||
|
|
5c537a5de0 | ||
|
|
5edd88c919 | ||
|
|
90cc944625 | ||
|
|
590fbff06e | ||
|
|
380940271b | ||
|
|
7d75177446 | ||
|
|
0a4ac4b585 | ||
|
|
7d4a221579 | ||
|
|
d3a9c7ef7f | ||
|
|
72c26f4f7f | ||
|
|
0e8b4adf22 | ||
|
|
8dfa61a61c | ||
|
|
99aa10b3ff | ||
|
|
b751edf624 | ||
|
|
fa8bf57768 | ||
|
|
80346b8813 | ||
|
|
13182b2801 | ||
|
|
dd09f0173e | ||
|
|
ce036a2fc0 | ||
|
|
ddf106f769 | ||
|
|
c35739db5e | ||
|
|
2f8220d757 | ||
|
|
5f6a609253 | ||
|
|
e02df9fc55 | ||
|
|
1c0a8a714a | ||
|
|
5e4f1e3677 | ||
|
|
af8843875a | ||
|
|
d1ee2e9c7d | ||
|
|
0925dfe2c9 | ||
|
|
1085775bc6 | ||
|
|
7d873a329f | ||
|
|
ef24712030 | ||
|
|
20581bf303 | ||
|
|
d17238599b | ||
|
|
3e8c448696 | ||
|
|
7f4aa106f2 | ||
|
|
a6ed4f0d37 | ||
|
|
b858e65476 | ||
|
|
d3d6601727 | ||
|
|
da5bd8b5e3 | ||
|
|
045ed5c91d | ||
|
|
4289cf048d | ||
|
|
59a1114d03 | ||
|
|
682d66555d | ||
|
|
beccb83b16 | ||
|
|
5fcacad32b | ||
|
|
bb1c4fa5bd | ||
|
|
7a2d1601ec | ||
|
|
45fdf951b6 | ||
|
|
cece3541ab | ||
|
|
8356a604f0 | ||
|
|
9df0953cde | ||
|
|
2ec9f3a8aa | ||
|
|
ef8f5fecc8 | ||
|
|
4c294336e6 | ||
|
|
8c68b6f26d | ||
|
|
349fb4910b | ||
|
|
7c72c45be6 | ||
|
|
32fee86033 | ||
|
|
72f3ce5f08 | ||
|
|
af19cda65a | ||
|
|
a3e80069fb | ||
|
|
f1e3305974 | ||
|
|
3cdfe33610 | ||
|
|
47171e4b93 | ||
|
|
7cddbf99b1 | ||
|
|
d1ed72fa87 | ||
|
|
806221440b | ||
|
|
cd10d1c03b | ||
|
|
2db1a99aca | ||
|
|
619588fbab | ||
|
|
f39301935c | ||
|
|
2e44ca0136 | ||
|
|
7d27b182fc | ||
|
|
1d83ca4bca | ||
|
|
bec9d9f63d | ||
|
|
89fc5b8f4f | ||
|
|
7fd12a5e69 | ||
|
|
2ba9a567aa | ||
|
|
b4b952eece | ||
|
|
7d1becc575 | ||
|
|
6bb1805ed6 | ||
|
|
0f0a0be95d | ||
|
|
dbbb39199f | ||
|
|
e9acb46431 | ||
|
|
c6c2a71fb7 | ||
|
|
cdb5d2737e | ||
|
|
13d411677f | ||
|
|
f9dba63c28 | ||
|
|
989e6bbdd3 | ||
|
|
04255be948 | ||
|
|
a7bc8ec1f1 | ||
|
|
8cd2b32fef | ||
|
|
4c766cd11f | ||
|
|
c28560129f | ||
|
|
b9e4fb206d | ||
|
|
b06880c2cd | ||
|
|
cbc583eb54 | ||
|
|
e5ba7c3235 | ||
|
|
435d84a7ce | ||
|
|
139f632ca4 | ||
|
|
c17d6dacb2 | ||
|
|
44d0032f3b | ||
|
|
5d86becdae | ||
|
|
76ea8db4da | ||
|
|
aa50185647 | ||
|
|
fee5abd84b | ||
|
|
478d1086c1 | ||
|
|
93c8bafff5 | ||
|
|
6b58bca18b | ||
|
|
b5858c4472 | ||
|
|
fa777f5517 | ||
|
|
8592c21af4 | ||
|
|
3e79f6d89a | ||
|
|
323d7da4f7 | ||
|
|
f57fc932ac | ||
|
|
91ec21202b | ||
|
|
72e070539c | ||
|
|
02c6e764f2 | ||
|
|
5dc7c3c8e5 | ||
|
|
642c393879 | ||
|
|
ae3f5c737c | ||
|
|
0d72d75bf9 | ||
|
|
ca7682e3a3 | ||
|
|
9967e61abb | ||
|
|
a87736346f | ||
|
|
4c9d9940fd | ||
|
|
13b32f69b7 | ||
|
|
3d8c6d9607 | ||
|
|
49b61a3f30 | ||
|
|
f88470323b | ||
|
|
9186456a12 | ||
|
|
6022e5629c | ||
|
|
57ed58cefe | ||
|
|
17d32a4a82 | ||
|
|
59cb5de46b | ||
|
|
4271cfcc6f | ||
|
|
be3349405d | ||
|
|
0a2077901c | ||
|
|
e6d6d3ee43 | ||
|
|
0b8f7c8c10 | ||
|
|
34207bdf5b |
+167
-162
@@ -1,33 +1,38 @@
|
||||
# XXX: Precise is already deprecated, new default is Trusty.
|
||||
# https://blog.travis-ci.com/2017-07-11-trusty-as-default-linux-is-coming
|
||||
dist: precise
|
||||
dist: focal
|
||||
sudo: true
|
||||
language: c
|
||||
|
||||
matrix:
|
||||
include:
|
||||
- &test-ubuntu
|
||||
os: linux
|
||||
# os: linux
|
||||
compiler: gcc
|
||||
addons:
|
||||
apt:
|
||||
packages:
|
||||
- gfortran
|
||||
# before_script: &common-before
|
||||
# - COMMON_FLAGS="DYNAMIC_ARCH=1 TARGET=NEHALEM NUM_THREADS=32"
|
||||
# script:
|
||||
# - make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
# - make -C test $COMMON_FLAGS $BTYPE
|
||||
# - make -C ctest $COMMON_FLAGS $BTYPE
|
||||
# - make -C utest $COMMON_FLAGS $BTYPE
|
||||
# env:
|
||||
# - TARGET_BOX=LINUX64
|
||||
# - BTYPE="BINARY=64"
|
||||
#
|
||||
# - <<: *test-ubuntu
|
||||
os: linux-ppc64le
|
||||
before_script: &common-before
|
||||
- COMMON_FLAGS="DYNAMIC_ARCH=1 TARGET=NEHALEM NUM_THREADS=32"
|
||||
- COMMON_FLAGS="DYNAMIC_ARCH=1 TARGET=POWER8 NUM_THREADS=32"
|
||||
script:
|
||||
- make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
- make -C test $COMMON_FLAGS $BTYPE
|
||||
- make -C ctest $COMMON_FLAGS $BTYPE
|
||||
- make -C utest $COMMON_FLAGS $BTYPE
|
||||
env:
|
||||
- TARGET_BOX=LINUX64
|
||||
- BTYPE="BINARY=64"
|
||||
|
||||
- <<: *test-ubuntu
|
||||
os: linux-ppc64le
|
||||
before_script:
|
||||
- COMMON_FLAGS="DYNAMIC_ARCH=1 TARGET=POWER8 NUM_THREADS=32"
|
||||
env:
|
||||
# for matrix annotation only
|
||||
- TARGET_BOX=PPC64LE_LINUX
|
||||
@@ -55,38 +60,38 @@ matrix:
|
||||
- TARGET_BOX=IBMZ_LINUX
|
||||
- BTYPE="BINARY=64 USE_OPENMP=0 CC=clang"
|
||||
|
||||
- <<: *test-ubuntu
|
||||
env:
|
||||
- TARGET_BOX=LINUX64
|
||||
- BTYPE="BINARY=64 USE_OPENMP=1"
|
||||
|
||||
- <<: *test-ubuntu
|
||||
env:
|
||||
- TARGET_BOX=LINUX64
|
||||
- BTYPE="BINARY=64 INTERFACE64=1"
|
||||
|
||||
- <<: *test-ubuntu
|
||||
compiler: clang
|
||||
env:
|
||||
- TARGET_BOX=LINUX64
|
||||
- BTYPE="BINARY=64 CC=clang"
|
||||
|
||||
- <<: *test-ubuntu
|
||||
compiler: clang
|
||||
env:
|
||||
- TARGET_BOX=LINUX64
|
||||
- BTYPE="BINARY=64 INTERFACE64=1 CC=clang"
|
||||
|
||||
- <<: *test-ubuntu
|
||||
addons:
|
||||
apt:
|
||||
packages:
|
||||
- gcc-multilib
|
||||
- gfortran-multilib
|
||||
env:
|
||||
- TARGET_BOX=LINUX32
|
||||
- BTYPE="BINARY=32"
|
||||
|
||||
# - <<: *test-ubuntu
|
||||
# env:
|
||||
# - TARGET_BOX=LINUX64
|
||||
# - BTYPE="BINARY=64 USE_OPENMP=1"
|
||||
#
|
||||
# - <<: *test-ubuntu
|
||||
# env:
|
||||
# - TARGET_BOX=LINUX64
|
||||
# - BTYPE="BINARY=64 INTERFACE64=1"
|
||||
#
|
||||
# - <<: *test-ubuntu
|
||||
# compiler: clang
|
||||
# env:
|
||||
# - TARGET_BOX=LINUX64
|
||||
# - BTYPE="BINARY=64 CC=clang"
|
||||
#
|
||||
# - <<: *test-ubuntu
|
||||
# compiler: clang
|
||||
# env:
|
||||
# - TARGET_BOX=LINUX64
|
||||
# - BTYPE="BINARY=64 INTERFACE64=1 CC=clang"
|
||||
#
|
||||
# - <<: *test-ubuntu
|
||||
# addons:
|
||||
# apt:
|
||||
# packages:
|
||||
# - gcc-multilib
|
||||
# - gfortran-multilib
|
||||
# env:
|
||||
# - TARGET_BOX=LINUX32
|
||||
# - BTYPE="BINARY=32"
|
||||
#
|
||||
- os: linux
|
||||
arch: ppc64le
|
||||
dist: bionic
|
||||
@@ -121,47 +126,47 @@ matrix:
|
||||
# for matrix annotation only
|
||||
- TARGET_BOX=PPC64LE_LINUX_P9
|
||||
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
addons:
|
||||
apt:
|
||||
packages:
|
||||
- binutils-mingw-w64-x86-64
|
||||
- gcc-mingw-w64-x86-64
|
||||
- gfortran-mingw-w64-x86-64
|
||||
before_script: *common-before
|
||||
script:
|
||||
- travis_wait 45 make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
env:
|
||||
- TARGET_BOX=WIN64
|
||||
- BTYPE="BINARY=64 HOSTCC=gcc CC=x86_64-w64-mingw32-gcc FC=x86_64-w64-mingw32-gfortran"
|
||||
|
||||
# - os: linux
|
||||
# compiler: gcc
|
||||
# addons:
|
||||
# apt:
|
||||
# packages:
|
||||
# - binutils-mingw-w64-x86-64
|
||||
# - gcc-mingw-w64-x86-64
|
||||
# - gfortran-mingw-w64-x86-64
|
||||
# before_script: *common-before
|
||||
# script:
|
||||
# - travis_wait 45 make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
# env:
|
||||
# - TARGET_BOX=WIN64
|
||||
# - BTYPE="BINARY=64 HOSTCC=gcc CC=x86_64-w64-mingw32-gcc FC=x86_64-w64-mingw32-gfortran"
|
||||
#
|
||||
# Build & test on Alpine Linux inside chroot, i.e. on system with musl libc.
|
||||
# These jobs needs sudo, so Travis runs them on VM-based infrastructure
|
||||
# which is slower than container-based infrastructure used for jobs
|
||||
# that don't require sudo.
|
||||
- &test-alpine
|
||||
os: linux
|
||||
dist: trusty
|
||||
sudo: true
|
||||
language: minimal
|
||||
before_install:
|
||||
- "wget 'https://raw.githubusercontent.com/alpinelinux/alpine-chroot-install/v0.9.0/alpine-chroot-install' \
|
||||
&& echo 'e5dfbbdc0c4b3363b99334510976c86bfa6cb251 alpine-chroot-install' | sha1sum -c || exit 1"
|
||||
- alpine() { /alpine/enter-chroot -u "$USER" "$@"; }
|
||||
install:
|
||||
- sudo sh alpine-chroot-install -p 'build-base gfortran perl linux-headers'
|
||||
before_script: *common-before
|
||||
script:
|
||||
# XXX: Disable some warnings for now to avoid exceeding Travis limit for log size.
|
||||
- alpine make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
CFLAGS="-Wno-misleading-indentation -Wno-sign-conversion -Wno-incompatible-pointer-types"
|
||||
- alpine make -C test $COMMON_FLAGS $BTYPE
|
||||
- alpine make -C ctest $COMMON_FLAGS $BTYPE
|
||||
- alpine make -C utest $COMMON_FLAGS $BTYPE
|
||||
env:
|
||||
- TARGET_BOX=LINUX64_MUSL
|
||||
- BTYPE="BINARY=64"
|
||||
# - &test-alpine
|
||||
# os: linux
|
||||
# dist: trusty
|
||||
# sudo: true
|
||||
# language: minimal
|
||||
# before_install:
|
||||
# - "wget 'https://raw.githubusercontent.com/alpinelinux/alpine-chroot-install/v0.9.0/alpine-chroot-install' \
|
||||
# && echo 'e5dfbbdc0c4b3363b99334510976c86bfa6cb251 alpine-chroot-install' | sha1sum -c || exit 1"
|
||||
# - alpine() { /alpine/enter-chroot -u "$USER" "$@"; }
|
||||
# install:
|
||||
# - sudo sh alpine-chroot-install -p 'build-base gfortran perl linux-headers'
|
||||
# before_script: *common-before
|
||||
# script:
|
||||
# # XXX: Disable some warnings for now to avoid exceeding Travis limit for log size.
|
||||
# - alpine make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
# CFLAGS="-Wno-misleading-indentation -Wno-sign-conversion -Wno-incompatible-pointer-types"
|
||||
# - alpine make -C test $COMMON_FLAGS $BTYPE
|
||||
# - alpine make -C ctest $COMMON_FLAGS $BTYPE
|
||||
# - alpine make -C utest $COMMON_FLAGS $BTYPE
|
||||
# env:
|
||||
# - TARGET_BOX=LINUX64_MUSL
|
||||
# - BTYPE="BINARY=64"
|
||||
|
||||
# XXX: This job segfaults in TESTS OF THE COMPLEX LEVEL 3 BLAS,
|
||||
# but only on Travis CI, cannot reproduce it elsewhere.
|
||||
@@ -171,98 +176,98 @@ matrix:
|
||||
# - TARGET_BOX=LINUX64_MUSL
|
||||
# - BTYPE="BINARY=64 USE_OPENMP=1"
|
||||
|
||||
- <<: *test-alpine
|
||||
env:
|
||||
- TARGET_BOX=LINUX64_MUSL
|
||||
- BTYPE="BINARY=64 INTERFACE64=1"
|
||||
# - <<: *test-alpine
|
||||
# env:
|
||||
# - TARGET_BOX=LINUX64_MUSL
|
||||
# - BTYPE="BINARY=64 INTERFACE64=1"
|
||||
#
|
||||
# # Build with the same flags as Alpine do in OpenBLAS package.
|
||||
# - <<: *test-alpine
|
||||
# env:
|
||||
# - TARGET_BOX=LINUX64_MUSL
|
||||
# - BTYPE="BINARY=64 NO_AFFINITY=1 USE_OPENMP=0 NO_LAPACK=0 TARGET=CORE2"
|
||||
|
||||
# Build with the same flags as Alpine do in OpenBLAS package.
|
||||
- <<: *test-alpine
|
||||
env:
|
||||
- TARGET_BOX=LINUX64_MUSL
|
||||
- BTYPE="BINARY=64 NO_AFFINITY=1 USE_OPENMP=0 NO_LAPACK=0 TARGET=CORE2"
|
||||
# - &test-cmake
|
||||
# os: linux
|
||||
# compiler: clang
|
||||
# addons:
|
||||
# apt:
|
||||
# packages:
|
||||
# - gfortran
|
||||
# - cmake
|
||||
# dist: trusty
|
||||
# sudo: true
|
||||
# before_script:
|
||||
# - COMMON_ARGS="-DTARGET=NEHALEM -DNUM_THREADS=32"
|
||||
# script:
|
||||
# - mkdir build
|
||||
# - CONFIG=Release
|
||||
# - cmake -Bbuild -H. $CMAKE_ARGS $COMMON_ARGS -DCMAKE_BUILD_TYPE=$CONFIG
|
||||
# - cmake --build build --config $CONFIG -- -j2
|
||||
# env:
|
||||
# - CMAKE=1
|
||||
# - <<: *test-cmake
|
||||
# env:
|
||||
# - CMAKE=1 CMAKE_ARGS="-DNOFORTRAN=1"
|
||||
# - <<: *test-cmake
|
||||
# compiler: gcc
|
||||
# env:
|
||||
# - CMAKE=1
|
||||
|
||||
- &test-cmake
|
||||
os: linux
|
||||
compiler: clang
|
||||
addons:
|
||||
apt:
|
||||
packages:
|
||||
- gfortran
|
||||
- cmake
|
||||
dist: trusty
|
||||
sudo: true
|
||||
before_script:
|
||||
- COMMON_ARGS="-DTARGET=NEHALEM -DNUM_THREADS=32"
|
||||
script:
|
||||
- mkdir build
|
||||
- CONFIG=Release
|
||||
- cmake -Bbuild -H. $CMAKE_ARGS $COMMON_ARGS -DCMAKE_BUILD_TYPE=$CONFIG
|
||||
- cmake --build build --config $CONFIG -- -j2
|
||||
env:
|
||||
- CMAKE=1
|
||||
- <<: *test-cmake
|
||||
env:
|
||||
- CMAKE=1 CMAKE_ARGS="-DNOFORTRAN=1"
|
||||
- <<: *test-cmake
|
||||
compiler: gcc
|
||||
env:
|
||||
- CMAKE=1
|
||||
|
||||
- &test-macos
|
||||
os: osx
|
||||
osx_image: xcode11.5
|
||||
before_script:
|
||||
- COMMON_FLAGS="DYNAMIC_ARCH=1 NUM_THREADS=32"
|
||||
script:
|
||||
- travis_wait 45 make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
env:
|
||||
- BTYPE="TARGET=NEHALEM BINARY=64 INTERFACE64=1 FC=gfortran-9"
|
||||
|
||||
- <<: *test-macos
|
||||
osx_image: xcode12
|
||||
before_script:
|
||||
- COMMON_FLAGS="DYNAMIC_ARCH=1 NUM_THREADS=32"
|
||||
- brew update
|
||||
script:
|
||||
- travis_wait 45 make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
env:
|
||||
- BTYPE="TARGET=HASWELL USE_OPENMP=1 BINARY=64 INTERFACE64=1 CC=gcc-10 FC=gfortran-10"
|
||||
|
||||
- <<: *test-macos
|
||||
osx_image: xcode12
|
||||
before_script:
|
||||
- COMMON_FLAGS="DYNAMIC_ARCH=1 NUM_THREADS=32"
|
||||
- brew update
|
||||
script:
|
||||
- travis_wait 45 make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
env:
|
||||
- BTYPE="TARGET=NEHALEM BINARY=64 INTERFACE64=1 FC=gfortran-10"
|
||||
# - &test-macos
|
||||
# os: osx
|
||||
# osx_image: xcode11.5
|
||||
# before_script:
|
||||
# - COMMON_FLAGS="DYNAMIC_ARCH=1 NUM_THREADS=32"
|
||||
# script:
|
||||
# - travis_wait 45 make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
# env:
|
||||
# - BTYPE="TARGET=NEHALEM BINARY=64 INTERFACE64=1 FC=gfortran-9"
|
||||
#
|
||||
# - <<: *test-macos
|
||||
# osx_image: xcode12
|
||||
# before_script:
|
||||
# - COMMON_FLAGS="DYNAMIC_ARCH=1 NUM_THREADS=32"
|
||||
# - brew update
|
||||
# script:
|
||||
# - travis_wait 45 make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
# env:
|
||||
# - BTYPE="TARGET=HASWELL USE_OPENMP=1 BINARY=64 INTERFACE64=1 CC=gcc-10 FC=gfortran-10"
|
||||
#
|
||||
# - <<: *test-macos
|
||||
# osx_image: xcode12
|
||||
# before_script:
|
||||
# - COMMON_FLAGS="DYNAMIC_ARCH=1 NUM_THREADS=32"
|
||||
# - brew update
|
||||
# script:
|
||||
# - travis_wait 45 make QUIET_MAKE=1 $COMMON_FLAGS $BTYPE
|
||||
# env:
|
||||
# - BTYPE="TARGET=NEHALEM BINARY=64 INTERFACE64=1 FC=gfortran-10"
|
||||
|
||||
# - <<: *test-macos
|
||||
# osx_image: xcode10
|
||||
# env:
|
||||
# - BTYPE="TARGET=NEHALEM BINARY=32 NOFORTRAN=1"
|
||||
|
||||
- <<: *test-macos
|
||||
osx_image: xcode11.5
|
||||
before_script:
|
||||
- COMMON_FLAGS="DYNAMIC_ARCH=1 NUM_THREADS=32"
|
||||
- brew update
|
||||
env:
|
||||
# - <<: *test-macos
|
||||
# osx_image: xcode11.5
|
||||
# before_script:
|
||||
# - COMMON_FLAGS="DYNAMIC_ARCH=1 NUM_THREADS=32"
|
||||
# - brew update
|
||||
# env:
|
||||
# - CC="/Applications/Xcode-10.1.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang"
|
||||
# - CFLAGS="-O2 -Wno-macro-redefined -isysroot /Applications/Xcode-10.1.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS12.1.sdk -arch arm64 -miphoneos-version-min=10.0"
|
||||
- CC="/Applications/Xcode-11.5.GM.Seed.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang"
|
||||
- CFLAGS="-O2 -Wno-macro-redefined -isysroot /Applications/Xcode-11.5.GM.Seed.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS13.5.sdk -arch arm64 -miphoneos-version-min=10.0"
|
||||
- BTYPE="TARGET=ARMV8 BINARY=64 HOSTCC=clang NOFORTRAN=1"
|
||||
- <<: *test-macos
|
||||
osx_image: xcode11.5
|
||||
env:
|
||||
# - CC="/Applications/Xcode-10.1.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang"
|
||||
# - CFLAGS="-O2 -mno-thumb -Wno-macro-redefined -isysroot /Applications/Xcode-10.1.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS12.1.sdk -arch armv7 -miphoneos-version-min=5.1"
|
||||
- CC="/Applications/Xcode-11.5.GM.Seed.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang"
|
||||
- CFLAGS="-O2 -mno-thumb -Wno-macro-redefined -isysroot /Applications/Xcode-11.5.GM.Seed.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS13.5.sdk -arch armv7 -miphoneos-version-min=5.1"
|
||||
- BTYPE="TARGET=ARMV7 HOSTCC=clang NOFORTRAN=1"
|
||||
# - CC="/Applications/Xcode-11.5.GM.Seed.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang"
|
||||
# - CFLAGS="-O2 -Wno-macro-redefined -isysroot /Applications/Xcode-11.5.GM.Seed.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS13.5.sdk -arch arm64 -miphoneos-version-min=10.0"
|
||||
# - BTYPE="TARGET=ARMV8 BINARY=64 HOSTCC=clang NOFORTRAN=1"
|
||||
# - <<: *test-macos
|
||||
# osx_image: xcode11.5
|
||||
# env:
|
||||
## - CC="/Applications/Xcode-10.1.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang"
|
||||
## - CFLAGS="-O2 -mno-thumb -Wno-macro-redefined -isysroot /Applications/Xcode-10.1.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS12.1.sdk -arch armv7 -miphoneos-version-min=5.1"
|
||||
# - CC="/Applications/Xcode-11.5.GM.Seed.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang"
|
||||
# - CFLAGS="-O2 -mno-thumb -Wno-macro-redefined -isysroot /Applications/Xcode-11.5.GM.Seed.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS13.5.sdk -arch armv7 -miphoneos-version-min=5.1"
|
||||
# - BTYPE="TARGET=ARMV7 HOSTCC=clang NOFORTRAN=1"
|
||||
|
||||
- &test-graviton2
|
||||
os: linux
|
||||
|
||||
+219
-126
@@ -3,10 +3,13 @@
|
||||
##
|
||||
|
||||
cmake_minimum_required(VERSION 2.8.5)
|
||||
|
||||
project(OpenBLAS C ASM)
|
||||
|
||||
set(OpenBLAS_MAJOR_VERSION 0)
|
||||
set(OpenBLAS_MINOR_VERSION 3)
|
||||
set(OpenBLAS_PATCH_VERSION 17.dev)
|
||||
set(OpenBLAS_PATCH_VERSION 19)
|
||||
|
||||
set(OpenBLAS_VERSION "${OpenBLAS_MAJOR_VERSION}.${OpenBLAS_MINOR_VERSION}.${OpenBLAS_PATCH_VERSION}")
|
||||
|
||||
# Adhere to GNU filesystem layout conventions
|
||||
@@ -20,51 +23,68 @@ endif()
|
||||
|
||||
#######
|
||||
if(MSVC)
|
||||
option(BUILD_WITHOUT_LAPACK "Do not build LAPACK and LAPACKE (Only BLAS or CBLAS)" ON)
|
||||
option(BUILD_WITHOUT_LAPACK "Do not build LAPACK and LAPACKE (Only BLAS or CBLAS)" ON)
|
||||
endif()
|
||||
|
||||
option(BUILD_WITHOUT_CBLAS "Do not build the C interface (CBLAS) to the BLAS functions" OFF)
|
||||
|
||||
option(DYNAMIC_ARCH "Include support for multiple CPU targets, with automatic selection at runtime (x86/x86_64, aarch64 or ppc only)" OFF)
|
||||
|
||||
option(DYNAMIC_OLDER "Include specific support for older x86 cpu models (Penryn,Dunnington,Atom,Nano,Opteron) with DYNAMIC_ARCH" OFF)
|
||||
|
||||
option(BUILD_RELAPACK "Build with ReLAPACK (recursive implementation of several LAPACK functions on top of standard LAPACK)" OFF)
|
||||
|
||||
option(USE_LOCKING "Use locks even in single-threaded builds to make them callable from multiple threads" OFF)
|
||||
|
||||
if(${CMAKE_SYSTEM_NAME} MATCHES "Linux")
|
||||
option(NO_AFFINITY "Disable support for CPU affinity masks to avoid binding processes from e.g. R or numpy/scipy to a single core" ON)
|
||||
option(NO_AFFINITY "Disable support for CPU affinity masks to avoid binding processes from e.g. R or numpy/scipy to a single core" ON)
|
||||
else()
|
||||
set(NO_AFFINITY 1)
|
||||
set(NO_AFFINITY 1)
|
||||
endif()
|
||||
|
||||
option(CPP_THREAD_SAFETY_TEST "Run a massively parallel DGEMM test to confirm thread safety of the library (requires OpenMP and about 1.3GB of RAM)" OFF)
|
||||
|
||||
option(CPP_THREAD_SAFETY_GEMV "Run a massively parallel DGEMV test to confirm thread safety of the library (requires OpenMP)" OFF)
|
||||
option(BUILD_STATIC_LIBS "Build static library" OFF)
|
||||
if(NOT BUILD_STATIC_LIBS AND NOT BUILD_SHARED_LIBS)
|
||||
set(BUILD_STATIC_LIBS ON CACHE BOOL "Build static library" FORCE)
|
||||
endif()
|
||||
if((BUILD_STATIC_LIBS AND BUILD_SHARED_LIBS) AND MSVC)
|
||||
message(WARNING "Could not enable both BUILD_STATIC_LIBS and BUILD_SHARED_LIBS with MSVC, Disable BUILD_SHARED_LIBS")
|
||||
set(BUILD_SHARED_LIBS OFF CACHE BOOL "Build static library" FORCE)
|
||||
endif()
|
||||
|
||||
# Add a prefix or suffix to all exported symbol names in the shared library.
|
||||
# Avoids conflicts with other BLAS libraries, especially when using
|
||||
# 64 bit integer interfaces in OpenBLAS.
|
||||
|
||||
set(SYMBOLPREFIX "" CACHE STRING "Add a prefix to all exported symbol names in the shared library to avoid conflicts with other BLAS libraries" )
|
||||
|
||||
set(SYMBOLSUFFIX "" CACHE STRING "Add a suffix to all exported symbol names in the shared library, e.g. _64 for INTERFACE64 builds" )
|
||||
|
||||
#######
|
||||
if(BUILD_WITHOUT_LAPACK)
|
||||
set(NO_LAPACK 1)
|
||||
set(NO_LAPACKE 1)
|
||||
set(NO_LAPACK 1)
|
||||
set(NO_LAPACKE 1)
|
||||
endif()
|
||||
|
||||
if(BUILD_WITHOUT_CBLAS)
|
||||
set(NO_CBLAS 1)
|
||||
set(NO_CBLAS 1)
|
||||
endif()
|
||||
|
||||
#######
|
||||
|
||||
if(MSVC AND MSVC_STATIC_CRT)
|
||||
set(CompilerFlags
|
||||
CMAKE_CXX_FLAGS
|
||||
CMAKE_CXX_FLAGS_DEBUG
|
||||
CMAKE_CXX_FLAGS_RELEASE
|
||||
CMAKE_C_FLAGS
|
||||
CMAKE_C_FLAGS_DEBUG
|
||||
CMAKE_C_FLAGS_RELEASE
|
||||
)
|
||||
foreach(CompilerFlag ${CompilerFlags})
|
||||
string(REPLACE "/MD" "/MT" ${CompilerFlag} "${${CompilerFlag}}")
|
||||
endforeach()
|
||||
set(CompilerFlags
|
||||
CMAKE_CXX_FLAGS
|
||||
CMAKE_CXX_FLAGS_DEBUG
|
||||
CMAKE_CXX_FLAGS_RELEASE
|
||||
CMAKE_C_FLAGS
|
||||
CMAKE_C_FLAGS_DEBUG
|
||||
CMAKE_C_FLAGS_RELEASE
|
||||
)
|
||||
foreach(CompilerFlag ${CompilerFlags})
|
||||
string(REPLACE "/MD" "/MT" ${CompilerFlag} "${${CompilerFlag}}")
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
message(WARNING "CMake support is experimental. It does not yet support all build options and may not produce the same Makefiles that OpenBLAS ships with.")
|
||||
@@ -98,7 +118,7 @@ endif ()
|
||||
# set which float types we want to build for
|
||||
if (NOT DEFINED BUILD_SINGLE AND NOT DEFINED BUILD_DOUBLE AND NOT DEFINED BUILD_COMPLEX AND NOT DEFINED BUILD_COMPLEX16)
|
||||
# if none are defined, build for all
|
||||
# set(BUILD_BFLOAT16 true)
|
||||
# set(BUILD_BFLOAT16 true)
|
||||
set(BUILD_SINGLE true)
|
||||
set(BUILD_DOUBLE true)
|
||||
set(BUILD_COMPLEX true)
|
||||
@@ -132,7 +152,7 @@ endif ()
|
||||
|
||||
if (BUILD_BFLOAT16)
|
||||
message(STATUS "Building Half Precision")
|
||||
list(APPEND FLOAT_TYPES "BFLOAT16") # defines nothing
|
||||
# list(APPEND FLOAT_TYPES "BFLOAT16") # defines nothing
|
||||
endif ()
|
||||
|
||||
if (NOT DEFINED CORE OR "${CORE}" STREQUAL "UNKNOWN")
|
||||
@@ -143,9 +163,10 @@ endif ()
|
||||
set( CMAKE_LIBRARY_OUTPUT_DIRECTORY ${PROJECT_BINARY_DIR}/lib)
|
||||
set( CMAKE_ARCHIVE_OUTPUT_DIRECTORY ${PROJECT_BINARY_DIR}/lib)
|
||||
if(MSVC)
|
||||
set( CMAKE_LIBRARY_OUTPUT_DIRECTORY_DEBUG ${PROJECT_BINARY_DIR}/lib/Debug)
|
||||
set( CMAKE_ARCHIVE_OUTPUT_DIRECTORY_RELEASE ${PROJECT_BINARY_DIR}/lib/Release)
|
||||
set( CMAKE_LIBRARY_OUTPUT_DIRECTORY_DEBUG ${PROJECT_BINARY_DIR}/lib/Debug)
|
||||
set( CMAKE_ARCHIVE_OUTPUT_DIRECTORY_RELEASE ${PROJECT_BINARY_DIR}/lib/Release)
|
||||
endif ()
|
||||
|
||||
# get obj vars into format that add_library likes: $<TARGET_OBJS:objlib> (see http://www.cmake.org/cmake/help/v3.0/command/add_library.html)
|
||||
set(TARGET_OBJS "")
|
||||
foreach (SUBDIR ${SUBDIRS})
|
||||
@@ -183,12 +204,61 @@ if (${DYNAMIC_ARCH})
|
||||
endif ()
|
||||
|
||||
# add objects to the openblas lib
|
||||
add_library(${OpenBLAS_LIBNAME} ${LA_SOURCES} ${LAPACKE_SOURCES} ${RELA_SOURCES} ${TARGET_OBJS} ${OpenBLAS_DEF_FILE})
|
||||
target_include_directories(${OpenBLAS_LIBNAME} INTERFACE $<INSTALL_INTERFACE:include/openblas${SUFFIX64}>)
|
||||
if(NOT NO_LAPACK)
|
||||
add_library(LAPACK OBJECT ${LA_SOURCES})
|
||||
list(APPEND TARGET_OBJS "$<TARGET_OBJECTS:LAPACK>")
|
||||
endif()
|
||||
if(NOT NO_LAPACKE)
|
||||
add_library(LAPACKE OBJECT ${LAPACKE_SOURCES})
|
||||
list(APPEND TARGET_OBJS "$<TARGET_OBJECTS:LAPACKE>")
|
||||
endif()
|
||||
if(BUILD_RELAPACK)
|
||||
add_library(RELAPACK OBJECT ${RELA_SOURCES})
|
||||
list(APPEND TARGET_OBJS "$<TARGET_OBJECTS:RELAPACK>")
|
||||
endif()
|
||||
set(OpenBLAS_LIBS "")
|
||||
if(BUILD_STATIC_LIBS)
|
||||
add_library(${OpenBLAS_LIBNAME}_static STATIC ${TARGET_OBJS} ${OpenBLAS_DEF_FILE})
|
||||
target_include_directories(${OpenBLAS_LIBNAME}_static INTERFACE $<INSTALL_INTERFACE:include/openblas${SUFFIX64}>)
|
||||
list(APPEND OpenBLAS_LIBS ${OpenBLAS_LIBNAME}_static)
|
||||
endif()
|
||||
if(BUILD_SHARED_LIBS)
|
||||
add_library(${OpenBLAS_LIBNAME}_shared SHARED ${TARGET_OBJS} ${OpenBLAS_DEF_FILE})
|
||||
target_include_directories(${OpenBLAS_LIBNAME}_shared INTERFACE $<INSTALL_INTERFACE:include/openblas${SUFFIX64}>)
|
||||
list(APPEND OpenBLAS_LIBS ${OpenBLAS_LIBNAME}_shared)
|
||||
endif()
|
||||
if(BUILD_STATIC_LIBS)
|
||||
add_library(${OpenBLAS_LIBNAME} ALIAS ${OpenBLAS_LIBNAME}_static)
|
||||
else()
|
||||
add_library(${OpenBLAS_LIBNAME} ALIAS ${OpenBLAS_LIBNAME}_shared)
|
||||
endif()
|
||||
|
||||
set_target_properties(${OpenBLAS_LIBS} PROPERTIES OUTPUT_NAME ${OpenBLAS_LIBNAME})
|
||||
|
||||
# Android needs to explicitly link against libm
|
||||
if(ANDROID)
|
||||
target_link_libraries(${OpenBLAS_LIBNAME} m)
|
||||
if(BUILD_STATIC_LIBS)
|
||||
target_link_libraries(${OpenBLAS_LIBNAME}_static m)
|
||||
endif()
|
||||
if(BUILD_SHARED_LIBS)
|
||||
target_link_libraries(${OpenBLAS_LIBNAME}_shared m)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (APPLE AND DYNAMIC_ARCH AND BUILD_SHARED_LIBS)
|
||||
set (CMAKE_C_USE_RESPONSE_FILE_FOR_OBJECTS 1)
|
||||
if (NOT NOFORTRAN)
|
||||
set (CMAKE_Fortran_USE_RESPONSE_FILE_FOR_OBJECTS 1)
|
||||
set (CMAKE_Fortran_CREATE_SHARED_LIBRARY
|
||||
"sh -c 'cat ${CMAKE_BINARY_DIR}/CMakeFiles/openblas_shared.dir/objects*.rsp | xargs -n 1024 ar -ru libopenblas.a && exit 0' "
|
||||
"sh -c 'echo \"\" | ${CMAKE_Fortran_COMPILER} -o dummy.o -c -x f95-cpp-input - '"
|
||||
"sh -c '${CMAKE_Fortran_COMPILER} -fpic -shared -Wl,-all_load -Wl,-force_load,libopenblas.a -Wl,-noall_load dummy.o -o ${CMAKE_LIBRARY_OUTPUT_DIRECTORY}/libopenblas.${OpenBLAS_MAJOR_VERSION}.${OpenBLAS_MINOR_VERSION}.dylib'"
|
||||
"sh -c 'ls -l ${CMAKE_BINARY_DIR}/lib'")
|
||||
else ()
|
||||
set (CMAKE_C_CREATE_SHARED_LIBRARY
|
||||
"sh -c 'cat ${CMAKE_BINARY_DIR}/CMakeFiles/openblas_shared.dir/objects*.rsp | xargs -n 1024 ar -ru libopenblas.a && exit 0' "
|
||||
"sh -c '${CMAKE_C_COMPILER} -fpic -shared -Wl,-all_load -Wl,-force_load,libopenblas.a -Wl,-noall_load -o ${CMAKE_LIBRARY_OUTPUT_DIRECTORY}/libopenblas.${OpenBLAS_MAJOR_VERSION}.${OpenBLAS_MINOR_VERSION}.dylib'")
|
||||
endif ()
|
||||
endif()
|
||||
|
||||
# Handle MSVC exports
|
||||
@@ -197,21 +267,21 @@ if(MSVC AND BUILD_SHARED_LIBS)
|
||||
include("${PROJECT_SOURCE_DIR}/cmake/export.cmake")
|
||||
else()
|
||||
# Creates verbose .def file (51KB vs 18KB)
|
||||
set_target_properties(${OpenBLAS_LIBNAME} PROPERTIES WINDOWS_EXPORT_ALL_SYMBOLS true)
|
||||
set_target_properties(${OpenBLAS_LIBNAME}_shared PROPERTIES WINDOWS_EXPORT_ALL_SYMBOLS true)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Set output for libopenblas
|
||||
set_target_properties( ${OpenBLAS_LIBNAME} PROPERTIES RUNTIME_OUTPUT_DIRECTORY ${PROJECT_BINARY_DIR}/lib)
|
||||
set_target_properties( ${OpenBLAS_LIBNAME} PROPERTIES LIBRARY_OUTPUT_NAME_DEBUG "${OpenBLAS_LIBNAME}_d")
|
||||
set_target_properties( ${OpenBLAS_LIBNAME} PROPERTIES EXPORT_NAME "OpenBLAS")
|
||||
set_target_properties( ${OpenBLAS_LIBS} PROPERTIES RUNTIME_OUTPUT_DIRECTORY ${PROJECT_BINARY_DIR}/lib)
|
||||
set_target_properties( ${OpenBLAS_LIBS} PROPERTIES LIBRARY_OUTPUT_NAME_DEBUG "${OpenBLAS_LIBNAME}_d")
|
||||
set_target_properties( ${OpenBLAS_LIBS} PROPERTIES EXPORT_NAME "OpenBLAS")
|
||||
|
||||
foreach (OUTPUTCONFIG ${CMAKE_CONFIGURATION_TYPES})
|
||||
string( TOUPPER ${OUTPUTCONFIG} OUTPUTCONFIG )
|
||||
|
||||
set_target_properties( ${OpenBLAS_LIBNAME} PROPERTIES RUNTIME_OUTPUT_DIRECTORY_${OUTPUTCONFIG} ${PROJECT_BINARY_DIR}/lib/${OUTPUTCONFIG} )
|
||||
set_target_properties( ${OpenBLAS_LIBNAME} PROPERTIES LIBRARY_OUTPUT_DIRECTORY_${OUTPUTCONFIG} ${PROJECT_BINARY_DIR}/lib/${OUTPUTCONFIG} )
|
||||
set_target_properties( ${OpenBLAS_LIBNAME} PROPERTIES ARCHIVE_OUTPUT_DIRECTORY_${OUTPUTCONFIG} ${PROJECT_BINARY_DIR}/lib/${OUTPUTCONFIG} )
|
||||
set_target_properties( ${OpenBLAS_LIBS} PROPERTIES RUNTIME_OUTPUT_DIRECTORY_${OUTPUTCONFIG} ${PROJECT_BINARY_DIR}/lib/${OUTPUTCONFIG} )
|
||||
set_target_properties( ${OpenBLAS_LIBS} PROPERTIES LIBRARY_OUTPUT_DIRECTORY_${OUTPUTCONFIG} ${PROJECT_BINARY_DIR}/lib/${OUTPUTCONFIG} )
|
||||
set_target_properties( ${OpenBLAS_LIBS} PROPERTIES ARCHIVE_OUTPUT_DIRECTORY_${OUTPUTCONFIG} ${PROJECT_BINARY_DIR}/lib/${OUTPUTCONFIG} )
|
||||
endforeach()
|
||||
|
||||
enable_testing()
|
||||
@@ -220,10 +290,17 @@ if (USE_THREAD)
|
||||
# Add threading library to linker
|
||||
find_package(Threads)
|
||||
if (THREADS_HAVE_PTHREAD_ARG)
|
||||
set_property(TARGET ${OpenBLAS_LIBNAME} PROPERTY COMPILE_OPTIONS "-pthread")
|
||||
set_property(TARGET ${OpenBLAS_LIBNAME} PROPERTY INTERFACE_COMPILE_OPTIONS "-pthread")
|
||||
set_target_properties(${OpenBLAS_LIBS} PROPERTIES
|
||||
COMPILE_OPTIONS "-pthread"
|
||||
INTERFACE_COMPILE_OPTIONS "-pthread"
|
||||
)
|
||||
endif()
|
||||
if(BUILD_STATIC_LIBS)
|
||||
target_link_libraries(${OpenBLAS_LIBNAME}_static ${CMAKE_THREAD_LIBS_INIT})
|
||||
endif()
|
||||
if(BUILD_SHARED_LIBS)
|
||||
target_link_libraries(${OpenBLAS_LIBNAME}_shared ${CMAKE_THREAD_LIBS_INIT})
|
||||
endif()
|
||||
target_link_libraries(${OpenBLAS_LIBNAME} ${CMAKE_THREAD_LIBS_INIT})
|
||||
endif()
|
||||
|
||||
#if (MSVC OR NOT NOFORTRAN)
|
||||
@@ -239,97 +316,109 @@ if (NOT NOFORTRAN)
|
||||
add_subdirectory(ctest)
|
||||
endif()
|
||||
add_subdirectory(lapack-netlib/TESTING)
|
||||
if (CPP_THREAD_SAFETY_TEST OR CPP_THREAD_SAFETY_GEMV)
|
||||
add_subdirectory(cpp_thread_test)
|
||||
endif()
|
||||
if (CPP_THREAD_SAFETY_TEST OR CPP_THREAD_SAFETY_GEMV)
|
||||
add_subdirectory(cpp_thread_test)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
set_target_properties(${OpenBLAS_LIBNAME} PROPERTIES
|
||||
set_target_properties(${OpenBLAS_LIBS} PROPERTIES
|
||||
VERSION ${OpenBLAS_MAJOR_VERSION}.${OpenBLAS_MINOR_VERSION}
|
||||
SOVERSION ${OpenBLAS_MAJOR_VERSION}
|
||||
)
|
||||
|
||||
if (BUILD_SHARED_LIBS AND BUILD_RELAPACK)
|
||||
if (NOT MSVC)
|
||||
target_link_libraries(${OpenBLAS_LIBNAME} "-Wl,-allow-multiple-definition")
|
||||
target_link_libraries(${OpenBLAS_LIBNAME}_shared "-Wl,-allow-multiple-definition")
|
||||
else()
|
||||
set(CMAKE_SHARED_LINKER_FLAGS "${CMAKE_SHARED_LINKER_FLAGS} /FORCE:MULTIPLE")
|
||||
set(CMAKE_SHARED_LINKER_FLAGS "${CMAKE_SHARED_LINKER_FLAGS} /FORCE:MULTIPLE")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (BUILD_SHARED_LIBS AND NOT ${SYMBOLPREFIX}${SYMBOLSUFFIX} STREQUAL "")
|
||||
if (NOT DEFINED ARCH)
|
||||
set(ARCH_IN "x86_64")
|
||||
else()
|
||||
set(ARCH_IN ${ARCH})
|
||||
endif()
|
||||
if (NOT DEFINED ARCH)
|
||||
set(ARCH_IN "x86_64")
|
||||
else()
|
||||
set(ARCH_IN ${ARCH})
|
||||
endif()
|
||||
|
||||
if (${CORE} STREQUAL "generic")
|
||||
set(ARCH_IN "GENERIC")
|
||||
endif ()
|
||||
if (${CORE} STREQUAL "generic")
|
||||
set(ARCH_IN "GENERIC")
|
||||
endif ()
|
||||
|
||||
if (NOT DEFINED EXPRECISION)
|
||||
set(EXPRECISION_IN 0)
|
||||
else()
|
||||
set(EXPRECISION_IN ${EXPRECISION})
|
||||
endif()
|
||||
if (NOT DEFINED EXPRECISION)
|
||||
set(EXPRECISION_IN 0)
|
||||
else()
|
||||
set(EXPRECISION_IN ${EXPRECISION})
|
||||
endif()
|
||||
|
||||
if (NOT DEFINED NO_CBLAS)
|
||||
set(NO_CBLAS_IN 0)
|
||||
else()
|
||||
set(NO_CBLAS_IN ${NO_CBLAS})
|
||||
endif()
|
||||
if (NOT DEFINED NO_CBLAS)
|
||||
set(NO_CBLAS_IN 0)
|
||||
else()
|
||||
set(NO_CBLAS_IN ${NO_CBLAS})
|
||||
endif()
|
||||
|
||||
if (NOT DEFINED NO_LAPACK)
|
||||
set(NO_LAPACK_IN 0)
|
||||
else()
|
||||
set(NO_LAPACK_IN ${NO_LAPACK})
|
||||
endif()
|
||||
if (NOT DEFINED NO_LAPACK)
|
||||
set(NO_LAPACK_IN 0)
|
||||
else()
|
||||
set(NO_LAPACK_IN ${NO_LAPACK})
|
||||
endif()
|
||||
|
||||
if (NOT DEFINED NO_LAPACKE)
|
||||
set(NO_LAPACKE_IN 0)
|
||||
else()
|
||||
set(NO_LAPACKE_IN ${NO_LAPACKE})
|
||||
endif()
|
||||
if (NOT DEFINED NO_LAPACKE)
|
||||
set(NO_LAPACKE_IN 0)
|
||||
else()
|
||||
set(NO_LAPACKE_IN ${NO_LAPACKE})
|
||||
endif()
|
||||
|
||||
if (NOT DEFINED NEED2UNDERSCORES)
|
||||
set(NEED2UNDERSCORES_IN 0)
|
||||
else()
|
||||
set(NEED2UNDERSCORES_IN ${NEED2UNDERSCORES})
|
||||
endif()
|
||||
if (NOT DEFINED NEED2UNDERSCORES)
|
||||
set(NEED2UNDERSCORES_IN 0)
|
||||
else()
|
||||
set(NEED2UNDERSCORES_IN ${NEED2UNDERSCORES})
|
||||
endif()
|
||||
|
||||
if (NOT DEFINED ONLY_CBLAS)
|
||||
set(ONLY_CBLAS_IN 0)
|
||||
else()
|
||||
set(ONLY_CBLAS_IN ${ONLY_CBLAS})
|
||||
endif()
|
||||
if (NOT DEFINED ONLY_CBLAS)
|
||||
set(ONLY_CBLAS_IN 0)
|
||||
else()
|
||||
set(ONLY_CBLAS_IN ${ONLY_CBLAS})
|
||||
endif()
|
||||
|
||||
if (NOT DEFINED BU)
|
||||
set(BU _)
|
||||
endif()
|
||||
if (NOT DEFINED BU)
|
||||
set(BU _)
|
||||
endif()
|
||||
|
||||
if (NOT ${SYMBOLPREFIX} STREQUAL "")
|
||||
message(STATUS "adding prefix ${SYMBOLPREFIX} to names of exported symbols in ${OpenBLAS_LIBNAME}")
|
||||
endif()
|
||||
if (NOT ${SYMBOLSUFFIX} STREQUAL "")
|
||||
message(STATUS "adding suffix ${SYMBOLSUFFIX} to names of exported symbols in ${OpenBLAS_LIBNAME}")
|
||||
endif()
|
||||
add_custom_command(TARGET ${OpenBLAS_LIBNAME} POST_BUILD
|
||||
COMMAND perl ${PROJECT_SOURCE_DIR}/exports/gensymbol "objcopy" "${ARCH}" "${BU}" "${EXPRECISION_IN}" "${NO_CBLAS_IN}" "${NO_LAPACK_IN}" "${NO_LAPACKE_IN}" "${NEED2UNDERSCORES_IN}" "${ONLY_CBLAS_IN}" \"${SYMBOLPREFIX}\" \"${SYMBOLSUFFIX}\" "${BUILD_LAPACK_DEPRECATED}" > ${PROJECT_BINARY_DIR}/objcopy.def
|
||||
COMMAND objcopy -v --redefine-syms ${PROJECT_BINARY_DIR}/objcopy.def ${PROJECT_BINARY_DIR}/lib/lib${OpenBLAS_LIBNAME}.so
|
||||
COMMENT "renaming symbols"
|
||||
)
|
||||
if (NOT ${SYMBOLPREFIX} STREQUAL "")
|
||||
message(STATUS "adding prefix ${SYMBOLPREFIX} to names of exported symbols in ${OpenBLAS_LIBNAME}")
|
||||
endif()
|
||||
if (NOT ${SYMBOLSUFFIX} STREQUAL "")
|
||||
message(STATUS "adding suffix ${SYMBOLSUFFIX} to names of exported symbols in ${OpenBLAS_LIBNAME}")
|
||||
endif()
|
||||
|
||||
add_custom_command(TARGET ${OpenBLAS_LIBNAME}_shared POST_BUILD
|
||||
COMMAND perl ${PROJECT_SOURCE_DIR}/exports/gensymbol "objcopy" "${ARCH}" "${BU}" "${EXPRECISION_IN}" "${NO_CBLAS_IN}" "${NO_LAPACK_IN}" "${NO_LAPACKE_IN}" "${NEED2UNDERSCORES_IN}" "${ONLY_CBLAS_IN}" \"${SYMBOLPREFIX}\" \"${SYMBOLSUFFIX}\" "${BUILD_LAPACK_DEPRECATED}" > ${PROJECT_BINARY_DIR}/objcopy.def
|
||||
COMMAND objcopy -v --redefine-syms ${PROJECT_BINARY_DIR}/objcopy.def ${PROJECT_BINARY_DIR}/lib/lib${OpenBLAS_LIBNAME}.so
|
||||
COMMENT "renaming symbols"
|
||||
)
|
||||
endif()
|
||||
|
||||
|
||||
# Install project
|
||||
|
||||
# Install libraries
|
||||
install(TARGETS ${OpenBLAS_LIBNAME}
|
||||
EXPORT "OpenBLAS${SUFFIX64}Targets"
|
||||
RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR}
|
||||
ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR} )
|
||||
if(BUILD_SHARED_LIBS AND BUILD_STATIC_LIBS)
|
||||
install(TARGETS ${OpenBLAS_LIBNAME}_shared
|
||||
EXPORT "OpenBLAS${SUFFIX64}Targets"
|
||||
RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR}
|
||||
ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR} )
|
||||
install(TARGETS ${OpenBLAS_LIBNAME}_static
|
||||
ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR} )
|
||||
else()
|
||||
install(TARGETS ${OpenBLAS_LIBS}
|
||||
EXPORT "OpenBLAS${SUFFIX64}Targets"
|
||||
RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR}
|
||||
ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR} )
|
||||
endif()
|
||||
|
||||
# Install headers
|
||||
set(CMAKE_INSTALL_INCLUDEDIR ${CMAKE_INSTALL_INCLUDEDIR}/openblas${SUFFIX64})
|
||||
@@ -365,36 +454,41 @@ if(NOT NOFORTRAN)
|
||||
endif()
|
||||
|
||||
if(NOT NO_CBLAS)
|
||||
message (STATUS "Generating cblas.h in ${CMAKE_INSTALL_INCLUDEDIR}")
|
||||
set(CBLAS_H ${CMAKE_BINARY_DIR}/generated/cblas.h)
|
||||
file(READ ${CMAKE_CURRENT_SOURCE_DIR}/cblas.h CBLAS_H_CONTENTS)
|
||||
string(REPLACE "common" "openblas_config" CBLAS_H_CONTENTS_NEW "${CBLAS_H_CONTENTS}")
|
||||
if (NOT ${SYMBOLPREFIX} STREQUAL "")
|
||||
string(REPLACE " cblas" " ${SYMBOLPREFIX}cblas" CBLAS_H_CONTENTS "${CBLAS_H_CONTENTS_NEW}")
|
||||
string(REPLACE " openblas" " ${SYMBOLPREFIX}openblas" CBLAS_H_CONTENTS_NEW "${CBLAS_H_CONTENTS}")
|
||||
string (REPLACE " ${SYMBOLPREFIX}openblas_complex" " openblas_complex" CBLAS_H_CONTENTS "${CBLAS_H_CONTENTS_NEW}")
|
||||
string(REPLACE " goto" " ${SYMBOLPREFIX}goto" CBLAS_H_CONTENTS_NEW "${CBLAS_H_CONTENTS}")
|
||||
endif()
|
||||
if (NOT ${SYMBOLSUFFIX} STREQUAL "")
|
||||
string(REGEX REPLACE "(cblas[^ (]*)" "\\1${SYMBOLSUFFIX}" CBLAS_H_CONTENTS "${CBLAS_H_CONTENTS_NEW}")
|
||||
string(REGEX REPLACE "(openblas[^ (]*)" "\\1${SYMBOLSUFFIX}" CBLAS_H_CONTENTS_NEW "${CBLAS_H_CONTENTS}")
|
||||
string(REGEX REPLACE "(openblas_complex[^ ]*)${SYMBOLSUFFIX}" "\\1" CBLAS_H_CONTENTS "${CBLAS_H_CONTENTS_NEW}")
|
||||
string(REGEX REPLACE "(goto[^ (]*)" "\\1${SYMBOLSUFFIX}" CBLAS_H_CONTENTS_NEW "${CBLAS_H_CONTENTS}")
|
||||
endif()
|
||||
file(WRITE ${CBLAS_H} "${CBLAS_H_CONTENTS_NEW}")
|
||||
install (FILES ${CBLAS_H} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
|
||||
message (STATUS "Generating cblas.h in ${CMAKE_INSTALL_INCLUDEDIR}")
|
||||
set(CBLAS_H ${CMAKE_BINARY_DIR}/generated/cblas.h)
|
||||
file(READ ${CMAKE_CURRENT_SOURCE_DIR}/cblas.h CBLAS_H_CONTENTS)
|
||||
string(REPLACE "common" "openblas_config" CBLAS_H_CONTENTS_NEW "${CBLAS_H_CONTENTS}")
|
||||
if (NOT ${SYMBOLPREFIX} STREQUAL "")
|
||||
string(REPLACE " cblas" " ${SYMBOLPREFIX}cblas" CBLAS_H_CONTENTS "${CBLAS_H_CONTENTS_NEW}")
|
||||
string(REPLACE " openblas" " ${SYMBOLPREFIX}openblas" CBLAS_H_CONTENTS_NEW "${CBLAS_H_CONTENTS}")
|
||||
string (REPLACE " ${SYMBOLPREFIX}openblas_complex" " openblas_complex" CBLAS_H_CONTENTS "${CBLAS_H_CONTENTS_NEW}")
|
||||
string(REPLACE " goto" " ${SYMBOLPREFIX}goto" CBLAS_H_CONTENTS_NEW "${CBLAS_H_CONTENTS}")
|
||||
endif()
|
||||
if (NOT ${SYMBOLSUFFIX} STREQUAL "")
|
||||
string(REGEX REPLACE "(cblas[^ (]*)" "\\1${SYMBOLSUFFIX}" CBLAS_H_CONTENTS "${CBLAS_H_CONTENTS_NEW}")
|
||||
string(REGEX REPLACE "(openblas[^ (]*)" "\\1${SYMBOLSUFFIX}" CBLAS_H_CONTENTS_NEW "${CBLAS_H_CONTENTS}")
|
||||
string(REGEX REPLACE "(openblas_complex[^ ]*)${SYMBOLSUFFIX}" "\\1" CBLAS_H_CONTENTS "${CBLAS_H_CONTENTS_NEW}")
|
||||
string(REGEX REPLACE "(goto[^ (]*)" "\\1${SYMBOLSUFFIX}" CBLAS_H_CONTENTS_NEW "${CBLAS_H_CONTENTS}")
|
||||
endif()
|
||||
file(WRITE ${CBLAS_H} "${CBLAS_H_CONTENTS_NEW}")
|
||||
install (FILES ${CBLAS_H} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
|
||||
endif()
|
||||
|
||||
if(NOT NO_LAPACKE)
|
||||
message (STATUS "Copying LAPACKE header files to ${CMAKE_INSTALL_INCLUDEDIR}")
|
||||
add_dependencies( ${OpenBLAS_LIBNAME} genlapacke)
|
||||
FILE(GLOB_RECURSE INCLUDE_FILES "${CMAKE_CURRENT_SOURCE_DIR}/lapack-netlib/LAPACKE/*.h")
|
||||
install (FILES ${INCLUDE_FILES} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
|
||||
message (STATUS "Copying LAPACKE header files to ${CMAKE_INSTALL_INCLUDEDIR}")
|
||||
if(BUILD_STATIC_LIBS)
|
||||
add_dependencies( ${OpenBLAS_LIBNAME}_static genlapacke)
|
||||
endif()
|
||||
if(BUILD_SHARED_LIBS)
|
||||
add_dependencies( ${OpenBLAS_LIBNAME}_shared genlapacke)
|
||||
endif()
|
||||
FILE(GLOB_RECURSE INCLUDE_FILES "${CMAKE_CURRENT_SOURCE_DIR}/lapack-netlib/LAPACKE/*.h")
|
||||
install (FILES ${INCLUDE_FILES} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
|
||||
|
||||
ADD_CUSTOM_TARGET(genlapacke
|
||||
COMMAND ${CMAKE_COMMAND} -E copy ${CMAKE_CURRENT_SOURCE_DIR}/lapack-netlib/LAPACKE/include/lapacke_mangling_with_flags.h.in "${CMAKE_BINARY_DIR}/lapacke_mangling.h"
|
||||
)
|
||||
install (FILES ${CMAKE_BINARY_DIR}/lapacke_mangling.h DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}/openblas${SUFFIX64})
|
||||
ADD_CUSTOM_TARGET(genlapacke
|
||||
COMMAND ${CMAKE_COMMAND} -E copy ${CMAKE_CURRENT_SOURCE_DIR}/lapack-netlib/LAPACKE/include/lapacke_mangling_with_flags.h.in "${CMAKE_BINARY_DIR}/lapacke_mangling.h"
|
||||
)
|
||||
install (FILES ${CMAKE_BINARY_DIR}/lapacke_mangling.h DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}/openblas${SUFFIX64})
|
||||
endif()
|
||||
|
||||
# Install pkg-config files
|
||||
@@ -419,4 +513,3 @@ install(FILES ${CMAKE_CURRENT_BINARY_DIR}/${PN}ConfigVersion.cmake
|
||||
install(EXPORT "${PN}${SUFFIX64}Targets"
|
||||
NAMESPACE "${PN}${SUFFIX64}::"
|
||||
DESTINATION ${CMAKECONFIG_INSTALL_DIR})
|
||||
|
||||
|
||||
@@ -197,3 +197,7 @@ In chronological order:
|
||||
|
||||
* River Dillon <oss@outerpassage.net>
|
||||
* [2021-07-10] fix compilation with musl libc
|
||||
|
||||
* Bine Brank <https://github.com/binebrank>
|
||||
* [2021-10-27] Add vector-length-agnostic DGEMM kernels for Arm SVE
|
||||
* [2021-11-20] Vector-length-agnostic Arm SVE copy routines for DGEMM, DTRMM, DSYMM
|
||||
|
||||
@@ -1,4 +1,94 @@
|
||||
OpenBLAS ChangeLog
|
||||
====================================================================
|
||||
Version 0.3.19
|
||||
19-Dec-2021
|
||||
|
||||
general:
|
||||
- reverted unsafe TRSV/ZRSV optimizations introduced in 0.3.16
|
||||
- fixed a potential thread race in the thread buffer reallocation routines
|
||||
that were introduced in 0.3.18
|
||||
- fixed miscounting of thread pool size on Linux with OMP_PROC_BIND=TRUE
|
||||
- fixed CBLAS interfaces for CSROT/ZSROT and CROTG/ZROTG
|
||||
- made automatic library suffix for CMAKE builds with INTERFACE64 available
|
||||
to CBLAS-only builds
|
||||
|
||||
x86_64:
|
||||
- DYNAMIC_ARCH builds now fall back to the cpu with most similar capabilities
|
||||
when an unknown CPUID is encountered, instead of defaulting to Prescott
|
||||
- added cpu detection for Intel Alder Lake
|
||||
- added cpu detection for Intel Sapphire Rapids
|
||||
- added an optimized SBGEMM kernel for Sapphire Rapids
|
||||
- fixed DYNAMIC_ARCH builds on OSX with CMAKE
|
||||
- worked around DYNAMIC_ARCH builds made on Sandybridge failing on SkylakeX
|
||||
- fixed missing thread initialization for static builds on Windows/MSVC
|
||||
- fixed an excessive read in ZSYMV
|
||||
|
||||
POWER:
|
||||
- added support for POWER10 in big-endian mode
|
||||
- added support for building with CMAKE
|
||||
- added optimized SGEMM and DGEMM kernels for small matrix sizes
|
||||
|
||||
ARMV8:
|
||||
- added basic support and cputype detection for Fujitsu A64FX
|
||||
- added a generic ARMV8SVE target
|
||||
- added SVE-enabled SGEMM and DGEMM kernels for ARMV8SVE and A64FX
|
||||
- added optimized CGEMM and ZGEMM kernels for Cortex A53 and A55 cpus
|
||||
- fixed cpuid detection for Apple M1 and improved performance
|
||||
- improved compiler flag setting in CMAKE builds
|
||||
|
||||
RISCV64:
|
||||
- fixed improper initialization in CSCAL/ZSCAL for strided access patterns
|
||||
|
||||
MIPS:
|
||||
- added a GENERIC target for MIPS32
|
||||
- added support for cross-compiling to MIPS32 on x86_64 using CMAKE
|
||||
|
||||
MIPS64:
|
||||
- fixed misdetection of MSA capability
|
||||
|
||||
====================================================================
|
||||
Version 0.3.18
|
||||
02-Oct-2021
|
||||
|
||||
general:
|
||||
- when the build-time number of preconfigured threads is exceeded
|
||||
at runtime (typically by an external program calling BLAS functions
|
||||
from a larger number of threads in parallel), OpenBLAS will now
|
||||
allocate an auxiliary control structure for up to 512 additional
|
||||
threads instead of aborting
|
||||
- added support for Loongson's LoongArch64 cpu architecture
|
||||
- fixed building OpenBLAS with CMAKE and -DBUILD_BFLOAT16=ON
|
||||
- added support for building OpenBLAS as a CMAKE subproject
|
||||
- added support for building for Windows/ARM64 targets with clang
|
||||
- improved support for building with the IBM xlf compiler
|
||||
- imported Reference-LAPACK PR 625 (out-of-bounds reads in ?LARRV)
|
||||
- imported Reference-LAPACK PR 597 for testsuite compatibility with
|
||||
LLVM's libomp
|
||||
|
||||
x86_64:
|
||||
- added SkylakeX S/DGEMM kernels for small problem sizes (M*N*K<=1000000)
|
||||
- added optimized SBGEMM for Intel Cooper Lake
|
||||
- reinstated the performance patch for AVX512 SGEMV_T with a proper fix
|
||||
- added a workaround for a gcc11 tree-vectorizer bug that caused spurious
|
||||
failures in the test programs for complex BLAS3 when compiling at -O3
|
||||
(the default for cmake "release" builds)
|
||||
- added support for runtime cpu count detection under Haiku OS
|
||||
- worked around a long-standing miscompilation issue of the Haswell DGEMV_T
|
||||
kernel with gcc that could produce NaN output in some corner cases
|
||||
|
||||
POWER:
|
||||
- improved performance of DASUM on POWER10
|
||||
|
||||
ARMV8:
|
||||
- fixed crashes (use of reserved register x18) on Apple M1 under OSX
|
||||
- fixed building with gcc releases earlier than 5.1
|
||||
|
||||
MIPS:
|
||||
- fixed building under BSD
|
||||
|
||||
MIPS64:
|
||||
- fixed building under BSD
|
||||
|
||||
====================================================================
|
||||
Version 0.3.17
|
||||
15-Jul-2021
|
||||
|
||||
@@ -32,7 +32,7 @@ export NOFORTRAN
|
||||
export NO_LAPACK
|
||||
endif
|
||||
|
||||
LAPACK_NOOPT := $(filter-out -O0 -O1 -O2 -O3 -Ofast,$(LAPACK_FFLAGS))
|
||||
LAPACK_NOOPT := $(filter-out -O0 -O1 -O2 -O3 -Ofast -O -Og -Os,$(LAPACK_FFLAGS))
|
||||
|
||||
SUBDIRS_ALL = $(SUBDIRS) test ctest utest exports benchmark ../laswp ../bench cpp_thread_test
|
||||
|
||||
@@ -269,7 +269,7 @@ prof_lapack : lapack_prebuild
|
||||
lapack_prebuild :
|
||||
ifeq ($(NOFORTRAN), $(filter 0,$(NOFORTRAN)))
|
||||
-@echo "FC = $(FC)" > $(NETLIB_LAPACK_DIR)/make.inc
|
||||
-@echo "FFLAGS = $(LAPACK_FFLAGS)" >> $(NETLIB_LAPACK_DIR)/make.inc
|
||||
-@echo "override FFLAGS = $(LAPACK_FFLAGS)" >> $(NETLIB_LAPACK_DIR)/make.inc
|
||||
-@echo "FFLAGS_DRV = $(LAPACK_FFLAGS)" >> $(NETLIB_LAPACK_DIR)/make.inc
|
||||
-@echo "POPTS = $(LAPACK_FPFLAGS)" >> $(NETLIB_LAPACK_DIR)/make.inc
|
||||
-@echo "FFLAGS_NOOPT = -O0 $(LAPACK_NOOPT)" >> $(NETLIB_LAPACK_DIR)/make.inc
|
||||
|
||||
+24
-5
@@ -1,6 +1,9 @@
|
||||
ifneq ($(C_COMPILER), PGI)
|
||||
|
||||
ifneq ($(GCCVERSIONGT4), 1)
|
||||
ifeq ($(C_COMPILER), CLANG)
|
||||
ISCLANG=1
|
||||
endif
|
||||
ifneq (1, $(filter 1,$(GCCVERSIONGT4) $(ISCLANG)))
|
||||
CCOMMON_OPT += -march=armv8-a
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
FCOMMON_OPT += -march=armv8-a
|
||||
@@ -17,6 +20,13 @@ FCOMMON_OPT += -march=armv8-a
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(CORE), ARMV8SVE)
|
||||
CCOMMON_OPT += -march=armv8-a+sve
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
FCOMMON_OPT += -march=armv8-a+sve
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(CORE), CORTEXA53)
|
||||
CCOMMON_OPT += -march=armv8-a -mtune=cortex-a53
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
@@ -48,7 +58,7 @@ endif
|
||||
# Use a72 tunings because Neoverse-N1 is only available
|
||||
# in GCC>=9
|
||||
ifeq ($(CORE), NEOVERSEN1)
|
||||
ifeq ($(GCCVERSIONGTEQ7), 1)
|
||||
ifeq (1, $(filter 1,$(GCCVERSIONGTEQ7) $(ISCLANG)))
|
||||
ifeq ($(GCCVERSIONGTEQ9), 1)
|
||||
CCOMMON_OPT += -march=armv8.2-a -mtune=neoverse-n1
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
@@ -70,7 +80,7 @@ endif
|
||||
|
||||
# Use a53 tunings because a55 is only available in GCC>=8.1
|
||||
ifeq ($(CORE), CORTEXA55)
|
||||
ifeq ($(GCCVERSIONGTEQ7), 1)
|
||||
ifeq (1, $(filter 1,$(GCCVERSIONGTEQ7) $(ISCLANG)))
|
||||
ifeq ($(GCCVERSIONGTEQ8), 1)
|
||||
CCOMMON_OPT += -march=armv8.2-a -mtune=cortex-a55
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
@@ -132,7 +142,7 @@ FCOMMON_OPT += -march=armv8.3-a
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(GCCVERSIONGTEQ9), 1)
|
||||
ifeq (1, $(filter 1,$(GCCVERSIONGTEQ9) $(ISCLANG)))
|
||||
ifeq ($(CORE), TSV110)
|
||||
CCOMMON_OPT += -march=armv8.2-a -mtune=tsv110
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
@@ -150,6 +160,15 @@ endif
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq (1, $(filter 1,$(GCCVERSIONGTEQ11) $(ISCLANG)))
|
||||
ifeq ($(CORE), A64FX)
|
||||
CCOMMON_OPT += -march=armv8.2-a+sve -mtune=a64fx
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
FCOMMON_OPT += -march=armv8.2-a+sve -mtune=a64fx
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
endif
|
||||
endif
|
||||
|
||||
endif
|
||||
|
||||
@@ -12,9 +12,13 @@ endif
|
||||
ifeq ($(CORE), POWER10)
|
||||
ifneq ($(C_COMPILER), PGI)
|
||||
CCOMMON_OPT += -Ofast -mcpu=power10 -mtune=power10 -mvsx -fno-fast-math
|
||||
ifeq ($(F_COMPILER), IBM)
|
||||
FCOMMON_OPT += -O2 -qrecur -qnosave
|
||||
else
|
||||
FCOMMON_OPT += -O2 -frecursive -mcpu=power10 -mtune=power10 -fno-fast-math
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(CORE), POWER9)
|
||||
ifneq ($(C_COMPILER), PGI)
|
||||
@@ -33,7 +37,11 @@ else
|
||||
CCOMMON_OPT += -fast -Mvect=simd -Mcache_align
|
||||
endif
|
||||
ifneq ($(F_COMPILER), PGI)
|
||||
ifeq ($(F_COMPILER), IBM)
|
||||
FCOMMON_OPT += -O2 -qrecur -qnosave
|
||||
else
|
||||
FCOMMON_OPT += -O2 -frecursive -fno-fast-math
|
||||
endif
|
||||
ifeq ($(C_COMPILER), GCC)
|
||||
ifneq ($(GCCVERSIONGT4), 1)
|
||||
$(warning your compiler is too old to fully support POWER9, getting a newer version of gcc is recommended)
|
||||
@@ -57,7 +65,11 @@ CCOMMON_OPT += -fast -Mvect=simd -Mcache_align
|
||||
endif
|
||||
ifneq ($(F_COMPILER), PGI)
|
||||
ifeq ($(OSNAME), AIX)
|
||||
ifeq ($(F_COMPILER), IBM)
|
||||
FCOMMON_OPT += -O2 -qrecur -qnosave
|
||||
else
|
||||
FCOMMON_OPT += -O1 -frecursive -mcpu=power8 -mtune=power8 -fno-fast-math
|
||||
endif
|
||||
else
|
||||
FCOMMON_OPT += -O2 -frecursive -mcpu=power8 -mtune=power8 -fno-fast-math
|
||||
endif
|
||||
|
||||
+1
-1
@@ -3,7 +3,7 @@
|
||||
#
|
||||
|
||||
# This library's version
|
||||
VERSION = 0.3.17.dev
|
||||
VERSION = 0.3.19
|
||||
|
||||
# If you set the suffix, the library name will be libopenblas_$(LIBNAMESUFFIX).a
|
||||
# and libopenblas_$(LIBNAMESUFFIX).so. Meanwhile, the soname in shared library
|
||||
|
||||
+61
-13
@@ -9,11 +9,10 @@ ifndef TOPDIR
|
||||
TOPDIR = .
|
||||
endif
|
||||
|
||||
# If ARCH is not set, we use the host system's architecture for getarch compile options.
|
||||
ifndef ARCH
|
||||
# we need to use the host system's architecture for getarch compile options even especially when cross-compiling
|
||||
HOSTARCH := $(shell uname -m)
|
||||
else
|
||||
HOSTARCH = $(ARCH)
|
||||
ifeq ($(HOSTARCH), amd64)
|
||||
HOSTARCH=x86_64
|
||||
endif
|
||||
|
||||
# Catch conflicting usage of ARCH in some BSD environments
|
||||
@@ -33,6 +32,10 @@ else ifeq ($(ARCH), armv7)
|
||||
override ARCH=arm
|
||||
else ifeq ($(ARCH), aarch64)
|
||||
override ARCH=arm64
|
||||
else ifeq ($(ARCH), mipsel)
|
||||
override ARCH=mips
|
||||
else ifeq ($(ARCH), mips64el)
|
||||
override ARCH=mips64
|
||||
else ifeq ($(ARCH), zarch)
|
||||
override ARCH=zarch
|
||||
endif
|
||||
@@ -98,7 +101,7 @@ GETARCH_FLAGS += -DUSER_TARGET
|
||||
ifeq ($(TARGET), GENERIC)
|
||||
ifeq ($(DYNAMIC_ARCH), 1)
|
||||
override NO_EXPRECISION=1
|
||||
export NO_EXPRECiSION
|
||||
export NO_EXPRECISION
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
@@ -115,6 +118,9 @@ endif
|
||||
ifeq ($(TARGET), COOPERLAKE)
|
||||
GETARCH_FLAGS := -DFORCE_NEHALEM
|
||||
endif
|
||||
ifeq ($(TARGET), SAPPHIRERAPIDS)
|
||||
GETARCH_FLAGS := -DFORCE_NEHALEM
|
||||
endif
|
||||
ifeq ($(TARGET), SANDYBRIDGE)
|
||||
GETARCH_FLAGS := -DFORCE_NEHALEM
|
||||
endif
|
||||
@@ -139,8 +145,13 @@ endif
|
||||
ifeq ($(TARGET), POWER8)
|
||||
GETARCH_FLAGS := -DFORCE_POWER6
|
||||
endif
|
||||
ifeq ($(TARGET), POWER9)
|
||||
GETARCH_FLAGS := -DFORCE_POWER6
|
||||
endif
|
||||
ifeq ($(TARGET), POWER10)
|
||||
GETARCH_FLAGS := -DFORCE_POWER6
|
||||
endif
|
||||
endif
|
||||
|
||||
|
||||
#TARGET_CORE will override TARGET which is used in DYNAMIC_ARCH=1.
|
||||
#
|
||||
@@ -160,6 +171,9 @@ endif
|
||||
ifeq ($(TARGET_CORE), COOPERLAKE)
|
||||
GETARCH_FLAGS := -DFORCE_NEHALEM
|
||||
endif
|
||||
ifeq ($(TARGET_CORE), SAPPHIRERAPIDS)
|
||||
GETARCH_FLAGS := -DFORCE_NEHALEM
|
||||
endif
|
||||
ifeq ($(TARGET_CORE), SANDYBRIDGE)
|
||||
GETARCH_FLAGS := -DFORCE_NEHALEM
|
||||
endif
|
||||
@@ -244,10 +258,24 @@ else
|
||||
ONLY_CBLAS = 0
|
||||
endif
|
||||
|
||||
#For small matrix optimization
|
||||
ifeq ($(ARCH), x86_64)
|
||||
SMALL_MATRIX_OPT = 1
|
||||
else ifeq ($(CORE), POWER10)
|
||||
SMALL_MATRIX_OPT = 1
|
||||
endif
|
||||
ifeq ($(SMALL_MATRIX_OPT), 1)
|
||||
CCOMMON_OPT += -DSMALL_MATRIX_OPT
|
||||
endif
|
||||
|
||||
# This operation is expensive, so execution should be once.
|
||||
ifndef GOTOBLAS_MAKEFILE
|
||||
export GOTOBLAS_MAKEFILE = 1
|
||||
|
||||
# Determine if the assembler is GNU Assembler
|
||||
HAVE_GAS := $(shell $(AS) -v < /dev/null 2>&1 | grep GNU 2>&1 >/dev/null ; echo $$?)
|
||||
GETARCH_FLAGS += -DHAVE_GAS=$(HAVE_GAS)
|
||||
|
||||
# Generating Makefile.conf and config.h
|
||||
DUMMY := $(shell $(MAKE) -C $(TOPDIR) -f Makefile.prebuild CC="$(CC)" FC="$(FC)" HOSTCC="$(HOSTCC)" HOST_CFLAGS="$(GETARCH_FLAGS)" CFLAGS="$(CFLAGS)" BINARY=$(BINARY) USE_OPENMP=$(USE_OPENMP) TARGET_CORE=$(TARGET_CORE) ONLY_CBLAS=$(ONLY_CBLAS) TARGET=$(TARGET) all)
|
||||
|
||||
@@ -295,7 +323,7 @@ else
|
||||
SMP = 1
|
||||
endif
|
||||
else
|
||||
ifeq ($(NUM_THREAD), 1)
|
||||
ifeq ($(NUM_THREADS), 1)
|
||||
SMP =
|
||||
else
|
||||
SMP = 1
|
||||
@@ -856,7 +884,7 @@ BINARY_DEFINED = 1
|
||||
endif
|
||||
|
||||
ifeq ($(ARCH), loongarch64)
|
||||
ifeq ($(CORE), LOONGSONG3R5)
|
||||
ifeq ($(CORE), LOONGSON3R5)
|
||||
CCOMMON_OPT += -march=loongarch64 -mabi=lp64
|
||||
FCOMMON_OPT += -march=loongarch64 -mabi=lp64
|
||||
endif
|
||||
@@ -880,15 +908,25 @@ endif
|
||||
|
||||
ifeq ($(C_COMPILER), PGI)
|
||||
PGCVERSIONGT20 := $(shell expr `$(CC) --version|sed -n "2p" |sed -e "s/[^0-9.]//g" |cut -d "." -f 1` \> 20)
|
||||
PGCVERSIONGTEQ20 := $(shell expr `$(CC) --version|sed -n "2p" |sed -e "s/[^0-9.]//g" |cut -d "." -f 1` \>= 20)
|
||||
PGCMINORVERSIONGE11 := $(shell expr `$(CC) --version|sed -n "2p" |sed -e "s/[^0-9.]//g" |cut -c 4-5` == 11)
|
||||
PGCVERSIONEQ20 := $(shell expr `$(CC) --version|sed -n "2p" |sed -e "s/[^0-9.]//g" |cut -d "." -f 1` == 20)
|
||||
PGCMINORVERSIONGE11 := $(shell expr `$(CC) --version|sed -n "2p" |cut -d "-" -f 1 |sed -e "s/[^0-9.]//g" |cut -c 4-5` \>= 11)
|
||||
PGCVERSIONCHECK := $(PGCVERSIONGT20)$(PGCVERSIONEQ20)$(PGCMINORVERSIONGE11)
|
||||
ifeq ($(PGCVERSIONCHECK), $(filter $(PGCVERSIONCHECK), 110 111 011))
|
||||
ifeq ($(PGCVERSIONCHECK), $(filter $(PGCVERSIONCHECK), 100 101 011))
|
||||
NEWPGI := 1
|
||||
PGCVERSIONGT21 := $(shell expr `$(CC) --version|sed -n "2p" |sed -e "s/[^0-9.]//g" |cut -d "." -f 1` \> 21)
|
||||
PGCVERSIONEQ21 := $(shell expr `$(CC) --version|sed -n "2p" |sed -e "s/[^0-9.]//g" |cut -d "." -f 1` == 21)
|
||||
PGCVERSIONCHECK2 := $(PGCVERSIONGT21)$(PGCVERSIONEQ21)$(PGCMINORVERSIONGE11)
|
||||
ifeq ($(PGCVERSIONCHECK2), $(filter $(PGCVERSIONCHECK2), 100 101 011))
|
||||
NEWPGI2 := 1
|
||||
endif
|
||||
endif
|
||||
ifdef BINARY64
|
||||
ifeq ($(ARCH), x86_64)
|
||||
ifneq ($(NEWPGI2),1)
|
||||
CCOMMON_OPT += -tp p7-64
|
||||
else
|
||||
CCOMMON_OPT += -tp px
|
||||
endif
|
||||
ifneq ($(NEWPGI),1)
|
||||
CCOMMON_OPT += -D__MMX__ -Mnollvm
|
||||
endif
|
||||
@@ -903,7 +941,11 @@ endif
|
||||
endif
|
||||
endif
|
||||
else
|
||||
ifneq ($(NEWPGI2),1)
|
||||
CCOMMON_OPT += -tp p7
|
||||
else
|
||||
CCOMMON_OPT += -tp px
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
@@ -1080,8 +1122,12 @@ FCOMMON_OPT += -i8
|
||||
endif
|
||||
endif
|
||||
ifeq ($(ARCH), x86_64)
|
||||
ifneq ($(NEWPGI2),1)
|
||||
FCOMMON_OPT += -tp p7-64
|
||||
else
|
||||
FCOMMON_OPT += -tp px
|
||||
endif
|
||||
else
|
||||
ifeq ($(ARCH), power)
|
||||
ifeq ($(CORE), POWER6)
|
||||
$(warning NVIDIA HPC compilers do not support POWER6.)
|
||||
@@ -1631,8 +1677,10 @@ export HAVE_VFP
|
||||
export HAVE_VFPV3
|
||||
export HAVE_VFPV4
|
||||
export HAVE_NEON
|
||||
export HAVE_MSA
|
||||
export MSA_FLAGS
|
||||
ifndef NO_MSA
|
||||
export HAVE_MSA
|
||||
export MSA_FLAGS
|
||||
endif
|
||||
export KERNELDIR
|
||||
export FUNCTION_PROFILE
|
||||
export TARGET_CORE
|
||||
|
||||
@@ -81,6 +81,40 @@ CCOMMON_OPT += -march=cooperlake
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
FCOMMON_OPT += -march=cooperlake
|
||||
endif
|
||||
else # gcc not support, fallback to avx512
|
||||
CCOMMON_OPT += -march=skylake-avx512
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
FCOMMON_OPT += -march=skylake-avx512
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
ifeq ($(OSNAME), CYGWIN_NT)
|
||||
CCOMMON_OPT += -fno-asynchronous-unwind-tables
|
||||
FCOMMON_OPT += -fno-asynchronous-unwind-tables
|
||||
endif
|
||||
ifeq ($(OSNAME), WINNT)
|
||||
ifeq ($(C_COMPILER), GCC)
|
||||
CCOMMON_OPT += -fno-asynchronous-unwind-tables
|
||||
FCOMMON_OPT += -fno-asynchronous-unwind-tables
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(CORE), SAPPHIRERAPIDS)
|
||||
ifndef NO_AVX512
|
||||
ifeq ($(C_COMPILER), GCC)
|
||||
# sapphire rapids support was added in 11
|
||||
ifeq ($(GCCVERSIONGTEQ11), 1)
|
||||
CCOMMON_OPT += -march=sapphirerapids
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
FCOMMON_OPT += -march=sapphirerapids
|
||||
endif
|
||||
else # gcc not support, fallback to avx512
|
||||
CCOMMON_OPT += -march=skylake-avx512
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
FCOMMON_OPT += -march=skylake-avx512
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
ifeq ($(OSNAME), CYGWIN_NT)
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
[](https://gitter.im/xianyi/OpenBLAS?utm_source=badge&utm_medium=badge&utm_campaign=pr-badge&utm_content=badge)
|
||||
|
||||
Travis CI: [](https://travis-ci.org/xianyi/OpenBLAS)
|
||||
Travis CI: [](https://travis-ci.com/xianyi/OpenBLAS)
|
||||
|
||||
AppVeyor: [](https://ci.appveyor.com/project/xianyi/openblas/branch/develop)
|
||||
|
||||
@@ -128,6 +128,7 @@ Please read `GotoBLAS_01Readme.txt` for older CPU models already supported by th
|
||||
- **Intel Sandy Bridge**: Optimized Level-3 and Level-2 BLAS with AVX on x86-64.
|
||||
- **Intel Haswell**: Optimized Level-3 and Level-2 BLAS with AVX2 and FMA on x86-64.
|
||||
- **Intel Skylake-X**: Optimized Level-3 and Level-2 BLAS with AVX512 and FMA on x86-64.
|
||||
- **Intel Cooper Lake**: as Skylake-X with improved BFLOAT16 support.
|
||||
- **AMD Bobcat**: Used GotoBLAS2 Barcelona codes.
|
||||
- **AMD Bulldozer**: x86-64 ?GEMM FMA4 kernels. (Thanks to Werner Saar)
|
||||
- **AMD PILEDRIVER**: Uses Bulldozer codes with some optimizations.
|
||||
@@ -153,6 +154,7 @@ Please read `GotoBLAS_01Readme.txt` for older CPU models already supported by th
|
||||
|
||||
- **ARMv8**: Basic ARMV8 with small caches, optimized Level-3 and Level-2 BLAS
|
||||
- **Cortex-A53**: same as ARMV8 (different cpu specifications)
|
||||
- **Cortex-A55**: same as ARMV8 (different cpu specifications)
|
||||
- **Cortex A57**: Optimized Level-3 and Level-2 functions
|
||||
- **Cortex A72**: same as A57 ( different cpu specifications)
|
||||
- **Cortex A73**: same as A57 (different cpu specifications)
|
||||
@@ -178,10 +180,11 @@ Please read `GotoBLAS_01Readme.txt` for older CPU models already supported by th
|
||||
|
||||
#### RISC-V
|
||||
|
||||
- **C910V**: Optimized Leve-3 BLAS (real) and Level-1,2 by RISC-V Vector extension 0.7.1.
|
||||
- **C910V**: Optimized Level-3 BLAS (real) and Level-1,2 by RISC-V Vector extension 0.7.1.
|
||||
```sh
|
||||
make HOSTCC=gcc TARGET=C910V CC=riscv64-unknown-linux-gnu-gcc FC=riscv64-unknown-linux-gnu-gfortran
|
||||
```
|
||||
(also known to work on C906)
|
||||
|
||||
### Support for multiple targets in a single library
|
||||
|
||||
|
||||
@@ -23,6 +23,7 @@ HASWELL
|
||||
SKYLAKEX
|
||||
ATOM
|
||||
COOPERLAKE
|
||||
SAPPHIRERAPIDS
|
||||
|
||||
b)AMD CPU:
|
||||
ATHLON
|
||||
|
||||
+12
-11
@@ -29,15 +29,15 @@ environment:
|
||||
global:
|
||||
CONDA_INSTALL_LOCN: C:\\Miniconda36-x64
|
||||
matrix:
|
||||
- COMPILER: clang-cl
|
||||
WITH_FORTRAN: ON
|
||||
- COMPILER: clang-cl
|
||||
DYNAMIC_ARCH: ON
|
||||
WITH_FORTRAN: OFF
|
||||
- COMPILER: cl
|
||||
- COMPILER: MinGW64-gcc-7.2.0-mingw
|
||||
DYNAMIC_ARCH: OFF
|
||||
WITH_FORTRAN: ignore
|
||||
# - COMPILER: clang-cl
|
||||
# WITH_FORTRAN: ON
|
||||
# - COMPILER: clang-cl
|
||||
# DYNAMIC_ARCH: ON
|
||||
# WITH_FORTRAN: OFF
|
||||
# - COMPILER: cl
|
||||
# - COMPILER: MinGW64-gcc-7.2.0-mingw
|
||||
# DYNAMIC_ARCH: OFF
|
||||
# WITH_FORTRAN: ignore
|
||||
- APPVEYOR_BUILD_WORKER_IMAGE: Visual Studio 2015
|
||||
COMPILER: MinGW-gcc-6.3.0-32
|
||||
- APPVEYOR_BUILD_WORKER_IMAGE: Visual Studio 2015
|
||||
@@ -46,6 +46,7 @@ environment:
|
||||
|
||||
install:
|
||||
- if [%COMPILER%]==[clang-cl] call %CONDA_INSTALL_LOCN%\Scripts\activate.bat
|
||||
- if [%COMPILER%]==[clang-cl] conda update --yes -n base conda
|
||||
- if [%COMPILER%]==[clang-cl] conda config --add channels conda-forge --force
|
||||
- if [%COMPILER%]==[clang-cl] conda config --set auto_update_conda false
|
||||
- if [%COMPILER%]==[clang-cl] conda install --yes --quiet clangdev cmake ninja flang=11.0.1
|
||||
@@ -64,8 +65,8 @@ before_build:
|
||||
- if [%COMPILER%]==[MinGW64-gcc-7.2.0-mingw] cmake -G "MinGW Makefiles" -DNOFORTRAN=1 ..
|
||||
- if [%COMPILER%]==[MinGW-gcc-6.3.0-32] cmake -G "MSYS Makefiles" -DNOFORTRAN=1 ..
|
||||
- if [%COMPILER%]==[MinGW-gcc-5.3.0] cmake -G "MSYS Makefiles" -DNOFORTRAN=1 ..
|
||||
- if [%WITH_FORTRAN%]==[OFF] cmake -G "Ninja" -DCMAKE_CXX_COMPILER=clang-cl -DCMAKE_C_COMPILER=clang-cl -DMSVC_STATIC_CRT=ON ..
|
||||
- if [%WITH_FORTRAN%]==[ON] cmake -G "Ninja" -DCMAKE_CXX_COMPILER=clang-cl -DCMAKE_C_COMPILER=clang-cl -DCMAKE_Fortran_COMPILER=flang -DBUILD_WITHOUT_LAPACK=no -DNOFORTRAN=0 ..
|
||||
- if [%WITH_FORTRAN%]==[OFF] cmake -G "Ninja" -DCMAKE_CXX_COMPILER=clang-cl -DCMAKE_C_COMPILER=clang-cl -DCMAKE_MT=mt -DMSVC_STATIC_CRT=ON ..
|
||||
- if [%WITH_FORTRAN%]==[ON] cmake -G "Ninja" -DCMAKE_CXX_COMPILER=clang-cl -DCMAKE_C_COMPILER=clang-cl -DCMAKE_Fortran_COMPILER=flang -DCMAKE_MT=mt -DBUILD_WITHOUT_LAPACK=no -DNOFORTRAN=0 ..
|
||||
- if [%USE_OPENMP%]==[ON] cmake -DUSE_OPENMP=ON ..
|
||||
- if [%DYNAMIC_ARCH%]==[ON] cmake -DDYNAMIC_ARCH=ON -DDYNAMIC_LIST='CORE2;NEHALEM;SANDYBRIDGE;BULLDOZER;HASWELL' ..
|
||||
|
||||
|
||||
+103
-7
@@ -19,7 +19,7 @@ jobs:
|
||||
# of gcc / glibc
|
||||
- job: manylinux1_gcc
|
||||
pool:
|
||||
vmImage: 'ubuntu-16.04'
|
||||
vmImage: 'ubuntu-latest'
|
||||
steps:
|
||||
- script: |
|
||||
echo "FROM quay.io/pypa/manylinux1_x86_64
|
||||
@@ -35,7 +35,7 @@ jobs:
|
||||
displayName: Run manylinux1 docker build
|
||||
- job: Intel_SDE_skx
|
||||
pool:
|
||||
vmImage: 'ubuntu-16.04'
|
||||
vmImage: 'ubuntu-latest'
|
||||
steps:
|
||||
- script: |
|
||||
# at the time of writing the available Azure Ubuntu vm image
|
||||
@@ -75,7 +75,50 @@ jobs:
|
||||
cd utest
|
||||
dir
|
||||
openblas_utest.exe
|
||||
|
||||
|
||||
- job: Windows_mingw_gmake
|
||||
pool:
|
||||
vmImage: 'windows-latest'
|
||||
steps:
|
||||
- script: |
|
||||
mingw32-make CC=gcc FC=gfortran DYNAMIC_ARCH=1 DYNAMIC_LIST="NEHALEM SANDYBRIDGE HASWELL"
|
||||
|
||||
- job: Windows_clang_cmake
|
||||
pool:
|
||||
vmImage: 'windows-latest'
|
||||
steps:
|
||||
- script: |
|
||||
set "PATH=C:\Miniconda\Scripts;C:\Miniconda\Library\bin;C:\Miniconda\Library\usr\bin;C:\Miniconda\condabin;%PATH%"
|
||||
set "LIB=C:\Miniconda\Library\lib;%LIB%"
|
||||
set "CPATH=C:\Miniconda\Library\include;%CPATH%
|
||||
conda config --add channels conda-forge --force
|
||||
conda config --set auto_update_conda false
|
||||
conda install --yes ninja
|
||||
call "C:\Program Files (x86)\Microsoft Visual Studio\2019\Enterprise\VC\Auxiliary\Build\vcvars64.bat"
|
||||
mkdir build
|
||||
cd build
|
||||
cmake -G "Ninja" -DCMAKE_C_COMPILER=clang-cl -DCMAKE_CXX_COMPILER=clang-cl -DCMAKE_MT=mt -DCMAKE_BUILD_TYPE=Release -DNOFORTRAN=1 -DMSVC_STATIC_CRT=ON ..
|
||||
cmake --build . --config Release
|
||||
ctest
|
||||
|
||||
- job: Windows_flang_clang
|
||||
pool:
|
||||
vmImage: 'windows-latest'
|
||||
steps:
|
||||
- script: |
|
||||
set "PATH=C:\Miniconda\Scripts;C:\Miniconda\Library\bin;C:\Miniconda\Library\usr\bin;C:\Miniconda\condabin;%PATH%"
|
||||
set "LIB=C:\Miniconda\Library\lib;%LIB%"
|
||||
set "CPATH=C:\Miniconda\Library\include;%CPATH%"
|
||||
conda config --add channels conda-forge --force
|
||||
conda config --set auto_update_conda false
|
||||
conda install --yes --quiet ninja flang
|
||||
mkdir build
|
||||
cd build
|
||||
call "C:\Program Files (x86)\Microsoft Visual Studio\2019\Enterprise\VC\Auxiliary\Build\vcvars64.bat"
|
||||
cmake -G "Ninja" -DCMAKE_C_COMPILER=clang-cl -DCMAKE_CXX_COMPILER=clang-cl -DCMAKE_Fortran_COMPILER=flang -DCMAKE_MT=mt -DCMAKE_BUILD_TYPE=Release -DMSVC_STATIC_CRT=ON ..
|
||||
cmake --build . --config Release
|
||||
ctest
|
||||
|
||||
- job: OSX_OpenMP
|
||||
pool:
|
||||
vmImage: 'macOS-10.15'
|
||||
@@ -83,6 +126,8 @@ jobs:
|
||||
- script: |
|
||||
brew update
|
||||
make TARGET=CORE2 DYNAMIC_ARCH=1 USE_OPENMP=1 INTERFACE64=1 CC=gcc-10 FC=gfortran-10
|
||||
make TARGET=CORE2 DYNAMIC_ARCH=1 USE_OPENMP=1 INTERFACE64=1 CC=gcc-10 FC=gfortran-10 PREFIX=../blasinst install
|
||||
ls -lR ../blasinst
|
||||
|
||||
- job: OSX_GCC_Nothreads
|
||||
pool:
|
||||
@@ -104,6 +149,36 @@ jobs:
|
||||
brew install llvm libomp
|
||||
make TARGET=CORE2 USE_OPENMP=1 INTERFACE64=1 DYNAMIC_ARCH=1 CC=/usr/local/opt/llvm/bin/clang FC=gfortran-10
|
||||
|
||||
- job: OSX_OpenMP_Clang_cmake
|
||||
pool:
|
||||
vmImage: 'macOS-10.15'
|
||||
variables:
|
||||
LD_LIBRARY_PATH: /usr/local/opt/llvm/lib
|
||||
LIBRARY_PATH: /usr/local/opt/llvm/lib
|
||||
steps:
|
||||
- script: |
|
||||
brew update
|
||||
brew install llvm libomp
|
||||
mkdir build
|
||||
cd build
|
||||
cmake -DTARGET=CORE2 -DUSE_OPENMP=1 -DINTERFACE64=1 -DDYNAMIC_ARCH=1 -DCMAKE_C_COMPILER=/usr/local/opt/llvm/bin/clang -DNOFORTRAN=1 -DNO_AVX512=1 ..
|
||||
make
|
||||
ctest
|
||||
|
||||
- job: OSX_dynarch_cmake
|
||||
pool:
|
||||
vmImage: 'macOS-10.15'
|
||||
variables:
|
||||
LD_LIBRARY_PATH: /usr/local/opt/llvm/lib
|
||||
LIBRARY_PATH: /usr/local/opt/llvm/lib
|
||||
steps:
|
||||
- script: |
|
||||
mkdir build
|
||||
cd build
|
||||
cmake -DTARGET=CORE2 -DDYNAMIC_ARCH=1 -DCMAKE_C_COMPILER=gcc-10 -DCMAKE_Fortran_COMPILER=gfortran-10 -DBUILD_SHARED_LIBS=ON ..
|
||||
cmake --build .
|
||||
ctest
|
||||
|
||||
- job: OSX_Ifort_Clang
|
||||
pool:
|
||||
vmImage: 'macOS-10.15'
|
||||
@@ -145,15 +220,36 @@ jobs:
|
||||
brew update
|
||||
brew install --cask android-ndk
|
||||
export ANDROID_NDK_HOME=/usr/local/share/android-ndk
|
||||
make TARGET=ARMV7 ONLY_CBLAS=1 CC=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/darwin-x86_64/bin/armv7a-linux-androideabi21-clang AR=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/darwin-x86_64/bin/arm-linux-androideabi-ar HOSTCC=gcc ARM_SOFTFP_ABI=1 -j4
|
||||
|
||||
make TARGET=ARMV7 ONLY_CBLAS=1 CC=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/darwin-x86_64/bin/armv7a-linux-androideabi21-clang AR=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/darwin-x86_64/bin/llvm-ar HOSTCC=gcc ARM_SOFTFP_ABI=1 -j4
|
||||
|
||||
- job: OSX_IOS_ARMV8
|
||||
pool:
|
||||
vmImage: 'macOS-10.15'
|
||||
variables:
|
||||
CC: /Applications/Xcode_12.4.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang
|
||||
CFLAGS: -O2 -Wno-macro-redefined -isysroot /Applications/Xcode_12.4.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS14.4.sdk -arch arm64 -miphoneos-version-min=10.0
|
||||
steps:
|
||||
- script: |
|
||||
make TARGET=ARMV8 DYNAMIC_ARCH=1 NUM_THREADS=32 HOSTCC=clang NOFORTRAN=1
|
||||
|
||||
- job: OSX_IOS_ARMV7
|
||||
pool:
|
||||
vmImage: 'macOS-10.15'
|
||||
variables:
|
||||
CC: /Applications/Xcode_12.4.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang
|
||||
CFLAGS: -O2 -mno-thumb -Wno-macro-redefined -isysroot /Applications/Xcode_12.4.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS14.4.sdk -arch armv7 -miphoneos-version-min=5.1
|
||||
steps:
|
||||
- script: |
|
||||
make TARGET=ARMV7 DYNAMIC_ARCH=1 NUM_THREADS=32 HOSTCC=clang NOFORTRAN=1
|
||||
|
||||
- job: ALPINE_MUSL
|
||||
pool:
|
||||
vmImage: 'ubuntu-latest'
|
||||
steps:
|
||||
- script: |
|
||||
wget 'https://raw.githubusercontent.com/alpinelinux/alpine-chroot-install/v0.9.0/alpine-chroot-install' \
|
||||
&& echo 'e5dfbbdc0c4b3363b99334510976c86bfa6cb251 alpine-chroot-install' | sha1sum -c || exit 1
|
||||
wget https://raw.githubusercontent.com/alpinelinux/alpine-chroot-install/v0.13.2/alpine-chroot-install \
|
||||
&& echo '60c7e0b5d82e21d1a549fc9a46ba3b36688c09dc alpine-chroot-install' | sha1sum -c \
|
||||
|| exit 1
|
||||
alpine() { /alpine/enter-chroot -u "$USER" "$@"; }
|
||||
sudo sh alpine-chroot-install -p 'build-base gfortran perl linux-headers sudo'
|
||||
alpine make DYNAMIC_ARCH=1 BINARY=64
|
||||
|
||||
+2
-2
@@ -125,7 +125,7 @@ int main(int argc, char *argv[]){
|
||||
fprintf(stderr, " %6dx%d : ", (int)m,(int)n);
|
||||
for(j = 0; j < m; j++){
|
||||
for(i = 0; i < n * COMPSIZE; i++){
|
||||
a[(long)j + (long)i * (long)m * COMPSIZE] = ((FLOAT) rand() / (FLOAT) RAND_MAX) - 0.5;
|
||||
a[(long)i + (long)j * (long)m * COMPSIZE] = ((FLOAT) rand() / (FLOAT) RAND_MAX) - 0.5;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -162,7 +162,7 @@ int main(int argc, char *argv[]){
|
||||
fprintf(stderr, " %6dx%d : ", (int)m,(int)n);
|
||||
for(j = 0; j < m; j++){
|
||||
for(i = 0; i < n * COMPSIZE; i++){
|
||||
a[(long)j + (long)i * (long)m * COMPSIZE] = ((FLOAT) rand() / (FLOAT) RAND_MAX) - 0.5;
|
||||
a[(long)i + (long)j * (long)m * COMPSIZE] = ((FLOAT) rand() / (FLOAT) RAND_MAX) - 0.5;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -400,6 +400,8 @@ void cblas_dbf16tod(OPENBLAS_CONST blasint n, OPENBLAS_CONST bfloat16 *in, OPE
|
||||
float cblas_sbdot(OPENBLAS_CONST blasint n, OPENBLAS_CONST bfloat16 *x, OPENBLAS_CONST blasint incx, OPENBLAS_CONST bfloat16 *y, OPENBLAS_CONST blasint incy);
|
||||
void cblas_sbgemv(OPENBLAS_CONST enum CBLAS_ORDER order, OPENBLAS_CONST enum CBLAS_TRANSPOSE trans, OPENBLAS_CONST blasint m, OPENBLAS_CONST blasint n, OPENBLAS_CONST float alpha, OPENBLAS_CONST bfloat16 *a, OPENBLAS_CONST blasint lda, OPENBLAS_CONST bfloat16 *x, OPENBLAS_CONST blasint incx, OPENBLAS_CONST float beta, float *y, OPENBLAS_CONST blasint incy);
|
||||
|
||||
void cblas_sbgemm(OPENBLAS_CONST enum CBLAS_ORDER Order, OPENBLAS_CONST enum CBLAS_TRANSPOSE TransA, OPENBLAS_CONST enum CBLAS_TRANSPOSE TransB, OPENBLAS_CONST blasint M, OPENBLAS_CONST blasint N, OPENBLAS_CONST blasint K,
|
||||
OPENBLAS_CONST float alpha, OPENBLAS_CONST bfloat16 *A, OPENBLAS_CONST blasint lda, OPENBLAS_CONST bfloat16 *B, OPENBLAS_CONST blasint ldb, OPENBLAS_CONST float beta, float *C, OPENBLAS_CONST blasint ldc);
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif /* __cplusplus */
|
||||
|
||||
+5
-1
@@ -109,7 +109,11 @@ if (${ARCH} STREQUAL "ia64")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (MIPS64)
|
||||
if (MIPS32 OR MIPS64)
|
||||
set(NO_BINARY_MODE 1)
|
||||
endif ()
|
||||
|
||||
if (LOONGARCH64)
|
||||
set(NO_BINARY_MODE 1)
|
||||
endif ()
|
||||
|
||||
|
||||
@@ -15,6 +15,11 @@ if (${CMAKE_C_COMPILER_ID} STREQUAL "GNU" OR ${CMAKE_C_COMPILER_ID} STREQUAL "LS
|
||||
|
||||
if (NO_BINARY_MODE)
|
||||
|
||||
if (MIPS32)
|
||||
set(CCOMMON_OPT "${CCOMMON_OPT} -mabi=32")
|
||||
set(BINARY_DEFINED 1)
|
||||
endif ()
|
||||
|
||||
if (MIPS64)
|
||||
if (BINARY64)
|
||||
set(CCOMMON_OPT "${CCOMMON_OPT} -mabi=64")
|
||||
@@ -29,6 +34,15 @@ if (${CMAKE_C_COMPILER_ID} STREQUAL "GNU" OR ${CMAKE_C_COMPILER_ID} STREQUAL "LS
|
||||
set(FCOMMON_OPT "${FCOMMON_OPT} -march=mips64")
|
||||
endif ()
|
||||
|
||||
if (LOONGARCH64)
|
||||
if (BINARY64)
|
||||
set(CCOMMON_OPT "${CCOMMON_OPT} -mabi=lp64")
|
||||
else ()
|
||||
set(CCOMMON_OPT "${CCOMMON_OPT} -mabi=lp32")
|
||||
endif ()
|
||||
set(BINARY_DEFINED 1)
|
||||
endif ()
|
||||
|
||||
if (CMAKE_SYSTEM_NAME STREQUAL "AIX")
|
||||
set(BINARY_DEFINED 1)
|
||||
endif ()
|
||||
@@ -117,6 +131,65 @@ if (${CORE} STREQUAL COOPERLAKE)
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (${CORE} STREQUAL SAPPHIRERAPIDS)
|
||||
if (NOT DYNAMIC_ARCH)
|
||||
if (NOT NO_AVX512)
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -dumpversion OUTPUT_VARIABLE GCC_VERSION)
|
||||
if (${GCC_VERSION} VERSION_GREATER 11.0 OR ${GCC_VERSION} VERSION_EQUAL 11.0)
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -march=sapphirerapids")
|
||||
else ()
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -march=skylake-avx512")
|
||||
endif()
|
||||
endif ()
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (${CORE} STREQUAL A64FX)
|
||||
if (NOT DYNAMIC_ARCH)
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -dumpversion OUTPUT_VARIABLE GCC_VERSION)
|
||||
if (${GCC_VERSION} VERSION_GREATER 11.0 OR ${GCC_VERSION} VERSION_EQUAL 11.0)
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -march=armv8.2-a+sve -mtune=a64fx")
|
||||
else ()
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -march=armv8.2-a+sve")
|
||||
endif()
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (${CORE} STREQUAL ARMV8SVE)
|
||||
if (NOT DYNAMIC_ARCH)
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -march=armv8-a+sve")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (${CORE} STREQUAL POWER10)
|
||||
if (NOT DYNAMIC_ARCH)
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -dumpversion OUTPUT_VARIABLE GCC_VERSION)
|
||||
if (${GCC_VERSION} VERSION_GREATER 10.2 OR ${GCC_VERSION} VERSION_EQUAL 10.2)
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -mcpu=power10 -mtune=power10 -mvsx -fno-fast-math")
|
||||
else ()
|
||||
message(FATAL_ERROR "Compiler GCC.${GCC_VERSION} does not support Power10." )
|
||||
endif()
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (${CORE} STREQUAL POWER9)
|
||||
if (NOT DYNAMIC_ARCH)
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -dumpversion OUTPUT_VARIABLE GCC_VERSION)
|
||||
if (${GCC_VERSION} VERSION_GREATER 5.0 OR ${GCC_VERSION} VERSION_EQUAL 5.0)
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -mcpu=power9 -mtune=power9 -mvsx -fno-fast-math")
|
||||
else ()
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -mcpu=power8 -mtune=power8 -mvsx -fno-fast-math")
|
||||
message(WARNING "Compiler GCC.${GCC_VERSION} does not fully support Power9.")
|
||||
endif ()
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (${CORE} STREQUAL POWER8)
|
||||
if (NOT DYNAMIC_ARCH)
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -mcpu=power8 -mtune=power8 -mvsx -fno-fast-math")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (NOT DYNAMIC_ARCH)
|
||||
if (HAVE_AVX2)
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -mavx2")
|
||||
|
||||
+8
-6
@@ -3,11 +3,6 @@
|
||||
## Description: Ported from portion of OpenBLAS/Makefile.system
|
||||
## Sets Fortran related variables.
|
||||
|
||||
if (INTERFACE64)
|
||||
set(SUFFIX64 64)
|
||||
set(SUFFIX64_UNDERSCORE _64)
|
||||
endif()
|
||||
|
||||
if (${F_COMPILER} STREQUAL "FLANG")
|
||||
set(CCOMMON_OPT "${CCOMMON_OPT} -DF_INTERFACE_FLANG")
|
||||
if (BINARY64 AND INTERFACE64)
|
||||
@@ -61,6 +56,13 @@ if (${F_COMPILER} STREQUAL "GFORTRAN")
|
||||
set(FCOMMON_OPT "${FCOMMON_OPT} -mabi=n32")
|
||||
endif ()
|
||||
endif ()
|
||||
if (LOONGARCH64)
|
||||
if (BINARY64)
|
||||
set(FCOMMON_OPT "${FCOMMON_OPT} -mabi=lp64")
|
||||
else ()
|
||||
set(FCOMMON_OPT "${FCOMMON_OPT} -mabi=lp32")
|
||||
endif ()
|
||||
endif ()
|
||||
else ()
|
||||
if (BINARY64)
|
||||
set(FCOMMON_OPT "${FCOMMON_OPT} -m64")
|
||||
@@ -97,7 +99,7 @@ endif ()
|
||||
|
||||
if (${F_COMPILER} STREQUAL "IBM")
|
||||
set(CCOMMON_OPT "${CCOMMON_OPT} -DF_INTERFACE_IBM")
|
||||
# FCOMMON_OPT += -qarch=440
|
||||
set(FCOMMON_OPT "${FCOMMON_OPT} -qrecur")
|
||||
if (BINARY64)
|
||||
set(FCOMMON_OPT "${FCOMMON_OPT} -q64")
|
||||
if (INTERFACE64)
|
||||
|
||||
+200
-194
@@ -1,212 +1,218 @@
|
||||
# helper functions for the kernel CMakeLists.txt
|
||||
|
||||
function(SetFallback KERNEL SOURCE_PATH)
|
||||
if (NOT (DEFINED ${KERNEL}))
|
||||
set(${KERNEL} ${SOURCE_PATH} PARENT_SCOPE)
|
||||
endif ()
|
||||
endfunction()
|
||||
|
||||
# Set the default filenames for L1 objects. Most of these will be overridden by the appropriate KERNEL file.
|
||||
macro(SetDefaultL1)
|
||||
set(SAMAXKERNEL amax.S)
|
||||
set(DAMAXKERNEL amax.S)
|
||||
set(QAMAXKERNEL amax.S)
|
||||
set(CAMAXKERNEL zamax.S)
|
||||
set(ZAMAXKERNEL zamax.S)
|
||||
set(XAMAXKERNEL zamax.S)
|
||||
set(SAMINKERNEL amin.S)
|
||||
set(DAMINKERNEL amin.S)
|
||||
set(QAMINKERNEL amin.S)
|
||||
set(CAMINKERNEL zamin.S)
|
||||
set(ZAMINKERNEL zamin.S)
|
||||
set(XAMINKERNEL zamin.S)
|
||||
set(SMAXKERNEL max.S)
|
||||
set(DMAXKERNEL max.S)
|
||||
set(QMAXKERNEL max.S)
|
||||
set(SMINKERNEL min.S)
|
||||
set(DMINKERNEL min.S)
|
||||
set(QMINKERNEL min.S)
|
||||
set(ISAMAXKERNEL iamax.S)
|
||||
set(IDAMAXKERNEL iamax.S)
|
||||
set(IQAMAXKERNEL iamax.S)
|
||||
set(ICAMAXKERNEL izamax.S)
|
||||
set(IZAMAXKERNEL izamax.S)
|
||||
set(IXAMAXKERNEL izamax.S)
|
||||
set(ISAMINKERNEL iamin.S)
|
||||
set(IDAMINKERNEL iamin.S)
|
||||
set(IQAMINKERNEL iamin.S)
|
||||
set(ICAMINKERNEL izamin.S)
|
||||
set(IZAMINKERNEL izamin.S)
|
||||
set(IXAMINKERNEL izamin.S)
|
||||
set(ISMAXKERNEL iamax.S)
|
||||
set(IDMAXKERNEL iamax.S)
|
||||
set(IQMAXKERNEL iamax.S)
|
||||
set(ISMINKERNEL iamin.S)
|
||||
set(IDMINKERNEL iamin.S)
|
||||
set(IQMINKERNEL iamin.S)
|
||||
set(SASUMKERNEL asum.S)
|
||||
set(DASUMKERNEL asum.S)
|
||||
set(CASUMKERNEL zasum.S)
|
||||
set(ZASUMKERNEL zasum.S)
|
||||
set(QASUMKERNEL asum.S)
|
||||
set(XASUMKERNEL zasum.S)
|
||||
set(SAXPYKERNEL axpy.S)
|
||||
set(DAXPYKERNEL axpy.S)
|
||||
set(CAXPYKERNEL zaxpy.S)
|
||||
set(ZAXPYKERNEL zaxpy.S)
|
||||
set(QAXPYKERNEL axpy.S)
|
||||
set(XAXPYKERNEL zaxpy.S)
|
||||
set(SCOPYKERNEL copy.S)
|
||||
set(DCOPYKERNEL copy.S)
|
||||
set(CCOPYKERNEL zcopy.S)
|
||||
set(ZCOPYKERNEL zcopy.S)
|
||||
set(QCOPYKERNEL copy.S)
|
||||
set(XCOPYKERNEL zcopy.S)
|
||||
set(SDOTKERNEL dot.S)
|
||||
set(DDOTKERNEL dot.S)
|
||||
set(CDOTKERNEL zdot.S)
|
||||
set(ZDOTKERNEL zdot.S)
|
||||
set(QDOTKERNEL dot.S)
|
||||
set(XDOTKERNEL zdot.S)
|
||||
set(SNRM2KERNEL nrm2.S)
|
||||
set(DNRM2KERNEL nrm2.S)
|
||||
set(QNRM2KERNEL nrm2.S)
|
||||
set(CNRM2KERNEL znrm2.S)
|
||||
set(ZNRM2KERNEL znrm2.S)
|
||||
set(XNRM2KERNEL znrm2.S)
|
||||
set(SROTKERNEL rot.S)
|
||||
set(DROTKERNEL rot.S)
|
||||
set(QROTKERNEL rot.S)
|
||||
set(CROTKERNEL zrot.S)
|
||||
set(ZROTKERNEL zrot.S)
|
||||
set(XROTKERNEL zrot.S)
|
||||
set(SSCALKERNEL scal.S)
|
||||
set(DSCALKERNEL scal.S)
|
||||
set(CSCALKERNEL zscal.S)
|
||||
set(ZSCALKERNEL zscal.S)
|
||||
set(QSCALKERNEL scal.S)
|
||||
set(XSCALKERNEL zscal.S)
|
||||
set(SSWAPKERNEL swap.S)
|
||||
set(DSWAPKERNEL swap.S)
|
||||
set(CSWAPKERNEL zswap.S)
|
||||
set(ZSWAPKERNEL zswap.S)
|
||||
set(QSWAPKERNEL swap.S)
|
||||
set(XSWAPKERNEL zswap.S)
|
||||
set(SGEMVNKERNEL gemv_n.S)
|
||||
set(SGEMVTKERNEL gemv_t.S)
|
||||
set(DGEMVNKERNEL gemv_n.S)
|
||||
set(DGEMVTKERNEL gemv_t.S)
|
||||
set(CGEMVNKERNEL zgemv_n.S)
|
||||
set(CGEMVTKERNEL zgemv_t.S)
|
||||
set(ZGEMVNKERNEL zgemv_n.S)
|
||||
set(ZGEMVTKERNEL zgemv_t.S)
|
||||
set(QGEMVNKERNEL gemv_n.S)
|
||||
set(QGEMVTKERNEL gemv_t.S)
|
||||
set(XGEMVNKERNEL zgemv_n.S)
|
||||
set(XGEMVTKERNEL zgemv_t.S)
|
||||
set(SCABS_KERNEL ../generic/cabs.c)
|
||||
set(DCABS_KERNEL ../generic/cabs.c)
|
||||
set(QCABS_KERNEL ../generic/cabs.c)
|
||||
set(LSAME_KERNEL ../generic/lsame.c)
|
||||
set(SAXPBYKERNEL ../arm/axpby.c)
|
||||
set(DAXPBYKERNEL ../arm/axpby.c)
|
||||
set(CAXPBYKERNEL ../arm/zaxpby.c)
|
||||
set(ZAXPBYKERNEL ../arm/zaxpby.c)
|
||||
set(SSUMKERNEL sum.S)
|
||||
set(DSUMKERNEL sum.S)
|
||||
set(CSUMKERNEL zsum.S)
|
||||
set(ZSUMKERNEL zsum.S)
|
||||
set(QSUMKERNEL sum.S)
|
||||
set(XSUMKERNEL zsum.S)
|
||||
SetFallback(SAMAXKERNEL amax.S)
|
||||
SetFallback(DAMAXKERNEL amax.S)
|
||||
SetFallback(QAMAXKERNEL amax.S)
|
||||
SetFallback(CAMAXKERNEL zamax.S)
|
||||
SetFallback(ZAMAXKERNEL zamax.S)
|
||||
SetFallback(XAMAXKERNEL zamax.S)
|
||||
SetFallback(SAMINKERNEL amin.S)
|
||||
SetFallback(DAMINKERNEL amin.S)
|
||||
SetFallback(QAMINKERNEL amin.S)
|
||||
SetFallback(CAMINKERNEL zamin.S)
|
||||
SetFallback(ZAMINKERNEL zamin.S)
|
||||
SetFallback(XAMINKERNEL zamin.S)
|
||||
SetFallback(SMAXKERNEL max.S)
|
||||
SetFallback(DMAXKERNEL max.S)
|
||||
SetFallback(QMAXKERNEL max.S)
|
||||
SetFallback(SMINKERNEL min.S)
|
||||
SetFallback(DMINKERNEL min.S)
|
||||
SetFallback(QMINKERNEL min.S)
|
||||
SetFallback(ISAMAXKERNEL iamax.S)
|
||||
SetFallback(IDAMAXKERNEL iamax.S)
|
||||
SetFallback(IQAMAXKERNEL iamax.S)
|
||||
SetFallback(ICAMAXKERNEL izamax.S)
|
||||
SetFallback(IZAMAXKERNEL izamax.S)
|
||||
SetFallback(IXAMAXKERNEL izamax.S)
|
||||
SetFallback(ISAMINKERNEL iamin.S)
|
||||
SetFallback(IDAMINKERNEL iamin.S)
|
||||
SetFallback(IQAMINKERNEL iamin.S)
|
||||
SetFallback(ICAMINKERNEL izamin.S)
|
||||
SetFallback(IZAMINKERNEL izamin.S)
|
||||
SetFallback(IXAMINKERNEL izamin.S)
|
||||
SetFallback(ISMAXKERNEL iamax.S)
|
||||
SetFallback(IDMAXKERNEL iamax.S)
|
||||
SetFallback(IQMAXKERNEL iamax.S)
|
||||
SetFallback(ISMINKERNEL iamin.S)
|
||||
SetFallback(IDMINKERNEL iamin.S)
|
||||
SetFallback(IQMINKERNEL iamin.S)
|
||||
SetFallback(SASUMKERNEL asum.S)
|
||||
SetFallback(DASUMKERNEL asum.S)
|
||||
SetFallback(CASUMKERNEL zasum.S)
|
||||
SetFallback(ZASUMKERNEL zasum.S)
|
||||
SetFallback(QASUMKERNEL asum.S)
|
||||
SetFallback(XASUMKERNEL zasum.S)
|
||||
SetFallback(SAXPYKERNEL axpy.S)
|
||||
SetFallback(DAXPYKERNEL axpy.S)
|
||||
SetFallback(CAXPYKERNEL zaxpy.S)
|
||||
SetFallback(ZAXPYKERNEL zaxpy.S)
|
||||
SetFallback(QAXPYKERNEL axpy.S)
|
||||
SetFallback(XAXPYKERNEL zaxpy.S)
|
||||
SetFallback(SCOPYKERNEL copy.S)
|
||||
SetFallback(DCOPYKERNEL copy.S)
|
||||
SetFallback(CCOPYKERNEL zcopy.S)
|
||||
SetFallback(ZCOPYKERNEL zcopy.S)
|
||||
SetFallback(QCOPYKERNEL copy.S)
|
||||
SetFallback(XCOPYKERNEL zcopy.S)
|
||||
SetFallback(SDOTKERNEL dot.S)
|
||||
SetFallback(DDOTKERNEL dot.S)
|
||||
SetFallback(CDOTKERNEL zdot.S)
|
||||
SetFallback(ZDOTKERNEL zdot.S)
|
||||
SetFallback(QDOTKERNEL dot.S)
|
||||
SetFallback(XDOTKERNEL zdot.S)
|
||||
SetFallback(SNRM2KERNEL nrm2.S)
|
||||
SetFallback(DNRM2KERNEL nrm2.S)
|
||||
SetFallback(QNRM2KERNEL nrm2.S)
|
||||
SetFallback(CNRM2KERNEL znrm2.S)
|
||||
SetFallback(ZNRM2KERNEL znrm2.S)
|
||||
SetFallback(XNRM2KERNEL znrm2.S)
|
||||
SetFallback(SROTKERNEL rot.S)
|
||||
SetFallback(DROTKERNEL rot.S)
|
||||
SetFallback(QROTKERNEL rot.S)
|
||||
SetFallback(CROTKERNEL zrot.S)
|
||||
SetFallback(ZROTKERNEL zrot.S)
|
||||
SetFallback(XROTKERNEL zrot.S)
|
||||
SetFallback(SSCALKERNEL scal.S)
|
||||
SetFallback(DSCALKERNEL scal.S)
|
||||
SetFallback(CSCALKERNEL zscal.S)
|
||||
SetFallback(ZSCALKERNEL zscal.S)
|
||||
SetFallback(QSCALKERNEL scal.S)
|
||||
SetFallback(XSCALKERNEL zscal.S)
|
||||
SetFallback(SSWAPKERNEL swap.S)
|
||||
SetFallback(DSWAPKERNEL swap.S)
|
||||
SetFallback(CSWAPKERNEL zswap.S)
|
||||
SetFallback(ZSWAPKERNEL zswap.S)
|
||||
SetFallback(QSWAPKERNEL swap.S)
|
||||
SetFallback(XSWAPKERNEL zswap.S)
|
||||
SetFallback(SGEMVNKERNEL gemv_n.S)
|
||||
SetFallback(SGEMVTKERNEL gemv_t.S)
|
||||
SetFallback(DGEMVNKERNEL gemv_n.S)
|
||||
SetFallback(DGEMVTKERNEL gemv_t.S)
|
||||
SetFallback(CGEMVNKERNEL zgemv_n.S)
|
||||
SetFallback(CGEMVTKERNEL zgemv_t.S)
|
||||
SetFallback(ZGEMVNKERNEL zgemv_n.S)
|
||||
SetFallback(ZGEMVTKERNEL zgemv_t.S)
|
||||
SetFallback(QGEMVNKERNEL gemv_n.S)
|
||||
SetFallback(QGEMVTKERNEL gemv_t.S)
|
||||
SetFallback(XGEMVNKERNEL zgemv_n.S)
|
||||
SetFallback(XGEMVTKERNEL zgemv_t.S)
|
||||
SetFallback(SCABS_KERNEL ../generic/cabs.c)
|
||||
SetFallback(DCABS_KERNEL ../generic/cabs.c)
|
||||
SetFallback(QCABS_KERNEL ../generic/cabs.c)
|
||||
SetFallback(LSAME_KERNEL ../generic/lsame.c)
|
||||
SetFallback(SAXPBYKERNEL ../arm/axpby.c)
|
||||
SetFallback(DAXPBYKERNEL ../arm/axpby.c)
|
||||
SetFallback(CAXPBYKERNEL ../arm/zaxpby.c)
|
||||
SetFallback(ZAXPBYKERNEL ../arm/zaxpby.c)
|
||||
SetFallback(SSUMKERNEL sum.S)
|
||||
SetFallback(DSUMKERNEL sum.S)
|
||||
SetFallback(CSUMKERNEL zsum.S)
|
||||
SetFallback(ZSUMKERNEL zsum.S)
|
||||
SetFallback(QSUMKERNEL sum.S)
|
||||
SetFallback(XSUMKERNEL zsum.S)
|
||||
if (BUILD_BFLOAT16)
|
||||
set(SHAMINKERNEL ../arm/amin.c)
|
||||
set(SHAMAXKERNEL ../arm/amax.c)
|
||||
set(SHMAXKERNEL ../arm/max.c)
|
||||
set(SHMINKERNEL ../arm/min.c)
|
||||
set(ISHAMAXKERNEL ../arm/iamax.c)
|
||||
set(ISHAMINKERNEL ../arm/iamin.c)
|
||||
set(ISHMAXKERNEL ../arm/imax.c)
|
||||
set(ISHMINKERNEL ../arm/imin.c)
|
||||
set(SHASUMKERNEL ../arm/asum.c)
|
||||
set(SHAXPYKERNEL ../arm/axpy.c)
|
||||
set(SHAXPBYKERNEL ../arm/axpby.c)
|
||||
set(SHCOPYKERNEL ../arm/copy.c)
|
||||
set(SBDOTKERNEL ../x86_64/sbdot.c)
|
||||
set(SHROTKERNEL ../arm/rot.c)
|
||||
set(SHSCALKERNEL ../arm/scal.c)
|
||||
set(SHNRM2KERNEL ../arm/nrm2.c)
|
||||
set(SHSUMKERNEL ../arm/sum.c)
|
||||
set(SHSWAPKERNEL ../arm/swap.c)
|
||||
set(TOBF16KERNEL ../x86_64/tobf16.c)
|
||||
set(BF16TOKERNEL ../x86_64/bf16to.c)
|
||||
SetFallback(SHAMINKERNEL ../arm/amin.c)
|
||||
SetFallback(SHAMAXKERNEL ../arm/amax.c)
|
||||
SetFallback(SHMAXKERNEL ../arm/max.c)
|
||||
SetFallback(SHMINKERNEL ../arm/min.c)
|
||||
SetFallback(ISHAMAXKERNEL ../arm/iamax.c)
|
||||
SetFallback(ISHAMINKERNEL ../arm/iamin.c)
|
||||
SetFallback(ISHMAXKERNEL ../arm/imax.c)
|
||||
SetFallback(ISHMINKERNEL ../arm/imin.c)
|
||||
SetFallback(SHASUMKERNEL ../arm/asum.c)
|
||||
SetFallback(SHAXPYKERNEL ../arm/axpy.c)
|
||||
SetFallback(SHAXPBYKERNEL ../arm/axpby.c)
|
||||
SetFallback(SHCOPYKERNEL ../arm/copy.c)
|
||||
SetFallback(SBDOTKERNEL ../x86_64/sbdot.c)
|
||||
SetFallback(SHROTKERNEL ../arm/rot.c)
|
||||
SetFallback(SHSCALKERNEL ../arm/scal.c)
|
||||
SetFallback(SHNRM2KERNEL ../arm/nrm2.c)
|
||||
SetFallback(SHSUMKERNEL ../arm/sum.c)
|
||||
SetFallback(SHSWAPKERNEL ../arm/swap.c)
|
||||
SetFallback(TOBF16KERNEL ../x86_64/tobf16.c)
|
||||
SetFallback(BF16TOKERNEL ../x86_64/bf16to.c)
|
||||
SetFallback(SBGEMVNKERNEL ../x86_64/sbgemv_n.c)
|
||||
SetFallback(SBGEMVTKERNEL ../x86_64/sbgemv_t.c)
|
||||
endif ()
|
||||
endmacro ()
|
||||
|
||||
macro(SetDefaultL2)
|
||||
set(SGEMVNKERNEL ../arm/gemv_n.c)
|
||||
set(SGEMVTKERNEL ../arm/gemv_t.c)
|
||||
set(DGEMVNKERNEL gemv_n.S)
|
||||
set(DGEMVTKERNEL gemv_t.S)
|
||||
set(CGEMVNKERNEL zgemv_n.S)
|
||||
set(CGEMVTKERNEL zgemv_t.S)
|
||||
set(ZGEMVNKERNEL zgemv_n.S)
|
||||
set(ZGEMVTKERNEL zgemv_t.S)
|
||||
set(QGEMVNKERNEL gemv_n.S)
|
||||
set(QGEMVTKERNEL gemv_t.S)
|
||||
set(XGEMVNKERNEL zgemv_n.S)
|
||||
set(XGEMVTKERNEL zgemv_t.S)
|
||||
set(SGERKERNEL ../generic/ger.c)
|
||||
set(DGERKERNEL ../generic/ger.c)
|
||||
set(QGERKERNEL ../generic/ger.c)
|
||||
set(CGERUKERNEL ../generic/zger.c)
|
||||
set(CGERCKERNEL ../generic/zger.c)
|
||||
set(ZGERUKERNEL ../generic/zger.c)
|
||||
set(ZGERCKERNEL ../generic/zger.c)
|
||||
set(XGERUKERNEL ../generic/zger.c)
|
||||
set(XGERCKERNEL ../generic/zger.c)
|
||||
set(SSYMV_U_KERNEL ../generic/symv_k.c)
|
||||
set(SSYMV_L_KERNEL ../generic/symv_k.c)
|
||||
set(DSYMV_U_KERNEL ../generic/symv_k.c)
|
||||
set(DSYMV_L_KERNEL ../generic/symv_k.c)
|
||||
set(QSYMV_U_KERNEL ../generic/symv_k.c)
|
||||
set(QSYMV_L_KERNEL ../generic/symv_k.c)
|
||||
set(CSYMV_U_KERNEL ../generic/zsymv_k.c)
|
||||
set(CSYMV_L_KERNEL ../generic/zsymv_k.c)
|
||||
set(ZSYMV_U_KERNEL ../generic/zsymv_k.c)
|
||||
set(ZSYMV_L_KERNEL ../generic/zsymv_k.c)
|
||||
set(XSYMV_U_KERNEL ../generic/zsymv_k.c)
|
||||
set(XSYMV_L_KERNEL ../generic/zsymv_k.c)
|
||||
set(CHEMV_U_KERNEL ../generic/zhemv_k.c)
|
||||
set(CHEMV_L_KERNEL ../generic/zhemv_k.c)
|
||||
set(CHEMV_V_KERNEL ../generic/zhemv_k.c)
|
||||
set(CHEMV_M_KERNEL ../generic/zhemv_k.c)
|
||||
set(ZHEMV_U_KERNEL ../generic/zhemv_k.c)
|
||||
set(ZHEMV_L_KERNEL ../generic/zhemv_k.c)
|
||||
set(ZHEMV_V_KERNEL ../generic/zhemv_k.c)
|
||||
set(ZHEMV_M_KERNEL ../generic/zhemv_k.c)
|
||||
set(XHEMV_U_KERNEL ../generic/zhemv_k.c)
|
||||
set(XHEMV_L_KERNEL ../generic/zhemv_k.c)
|
||||
set(XHEMV_V_KERNEL ../generic/zhemv_k.c)
|
||||
set(XHEMV_M_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(SGEMVNKERNEL ../arm/gemv_n.c)
|
||||
SetFallback(SGEMVTKERNEL ../arm/gemv_t.c)
|
||||
SetFallback(DGEMVNKERNEL gemv_n.S)
|
||||
SetFallback(DGEMVTKERNEL gemv_t.S)
|
||||
SetFallback(CGEMVNKERNEL zgemv_n.S)
|
||||
SetFallback(CGEMVTKERNEL zgemv_t.S)
|
||||
SetFallback(ZGEMVNKERNEL zgemv_n.S)
|
||||
SetFallback(ZGEMVTKERNEL zgemv_t.S)
|
||||
SetFallback(QGEMVNKERNEL gemv_n.S)
|
||||
SetFallback(QGEMVTKERNEL gemv_t.S)
|
||||
SetFallback(XGEMVNKERNEL zgemv_n.S)
|
||||
SetFallback(XGEMVTKERNEL zgemv_t.S)
|
||||
SetFallback(SGERKERNEL ../generic/ger.c)
|
||||
SetFallback(DGERKERNEL ../generic/ger.c)
|
||||
SetFallback(QGERKERNEL ../generic/ger.c)
|
||||
SetFallback(CGERUKERNEL ../generic/zger.c)
|
||||
SetFallback(CGERCKERNEL ../generic/zger.c)
|
||||
SetFallback(ZGERUKERNEL ../generic/zger.c)
|
||||
SetFallback(ZGERCKERNEL ../generic/zger.c)
|
||||
SetFallback(XGERUKERNEL ../generic/zger.c)
|
||||
SetFallback(XGERCKERNEL ../generic/zger.c)
|
||||
SetFallback(SSYMV_U_KERNEL ../generic/symv_k.c)
|
||||
SetFallback(SSYMV_L_KERNEL ../generic/symv_k.c)
|
||||
SetFallback(DSYMV_U_KERNEL ../generic/symv_k.c)
|
||||
SetFallback(DSYMV_L_KERNEL ../generic/symv_k.c)
|
||||
SetFallback(QSYMV_U_KERNEL ../generic/symv_k.c)
|
||||
SetFallback(QSYMV_L_KERNEL ../generic/symv_k.c)
|
||||
SetFallback(CSYMV_U_KERNEL ../generic/zsymv_k.c)
|
||||
SetFallback(CSYMV_L_KERNEL ../generic/zsymv_k.c)
|
||||
SetFallback(ZSYMV_U_KERNEL ../generic/zsymv_k.c)
|
||||
SetFallback(ZSYMV_L_KERNEL ../generic/zsymv_k.c)
|
||||
SetFallback(XSYMV_U_KERNEL ../generic/zsymv_k.c)
|
||||
SetFallback(XSYMV_L_KERNEL ../generic/zsymv_k.c)
|
||||
SetFallback(CHEMV_U_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(CHEMV_L_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(CHEMV_V_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(CHEMV_M_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(ZHEMV_U_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(ZHEMV_L_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(ZHEMV_V_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(ZHEMV_M_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(XHEMV_U_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(XHEMV_L_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(XHEMV_V_KERNEL ../generic/zhemv_k.c)
|
||||
SetFallback(XHEMV_M_KERNEL ../generic/zhemv_k.c)
|
||||
if (BUILD_BFLOAT16)
|
||||
set(SBGEMVNKERNEL ../x86_64/sbgemv_n.c)
|
||||
set(SBGEMVTKERNEL ../x86_64/sbgemv_t.c)
|
||||
set(SHGERKERNEL ../generic/ger.c)
|
||||
SetFallback(SBGEMVNKERNEL ../x86_64/sbgemv_n.c)
|
||||
SetFallback(SBGEMVTKERNEL ../x86_64/sbgemv_t.c)
|
||||
SetFallback(SHGERKERNEL ../generic/ger.c)
|
||||
endif ()
|
||||
endmacro ()
|
||||
|
||||
macro(SetDefaultL3)
|
||||
set(SGEADD_KERNEL ../generic/geadd.c)
|
||||
set(DGEADD_KERNEL ../generic/geadd.c)
|
||||
set(CGEADD_KERNEL ../generic/zgeadd.c)
|
||||
set(ZGEADD_KERNEL ../generic/zgeadd.c)
|
||||
SetFallback(SGEADD_KERNEL ../generic/geadd.c)
|
||||
SetFallback(DGEADD_KERNEL ../generic/geadd.c)
|
||||
SetFallback(CGEADD_KERNEL ../generic/zgeadd.c)
|
||||
SetFallback(ZGEADD_KERNEL ../generic/zgeadd.c)
|
||||
if (BUILD_BFLOAT16)
|
||||
set(SHGEADD_KERNEL ../generic/geadd.c)
|
||||
set(SBGEMMKERNEL ../generic/gemmkernel_2x2.c)
|
||||
set(SBGEMM_BETA ../generic/gemm_beta.c)
|
||||
set(SBGEMMINCOPY ../generic/gemm_ncopy_2.c)
|
||||
set(SBGEMMITCOPY ../generic/gemm_tcopy_2.c)
|
||||
set(SBGEMMONCOPY ../generic/gemm_ncopy_2.c)
|
||||
set(SBGEMMOTCOPY ../generic/gemm_tcopy_2.c)
|
||||
set(SBGEMMINCOPYOBJ sbgemm_incopy.o)
|
||||
set(SBGEMMITCOPYOBJ sbgemm_itcopy.o)
|
||||
set(SBGEMMONCOPYOBJ sbgemm_oncopy.o)
|
||||
set(SBGEMMOTCOPYOBJ sbgemm_otcopy.o)
|
||||
SetFallback(SHGEADD_KERNEL ../generic/geadd.c)
|
||||
SetFallback(SBGEMMKERNEL ../generic/gemmkernel_2x2.c)
|
||||
SetFallback(SBGEMM_BETA ../generic/gemm_beta.c)
|
||||
SetFallback(SBGEMMINCOPY ../generic/gemm_ncopy_2.c)
|
||||
SetFallback(SBGEMMITCOPY ../generic/gemm_tcopy_2.c)
|
||||
SetFallback(SBGEMMONCOPY ../generic/gemm_ncopy_2.c)
|
||||
SetFallback(SBGEMMOTCOPY ../generic/gemm_tcopy_2.c)
|
||||
SetFallback(SBGEMMINCOPYOBJ sbgemm_incopy.o)
|
||||
SetFallback(SBGEMMITCOPYOBJ sbgemm_itcopy.o)
|
||||
SetFallback(SBGEMMONCOPYOBJ sbgemm_oncopy.o)
|
||||
SetFallback(SBGEMMOTCOPYOBJ sbgemm_otcopy.o)
|
||||
endif ()
|
||||
|
||||
endmacro ()
|
||||
|
||||
+29
-1
@@ -416,7 +416,7 @@ endif ()
|
||||
set(ZGEMM_UNROLL_M 4)
|
||||
set(ZGEMM_UNROLL_N 4)
|
||||
set(SYMV_P 16)
|
||||
elseif ("${TCORE}" STREQUAL "VORTEX")
|
||||
elseif ("${TCORE}" STREQUAL "VORTEX")
|
||||
file(APPEND ${TARGET_CONF_TEMP}
|
||||
"#define ARMV8\n"
|
||||
"#define L1_CODE_SIZE\t32768\n"
|
||||
@@ -439,6 +439,34 @@ elseif ("${TCORE}" STREQUAL "VORTEX")
|
||||
set(ZGEMM_UNROLL_M 4)
|
||||
set(ZGEMM_UNROLL_N 4)
|
||||
set(SYMV_P 16)
|
||||
elseif ("${TCORE}" STREQUAL "P5600")
|
||||
file(APPEND ${TARGET_CONF_TEMP}
|
||||
"#define L2_SIZE 1048576\n"
|
||||
"#define DTB_SIZE 4096\n"
|
||||
"#define DTB_DEFAULT_ENTRIES 64\n")
|
||||
set(SGEMM_UNROLL_M 2)
|
||||
set(SGEMM_UNROLL_N 2)
|
||||
set(DGEMM_UNROLL_M 2)
|
||||
set(DGEMM_UNROLL_N 2)
|
||||
set(CGEMM_UNROLL_M 2)
|
||||
set(CGEMM_UNROLL_N 2)
|
||||
set(ZGEMM_UNROLL_M 2)
|
||||
set(ZGEMM_UNROLL_N 2)
|
||||
set(SYMV_P 16)
|
||||
elseif ("${TCORE}" MATCHES "MIPS")
|
||||
file(APPEND ${TARGET_CONF_TEMP}
|
||||
"#define L2_SIZE 262144\n"
|
||||
"#define DTB_SIZE 4096\n"
|
||||
"#define DTB_DEFAULT_ENTRIES 64\n")
|
||||
set(SGEMM_UNROLL_M 2)
|
||||
set(SGEMM_UNROLL_N 2)
|
||||
set(DGEMM_UNROLL_M 2)
|
||||
set(DGEMM_UNROLL_N 2)
|
||||
set(CGEMM_UNROLL_M 2)
|
||||
set(CGEMM_UNROLL_N 2)
|
||||
set(ZGEMM_UNROLL_M 2)
|
||||
set(ZGEMM_UNROLL_N 2)
|
||||
set(SYMV_P 16)
|
||||
elseif ("${TCORE}" STREQUAL "POWER6")
|
||||
file(APPEND ${TARGET_CONF_TEMP}
|
||||
"#define L1_DATA_SIZE 32768\n"
|
||||
|
||||
+69
-2
@@ -33,7 +33,7 @@ endif ()
|
||||
if (DEFINED BINARY AND DEFINED TARGET AND BINARY EQUAL 32)
|
||||
message(STATUS "Compiling a ${BINARY}-bit binary.")
|
||||
set(NO_AVX 1)
|
||||
if (${TARGET} STREQUAL "HASWELL" OR ${TARGET} STREQUAL "SANDYBRIDGE" OR ${TARGET} STREQUAL "SKYLAKEX" OR ${TARGET} STREQUAL "COOPERLAKE")
|
||||
if (${TARGET} STREQUAL "HASWELL" OR ${TARGET} STREQUAL "SANDYBRIDGE" OR ${TARGET} STREQUAL "SKYLAKEX" OR ${TARGET} STREQUAL "COOPERLAKE" OR ${TARGET} STREQUAL "SAPPHIRERAPIDS")
|
||||
set(TARGET "NEHALEM")
|
||||
endif ()
|
||||
if (${TARGET} STREQUAL "BULLDOZER" OR ${TARGET} STREQUAL "PILEDRIVER" OR ${TARGET} STREQUAL "ZEN")
|
||||
@@ -42,6 +42,9 @@ if (DEFINED BINARY AND DEFINED TARGET AND BINARY EQUAL 32)
|
||||
if (${TARGET} STREQUAL "ARMV8" OR ${TARGET} STREQUAL "CORTEXA57" OR ${TARGET} STREQUAL "CORTEXA53" OR ${TARGET} STREQUAL "CORTEXA55")
|
||||
set(TARGET "ARMV7")
|
||||
endif ()
|
||||
if (${TARGET} STREQUAL "POWER8" OR ${TARGET} STREQUAL "POWER9" OR ${TARGET} STREQUAL "POWER10")
|
||||
set(TARGET "POWER6")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
|
||||
@@ -102,6 +105,18 @@ if (CMAKE_C_COMPILER STREQUAL loongcc)
|
||||
set(GETARCH_FLAGS "${GETARCH_FLAGS} -static")
|
||||
endif ()
|
||||
|
||||
if (POWER)
|
||||
set(NO_WARMUP 1)
|
||||
set(HAVE_GAS 1)
|
||||
if (CMAKE_ASM_COMPILER_ID STREQUAL "GNU")
|
||||
set(HAVE_GAS 0)
|
||||
elseif (CMAKE_ASM_COMPILER_ID STREQUAL "Clang")
|
||||
set(CCOMMON_OPT "${CCOMMON_OPT} -fno-integrated-as")
|
||||
set(HAVE_GAS 0)
|
||||
endif ()
|
||||
set(GETARCH_FLAGS "${GETARCH_FLAGS} -DHAVE_GAS=${HAVE_GAS}")
|
||||
endif ()
|
||||
|
||||
#if don't use Fortran, it will only compile CBLAS.
|
||||
if (ONLY_CBLAS)
|
||||
set(NO_LAPACK 1)
|
||||
@@ -163,6 +178,22 @@ if (DEFINED TARGET)
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
if (${TARGET} STREQUAL SAPPHIRERAPIDS AND NOT NO_AVX512)
|
||||
if (${CMAKE_C_COMPILER_ID} STREQUAL "GNU")
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -dumpversion OUTPUT_VARIABLE GCC_VERSION)
|
||||
if (${CMAKE_C_COMPILER_VERSION} VERSION_GREATER 11.0)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=sapphirerapids")
|
||||
else()
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=skylake-avx512")
|
||||
endif()
|
||||
elseif (${CMAKE_C_COMPILER_ID} STREQUAL "Clang" OR ${CMAKE_C_COMPILER_ID} STREQUAL "AppleClang")
|
||||
if (${CMAKE_C_COMPILER_VERSION} VERSION_GREATER 12.0)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=sapphirerapids")
|
||||
else()
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=skylake-avx512")
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
if (${TARGET} STREQUAL SKYLAKEX AND NOT NO_AVX512)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=skylake-avx512")
|
||||
endif()
|
||||
@@ -206,6 +237,27 @@ if (DEFINED TARGET)
|
||||
if (DEFINED HAVE_SSE4_1)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -msse4.1")
|
||||
endif()
|
||||
|
||||
if (${TARGET} STREQUAL POWER10)
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -dumpversion OUTPUT_VARIABLE GCC_VERSION)
|
||||
if (${GCC_VERSION} VERSION_GREATER 10.2 OR ${GCC_VERSION} VERSION_EQUAL 10.2)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -mcpu=power10 -mtune=power10 -mvsx -fno-fast-math")
|
||||
else ()
|
||||
message(FATAL_ERROR "Compiler GCC.${GCC_VERSION} does not support Power10.")
|
||||
endif()
|
||||
endif()
|
||||
if (${TARGET} STREQUAL POWER9)
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -dumpversion OUTPUT_VARIABLE GCC_VERSION)
|
||||
if (${GCC_VERSION} VERSION_GREATER 5.0 OR ${GCC_VERSION} VERSION_EQUAL 5.0)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -mcpu=power9 -mtune=power9 -mvsx -fno-fast-math")
|
||||
else ()
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -mcpu=power8 -mtune=power8 -mvsx -fno-fast-math")
|
||||
message(WARNING "Compiler GCC.${GCC_VERSION} does not support fully Power9.")
|
||||
endif()
|
||||
endif()
|
||||
if (${TARGET} STREQUAL POWER8)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -mcpu=power8 -mtune=power8 -mvsx -fno-fast-math")
|
||||
endif()
|
||||
endif()
|
||||
if (DEFINED BINARY)
|
||||
message(STATUS "Compiling a ${BINARY}-bit binary.")
|
||||
@@ -223,6 +275,11 @@ include("${PROJECT_SOURCE_DIR}/cmake/arch.cmake")
|
||||
# C Compiler dependent settings
|
||||
include("${PROJECT_SOURCE_DIR}/cmake/cc.cmake")
|
||||
|
||||
if (INTERFACE64)
|
||||
set(SUFFIX64 64)
|
||||
set(SUFFIX64_UNDERSCORE _64)
|
||||
endif()
|
||||
|
||||
if (NOT NOFORTRAN)
|
||||
# Fortran Compiler dependent settings
|
||||
include("${PROJECT_SOURCE_DIR}/cmake/fc.cmake")
|
||||
@@ -258,8 +315,15 @@ if (NEED_PIC)
|
||||
endif()
|
||||
endif ()
|
||||
|
||||
if (X86_64 OR ${CORE} STREQUAL POWER10)
|
||||
set(SMALL_MATRIX_OPT TRUE)
|
||||
endif ()
|
||||
if (SMALL_MATRIX_OPT)
|
||||
set(CCOMMON_OPT "${CCOMMON_OPT} -DSMALL_MATRIX_OPT")
|
||||
endif ()
|
||||
|
||||
if (DYNAMIC_ARCH)
|
||||
if (X86 OR X86_64 OR ARM64 OR PPC)
|
||||
if (X86 OR X86_64 OR ARM64 OR POWER)
|
||||
set(CCOMMON_OPT "${CCOMMON_OPT} -DDYNAMIC_ARCH")
|
||||
if (DYNAMIC_OLDER)
|
||||
set(CCOMMON_OPT "${CCOMMON_OPT} -DDYNAMIC_OLDER")
|
||||
@@ -462,6 +526,9 @@ endif()
|
||||
if (BUILD_COMPLEX16)
|
||||
set(CMAKE_C_FLAGS "${CMAKE_C_FLAGS} -DBUILD_COMPLEX16")
|
||||
endif()
|
||||
if (BUILD_BFLOAT16)
|
||||
set(CMAKE_C_FLAGS "${CMAKE_C_FLAGS} -DBUILD_BFLOAT16")
|
||||
endif()
|
||||
if(NOT MSVC)
|
||||
set(CMAKE_ASM_FLAGS "${CMAKE_ASM_FLAGS} ${CCOMMON_OPT}")
|
||||
endif()
|
||||
|
||||
@@ -20,11 +20,11 @@ endif()
|
||||
|
||||
|
||||
|
||||
if(CMAKE_COMPILER_IS_GNUCC AND WIN32)
|
||||
if(MINGW)
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -dumpmachine
|
||||
OUTPUT_VARIABLE OPENBLAS_GCC_TARGET_MACHINE
|
||||
OUTPUT_VARIABLE OPENBLAS_MINGW_TARGET_MACHINE
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
if(OPENBLAS_GCC_TARGET_MACHINE MATCHES "amd64|x86_64|AMD64")
|
||||
if(OPENBLAS_MINGW_TARGET_MACHINE MATCHES "amd64|x86_64|AMD64")
|
||||
set(MINGW64 1)
|
||||
endif()
|
||||
endif()
|
||||
@@ -35,9 +35,11 @@ if(CMAKE_CL_64 OR MINGW64)
|
||||
elseif(MINGW OR (MSVC AND NOT CMAKE_CROSSCOMPILING))
|
||||
set(X86 1)
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "ppc.*|power.*|Power.*")
|
||||
set(PPC 1)
|
||||
set(POWER 1)
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "mips64.*")
|
||||
set(MIPS64 1)
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "loongarch64.*")
|
||||
set(LOONGARCH64 1)
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "amd64.*|x86_64.*|AMD64.*")
|
||||
if (NOT BINARY)
|
||||
if("${CMAKE_SIZEOF_VOID_P}" EQUAL "8")
|
||||
@@ -71,6 +73,8 @@ elseif (${CMAKE_CROSSCOMPILING})
|
||||
else ()
|
||||
set(X86 1)
|
||||
endif()
|
||||
elseif (${TARGET} STREQUAL "P5600" OR ${TARGET} MATCHES "MIPS.*")
|
||||
set(MIPS32 1)
|
||||
elseif (${TARGET} STREQUAL "ARMV7")
|
||||
set(ARM 1)
|
||||
else()
|
||||
@@ -84,8 +88,12 @@ if (X86_64)
|
||||
set(ARCH "x86_64")
|
||||
elseif(X86)
|
||||
set(ARCH "x86")
|
||||
elseif(PPC)
|
||||
elseif(POWER)
|
||||
set(ARCH "power")
|
||||
elseif(MIPS32)
|
||||
set(ARCH "mips")
|
||||
elseif(MIPS64)
|
||||
set(ARCH "mips64")
|
||||
elseif(ARM)
|
||||
set(ARCH "arm")
|
||||
elseif(ARM64)
|
||||
@@ -95,7 +103,7 @@ else()
|
||||
endif ()
|
||||
|
||||
if (NOT BINARY)
|
||||
if (X86_64 OR ARM64 OR PPC OR MIPS64)
|
||||
if (X86_64 OR ARM64 OR POWER OR MIPS64 OR LOONGARCH64)
|
||||
set(BINARY 64)
|
||||
else ()
|
||||
set(BINARY 32)
|
||||
|
||||
+159
-57
@@ -15,35 +15,83 @@ endfunction ()
|
||||
# Reads a Makefile into CMake vars.
|
||||
macro(ParseMakefileVars MAKEFILE_IN)
|
||||
message(STATUS "Reading vars from ${MAKEFILE_IN}...")
|
||||
set (IfElse 0)
|
||||
set (ElseSeen 0)
|
||||
set (C_COMPILER ${CMAKE_C_COMPILER_ID})
|
||||
set (IfElse 0)
|
||||
set (ElseSeen 0)
|
||||
set (SkipIfs 0)
|
||||
set (SkipElse 0)
|
||||
file(STRINGS ${MAKEFILE_IN} makefile_contents)
|
||||
foreach (makefile_line ${makefile_contents})
|
||||
#message(STATUS "parsing ${makefile_line}")
|
||||
if (${IfElse} GREATER 0)
|
||||
#message(STATUS "parsing ${makefile_line}")
|
||||
# Skip the entire scope of the else statement given that the if statement that precedes it has the valid condition.
|
||||
# The variable SkipIfs is used to identify which endif statement closes the scope of the else statement.
|
||||
if (${SkipElse} EQUAL 1)
|
||||
#message(STATUS "skipping ${makefile_line}")
|
||||
string(REGEX MATCH "(ifeq|ifneq|ifdef|ifndef) .*$" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
MATH(EXPR SkipIfs "${SkipIfs}+1")
|
||||
endif ()
|
||||
string(REGEX MATCH "endif[ \t]*" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
# message(STATUS "ENDIF ${makefile_line}")
|
||||
set (IfElse 0)
|
||||
set (ElseSeen 0)
|
||||
if (${SkipIfs} EQUAL 0)
|
||||
set (SkipElse 0)
|
||||
else ()
|
||||
MATH(EXPR SkipIfs "${SkipIfs}-1")
|
||||
endif ()
|
||||
endif ()
|
||||
continue ()
|
||||
endif ()
|
||||
# The variable IfElse is greater than 0 if and only if the previously parsed line is an if statement.
|
||||
if (${IfElse} GREATER 0)
|
||||
# If the current scope is the one that has to be skipped, the if/endif/else statements
|
||||
# along with it till the endif that closes the current scope have to be ignored as well.
|
||||
string(REGEX MATCH "(ifeq|ifneq|ifdef|ifndef) .*$" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
if ((${IfElse} EQUAL 2 AND ${ElseSeen} EQUAL 0) OR (${IfElse} EQUAL 1 AND ${ElseSeen} EQUAL 1))
|
||||
#message(STATUS "skipping ${makefile_line}")
|
||||
MATH(EXPR SkipIfs "${SkipIfs}+1")
|
||||
continue ()
|
||||
endif ()
|
||||
endif ()
|
||||
string(REGEX MATCH "endif[ \t]*" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
if (${SkipIfs} EQUAL 0)
|
||||
#message(STATUS "ENDIF ${makefile_line}")
|
||||
set (IfElse 0)
|
||||
set (ElseSeen 0)
|
||||
else ()
|
||||
#message(STATUS "skipping ${makefile_line}")
|
||||
MATH(EXPR SkipIfs "${SkipIfs}-1")
|
||||
endif ()
|
||||
continue ()
|
||||
endif ()
|
||||
string(REGEX MATCH "else[ \t]*" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
# message(STATUS "ELSE ${makefile_line}")
|
||||
set (ElseSeen 1)
|
||||
continue ()
|
||||
endif()
|
||||
if ( (${IfElse} EQUAL 2 AND ${ElseSeen} EQUAL 0) OR ( ${IfElse} EQUAL 1 AND ${ElseSeen} EQUAL 1))
|
||||
# message(STATUS "skipping ${makefile_line}")
|
||||
continue ()
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
if (${SkipIfs} EQUAL 0)
|
||||
#message(STATUS "ELSE ${makefile_line}")
|
||||
set (ElseSeen 1)
|
||||
else ()
|
||||
#message(STATUS "skipping ${makefile_line}")
|
||||
endif ()
|
||||
continue ()
|
||||
endif()
|
||||
# Skip the lines that are not part of the path that has to be taken.
|
||||
if ((${IfElse} EQUAL 2 AND ${ElseSeen} EQUAL 0) OR (${IfElse} EQUAL 1 AND ${ElseSeen} EQUAL 1) OR (${SkipIfs} GREATER 0))
|
||||
#message(STATUS "skipping ${makefile_line}")
|
||||
continue ()
|
||||
endif ()
|
||||
endif ()
|
||||
endif ()
|
||||
# Skip commented lines (the ones that start with '#')
|
||||
string(REGEX MATCH "[ \t]*\\#.*$" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
#message(STATUS "skipping ${makefile_line}")
|
||||
continue ()
|
||||
endif ()
|
||||
string(REGEX MATCH "([0-9_a-zA-Z]+)[ \t]*=[ \t]*(.+)$" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
#message(STATUS "match on ${line_match}")
|
||||
#message(STATUS "match on ${line_match}")
|
||||
set(var_name ${CMAKE_MATCH_1})
|
||||
# set(var_value ${CMAKE_MATCH_2})
|
||||
#set(var_value ${CMAKE_MATCH_2})
|
||||
string(STRIP ${CMAKE_MATCH_2} var_value)
|
||||
# check for Makefile variables in the string, e.g. $(TSUFFIX)
|
||||
string(REGEX MATCHALL "\\$\\(([0-9_a-zA-Z]+)\\)" make_var_matches ${var_value})
|
||||
@@ -54,39 +102,93 @@ macro(ParseMakefileVars MAKEFILE_IN)
|
||||
string(REPLACE "$(${make_var})" "${${make_var}}" var_value ${var_value})
|
||||
endforeach ()
|
||||
set(${var_name} ${var_value})
|
||||
else ()
|
||||
string(REGEX MATCH "include \\$\\(KERNELDIR\\)/(.+)$" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
#message(STATUS "match on include ${line_match}")
|
||||
ParseMakefileVars(${KERNELDIR}/${CMAKE_MATCH_1})
|
||||
else ()
|
||||
# message(STATUS "unmatched line ${line_match}")
|
||||
string(REGEX MATCH "ifeq \\(\\$\\(([_A-Z]+)\\),[ \t]*([0-9_A-Z]+)\\)" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
# message(STATUS "IFEQ: ${line_match} first: ${CMAKE_MATCH_1} second: ${CMAKE_MATCH_2}")
|
||||
if (DEFINED ${${CMAKE_MATCH_1}} AND ${${CMAKE_MATCH_1}} STREQUAL ${CMAKE_MATCH_2})
|
||||
# message (STATUS "condition is true")
|
||||
set (IfElse 1)
|
||||
else ()
|
||||
set (IfElse 2)
|
||||
endif ()
|
||||
continue ()
|
||||
endif ()
|
||||
# Include a new file to be parsed
|
||||
string(REGEX MATCH "include \\$\\(KERNELDIR\\)/(.+)$" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
#message(STATUS "match on include ${line_match}")
|
||||
ParseMakefileVars(${KERNELDIR}/${CMAKE_MATCH_1})
|
||||
continue ()
|
||||
endif ()
|
||||
# The if statement that precedes this else has the path taken
|
||||
# Thus, this else statement has to be skipped.
|
||||
string(REGEX MATCH "else[ \t]*" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
#message(STATUS "skipping ${makefile_line}")
|
||||
set (SkipElse 1)
|
||||
continue()
|
||||
endif()
|
||||
# Example 1: ifdef HAVE_MSA
|
||||
# Example 2: ifndef ZNRM2KERNEL
|
||||
string(REGEX MATCH "(ifdef|ifndef) ([0-9_A-Z]+)" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
#message(STATUS "${CMAKE_MATCH_1} first: ${CMAKE_MATCH_2}")
|
||||
set (ElseSeen 0)
|
||||
if (DEFINED ${CMAKE_MATCH_2})
|
||||
if (${CMAKE_MATCH_1} STREQUAL "ifdef")
|
||||
#message (STATUS "condition is true")
|
||||
set (IfElse 1)
|
||||
else ()
|
||||
string(REGEX MATCH "ifneq \\(\\$\\(([_A-Z]+)\\),[ \t]*([0-9_A-Z]+)\\)" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
# message(STATUS "IFNEQ: ${line_match} first: ${CMAKE_MATCH_1} second: ${CMAKE_MATCH_2}")
|
||||
if ( ${CMAKE_MATCH_1} STREQUAL C_COMPILER)
|
||||
set (CMAKE_MATCH_1 CMAKE_C_COMPILER)
|
||||
endif ()
|
||||
if (NOT ( ${${CMAKE_MATCH_1}} STREQUAL ${CMAKE_MATCH_2}))
|
||||
# message (STATUS "condition is true")
|
||||
set (IfElse 1)
|
||||
else ()
|
||||
set (IfElse 2)
|
||||
endif ()
|
||||
endif ()
|
||||
set (IfElse 2)
|
||||
endif ()
|
||||
else ()
|
||||
if (${CMAKE_MATCH_1} STREQUAL "ifdef")
|
||||
set (IfElse 2)
|
||||
else ()
|
||||
#message (STATUS "condition is true")
|
||||
set (IfElse 1)
|
||||
endif ()
|
||||
endif ()
|
||||
continue ()
|
||||
endif ()
|
||||
# Example 1: ifeq ($(SGEMM_UNROLL_M), 16)
|
||||
# Example 2: ifeq ($(SGEMM_UNROLL_M)x$(SGEMM_UNROLL_N), 8x8)
|
||||
# Example 3: ifeq ($(__BYTE_ORDER__)$(ELF_VERSION),__ORDER_BIG_ENDIAN__2)
|
||||
# Ignore the second group since (?:...) does not work on cmake
|
||||
string(REGEX MATCH "ifeq \\(\\$\\(([0-9_A-Z]+)\\)(([0-9_A-Za-z]*)\\$\\(([0-9_A-Z]+)\\))?,[ \t]*([0-9_A-Za-z]+)\\)" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
#message(STATUS "IFEQ: ${line_match} first: ${CMAKE_MATCH_1} second: ${CMAKE_MATCH_3} third: ${CMAKE_MATCH_4} fourth: ${CMAKE_MATCH_5}")
|
||||
if (DEFINED ${CMAKE_MATCH_1})
|
||||
if (DEFINED ${CMAKE_MATCH_4})
|
||||
set (STR ${${CMAKE_MATCH_1}}${CMAKE_MATCH_3}${${CMAKE_MATCH_4}})
|
||||
else ()
|
||||
set (STR ${${CMAKE_MATCH_1}})
|
||||
endif ()
|
||||
if (${STR} STREQUAL ${CMAKE_MATCH_5})
|
||||
#message (STATUS "condition is true")
|
||||
set (IfElse 1)
|
||||
continue ()
|
||||
endif ()
|
||||
endif ()
|
||||
set (IfElse 2)
|
||||
continue ()
|
||||
endif ()
|
||||
# Example 1 (Group 3): ifneq ($(SGEMM_UNROLL_M), $(SGEMM_UNROLL_N))
|
||||
# Example 2 (Group 4): ifneq ($(C_COMPILER), PGI)
|
||||
string(REGEX MATCH "ifneq \\(\\$\\(([0-9_A-Z]+)\\),[ \t]*(\\$\\(([0-9_A-Z]+)\\)|([0-9_A-Z]+))\\)" line_match "${makefile_line}")
|
||||
if (NOT "${line_match}" STREQUAL "")
|
||||
#message(STATUS "IFNEQ: ${line_match} first: ${CMAKE_MATCH_1} second: ${CMAKE_MATCH_3} third: ${CMAKE_MATCH_4}")
|
||||
set (ElseSeen 0)
|
||||
set (HasValidGroup 0)
|
||||
if (DEFINED ${CMAKE_MATCH_3})
|
||||
set (HasValidGroup 1)
|
||||
set (STR ${${CMAKE_MATCH_3}})
|
||||
elseif (NOT ${CMAKE_MATCH_4} STREQUAL "")
|
||||
set (HasValidGroup 1)
|
||||
set (STR ${CMAKE_MATCH_4})
|
||||
endif ()
|
||||
if (DEFINED ${CMAKE_MATCH_1} AND ${HasValidGroup} EQUAL 1)
|
||||
if (NOT (${${CMAKE_MATCH_1}} STREQUAL ${STR}))
|
||||
#message (STATUS "condition is true")
|
||||
set (IfElse 1)
|
||||
continue ()
|
||||
endif ()
|
||||
endif ()
|
||||
set (IfElse 2)
|
||||
continue ()
|
||||
endif ()
|
||||
#message(STATUS "unmatched line ${line_match}")
|
||||
endforeach ()
|
||||
endmacro ()
|
||||
|
||||
@@ -157,31 +259,31 @@ endfunction ()
|
||||
# STRING - compiles only the given type (e.g. DOUBLE)
|
||||
function(GenerateNamedObjects sources_in)
|
||||
|
||||
if (DEFINED ARGV1)
|
||||
if (${ARGC} GREATER 1)
|
||||
set(defines_in ${ARGV1})
|
||||
endif ()
|
||||
|
||||
if (DEFINED ARGV2 AND NOT "${ARGV2}" STREQUAL "")
|
||||
if (${ARGC} GREATER 2 AND NOT "${ARGV2}" STREQUAL "")
|
||||
set(name_in ${ARGV2})
|
||||
# strip off extension for kernel files that pass in the object name.
|
||||
get_filename_component(name_in ${name_in} NAME_WE)
|
||||
endif ()
|
||||
|
||||
if (DEFINED ARGV3)
|
||||
if (${ARGC} GREATER 3)
|
||||
set(use_cblas ${ARGV3})
|
||||
else ()
|
||||
set(use_cblas false)
|
||||
endif ()
|
||||
|
||||
if (DEFINED ARGV4)
|
||||
if (${ARGC} GREATER 4)
|
||||
set(replace_last_with ${ARGV4})
|
||||
endif ()
|
||||
|
||||
if (DEFINED ARGV5)
|
||||
if (${ARGC} GREATER 5)
|
||||
set(append_with ${ARGV5})
|
||||
endif ()
|
||||
|
||||
if (DEFINED ARGV6)
|
||||
if (${ARGC} GREATER 6)
|
||||
set(no_float_type ${ARGV6})
|
||||
else ()
|
||||
set(no_float_type false)
|
||||
@@ -196,7 +298,7 @@ function(GenerateNamedObjects sources_in)
|
||||
set(real_only false)
|
||||
set(complex_only false)
|
||||
set(mangle_complex_sources false)
|
||||
if (DEFINED ARGV7 AND NOT "${ARGV7}" STREQUAL "")
|
||||
if (${ARGC} GREATER 7 AND NOT "${ARGV7}" STREQUAL "")
|
||||
if (${ARGV7} EQUAL 1)
|
||||
set(real_only true)
|
||||
elseif (${ARGV7} EQUAL 2)
|
||||
@@ -342,17 +444,17 @@ endfunction ()
|
||||
function(GenerateCombinationObjects sources_in defines_in absent_codes_in all_defines_in replace_scheme)
|
||||
|
||||
set(alternate_name_in "")
|
||||
if (DEFINED ARGV5)
|
||||
if (${ARGC} GREATER 5)
|
||||
set(alternate_name_in ${ARGV5})
|
||||
endif ()
|
||||
|
||||
set(no_float_type false)
|
||||
if (DEFINED ARGV6)
|
||||
if (${ARGC} GREATER 6)
|
||||
set(no_float_type ${ARGV6})
|
||||
endif ()
|
||||
|
||||
set(complex_filename_scheme "")
|
||||
if (DEFINED ARGV7)
|
||||
if (${ARGC} GREATER 7)
|
||||
set(complex_filename_scheme ${ARGV7})
|
||||
endif ()
|
||||
|
||||
|
||||
+1
-1
@@ -120,7 +120,7 @@ static inline int blas_quickdivide(blasint x, blasint y){
|
||||
.text ;
|
||||
.p2align 2 ;
|
||||
.global REALNAME ;
|
||||
#ifndef __APPLE__
|
||||
#if !defined(__APPLE__) && !defined(_WIN32)
|
||||
.type REALNAME, %function ;
|
||||
#endif
|
||||
REALNAME:
|
||||
|
||||
+45
@@ -232,6 +232,8 @@
|
||||
|
||||
#define CGEADD_K cgeadd_k
|
||||
|
||||
#define CGEMM_SMALL_MATRIX_PERMIT cgemm_small_matrix_permit
|
||||
|
||||
#else
|
||||
|
||||
#define CAMAX_K gotoblas -> camax_k
|
||||
@@ -426,8 +428,51 @@
|
||||
|
||||
#define CGEADD_K gotoblas -> cgeadd_k
|
||||
|
||||
#define CGEMM_SMALL_MATRIX_PERMIT gotoblas -> cgemm_small_matrix_permit
|
||||
|
||||
#endif
|
||||
|
||||
#define CGEMM_SMALL_KERNEL_NN FUNC_OFFSET(cgemm_small_kernel_nn)
|
||||
#define CGEMM_SMALL_KERNEL_NT FUNC_OFFSET(cgemm_small_kernel_nt)
|
||||
#define CGEMM_SMALL_KERNEL_NR FUNC_OFFSET(cgemm_small_kernel_nr)
|
||||
#define CGEMM_SMALL_KERNEL_NC FUNC_OFFSET(cgemm_small_kernel_nc)
|
||||
|
||||
#define CGEMM_SMALL_KERNEL_TN FUNC_OFFSET(cgemm_small_kernel_tn)
|
||||
#define CGEMM_SMALL_KERNEL_TT FUNC_OFFSET(cgemm_small_kernel_tt)
|
||||
#define CGEMM_SMALL_KERNEL_TR FUNC_OFFSET(cgemm_small_kernel_tr)
|
||||
#define CGEMM_SMALL_KERNEL_TC FUNC_OFFSET(cgemm_small_kernel_tc)
|
||||
|
||||
#define CGEMM_SMALL_KERNEL_RN FUNC_OFFSET(cgemm_small_kernel_rn)
|
||||
#define CGEMM_SMALL_KERNEL_RT FUNC_OFFSET(cgemm_small_kernel_rt)
|
||||
#define CGEMM_SMALL_KERNEL_RR FUNC_OFFSET(cgemm_small_kernel_rr)
|
||||
#define CGEMM_SMALL_KERNEL_RC FUNC_OFFSET(cgemm_small_kernel_rc)
|
||||
|
||||
#define CGEMM_SMALL_KERNEL_CN FUNC_OFFSET(cgemm_small_kernel_cn)
|
||||
#define CGEMM_SMALL_KERNEL_CT FUNC_OFFSET(cgemm_small_kernel_ct)
|
||||
#define CGEMM_SMALL_KERNEL_CR FUNC_OFFSET(cgemm_small_kernel_cr)
|
||||
#define CGEMM_SMALL_KERNEL_CC FUNC_OFFSET(cgemm_small_kernel_cc)
|
||||
|
||||
#define CGEMM_SMALL_KERNEL_B0_NN FUNC_OFFSET(cgemm_small_kernel_b0_nn)
|
||||
#define CGEMM_SMALL_KERNEL_B0_NT FUNC_OFFSET(cgemm_small_kernel_b0_nt)
|
||||
#define CGEMM_SMALL_KERNEL_B0_NR FUNC_OFFSET(cgemm_small_kernel_b0_nr)
|
||||
#define CGEMM_SMALL_KERNEL_B0_NC FUNC_OFFSET(cgemm_small_kernel_b0_nc)
|
||||
|
||||
#define CGEMM_SMALL_KERNEL_B0_TN FUNC_OFFSET(cgemm_small_kernel_b0_tn)
|
||||
#define CGEMM_SMALL_KERNEL_B0_TT FUNC_OFFSET(cgemm_small_kernel_b0_tt)
|
||||
#define CGEMM_SMALL_KERNEL_B0_TR FUNC_OFFSET(cgemm_small_kernel_b0_tr)
|
||||
#define CGEMM_SMALL_KERNEL_B0_TC FUNC_OFFSET(cgemm_small_kernel_b0_tc)
|
||||
|
||||
#define CGEMM_SMALL_KERNEL_B0_RN FUNC_OFFSET(cgemm_small_kernel_b0_rn)
|
||||
#define CGEMM_SMALL_KERNEL_B0_RT FUNC_OFFSET(cgemm_small_kernel_b0_rt)
|
||||
#define CGEMM_SMALL_KERNEL_B0_RR FUNC_OFFSET(cgemm_small_kernel_b0_rr)
|
||||
#define CGEMM_SMALL_KERNEL_B0_RC FUNC_OFFSET(cgemm_small_kernel_b0_rc)
|
||||
|
||||
#define CGEMM_SMALL_KERNEL_B0_CN FUNC_OFFSET(cgemm_small_kernel_b0_cn)
|
||||
#define CGEMM_SMALL_KERNEL_B0_CT FUNC_OFFSET(cgemm_small_kernel_b0_ct)
|
||||
#define CGEMM_SMALL_KERNEL_B0_CR FUNC_OFFSET(cgemm_small_kernel_b0_cr)
|
||||
#define CGEMM_SMALL_KERNEL_B0_CC FUNC_OFFSET(cgemm_small_kernel_b0_cc)
|
||||
|
||||
|
||||
#define CGEMM_NN cgemm_nn
|
||||
#define CGEMM_CN cgemm_cn
|
||||
#define CGEMM_TN cgemm_tn
|
||||
|
||||
+15
@@ -157,6 +157,8 @@
|
||||
#define DIMATCOPY_K_RT dimatcopy_k_rt
|
||||
#define DGEADD_K dgeadd_k
|
||||
|
||||
#define DGEMM_SMALL_MATRIX_PERMIT dgemm_small_matrix_permit
|
||||
|
||||
#else
|
||||
|
||||
#define DAMAX_K gotoblas -> damax_k
|
||||
@@ -281,8 +283,21 @@
|
||||
|
||||
#define DGEADD_K gotoblas -> dgeadd_k
|
||||
|
||||
#define DGEMM_SMALL_MATRIX_PERMIT gotoblas -> dgemm_small_matrix_permit
|
||||
|
||||
#endif
|
||||
|
||||
#define DGEMM_SMALL_KERNEL_NN FUNC_OFFSET(dgemm_small_kernel_nn)
|
||||
#define DGEMM_SMALL_KERNEL_NT FUNC_OFFSET(dgemm_small_kernel_nt)
|
||||
#define DGEMM_SMALL_KERNEL_TN FUNC_OFFSET(dgemm_small_kernel_tn)
|
||||
#define DGEMM_SMALL_KERNEL_TT FUNC_OFFSET(dgemm_small_kernel_tt)
|
||||
|
||||
#define DGEMM_SMALL_KERNEL_B0_NN FUNC_OFFSET(dgemm_small_kernel_b0_nn)
|
||||
#define DGEMM_SMALL_KERNEL_B0_NT FUNC_OFFSET(dgemm_small_kernel_b0_nt)
|
||||
#define DGEMM_SMALL_KERNEL_B0_TN FUNC_OFFSET(dgemm_small_kernel_b0_tn)
|
||||
#define DGEMM_SMALL_KERNEL_B0_TT FUNC_OFFSET(dgemm_small_kernel_b0_tt)
|
||||
|
||||
|
||||
#define DGEMM_NN dgemm_nn
|
||||
#define DGEMM_CN dgemm_tn
|
||||
#define DGEMM_TN dgemm_tn
|
||||
|
||||
+123
@@ -515,6 +515,129 @@ int qgemm_kernel(BLASLONG, BLASLONG, BLASLONG, xidouble *, xidouble *, xidouble
|
||||
int qgemm_kernel(BLASLONG, BLASLONG, BLASLONG, xdouble, xdouble *, xdouble *, xdouble *, BLASLONG);
|
||||
#endif
|
||||
|
||||
#ifdef SMALL_MATRIX_OPT
|
||||
int sbgemm_small_matrix_permit(int transa, int transb, BLASLONG m, BLASLONG n, BLASLONG k, float alpha, float beta);
|
||||
|
||||
int sbgemm_small_kernel_nn(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int sbgemm_small_kernel_nt(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int sbgemm_small_kernel_tn(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int sbgemm_small_kernel_tt(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
|
||||
int sgemm_small_matrix_permit(int transa, int transb, BLASLONG m, BLASLONG n, BLASLONG k, float alpha, float beta);
|
||||
|
||||
int sgemm_small_kernel_nn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int sgemm_small_kernel_nt(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int sgemm_small_kernel_tn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int sgemm_small_kernel_tt(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
|
||||
int dgemm_small_matrix_permit(int transa, int transb, BLASLONG m, BLASLONG n, BLASLONG k, double alpha, double beta);
|
||||
|
||||
int dgemm_small_kernel_nn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double beta, double * C, BLASLONG ldc);
|
||||
int dgemm_small_kernel_nt(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double beta, double * C, BLASLONG ldc);
|
||||
int dgemm_small_kernel_tn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double beta, double * C, BLASLONG ldc);
|
||||
int dgemm_small_kernel_tt(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double beta, double * C, BLASLONG ldc);
|
||||
|
||||
int sbgemm_small_kernel_b0_nn(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int sbgemm_small_kernel_b0_nt(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int sbgemm_small_kernel_b0_tn(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int sbgemm_small_kernel_b0_tt(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
|
||||
int sgemm_small_kernel_b0_nn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int sgemm_small_kernel_b0_nt(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int sgemm_small_kernel_b0_tn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int sgemm_small_kernel_b0_tt(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
|
||||
int dgemm_small_kernel_b0_nn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int dgemm_small_kernel_b0_nt(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int dgemm_small_kernel_b0_tn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int dgemm_small_kernel_b0_tt(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
|
||||
int cgemm_small_matrix_permit(int transa, int transb, BLASLONG m, BLASLONG n, BLASLONG k, float alpha0, float alpha1, float beta0, float beta1);
|
||||
|
||||
int cgemm_small_kernel_nn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_nt(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_nr(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_nc(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
|
||||
int cgemm_small_kernel_tn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_tt(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_tr(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_tc(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
|
||||
int cgemm_small_kernel_rn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_rt(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_rr(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_rc(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
|
||||
int cgemm_small_kernel_cn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_ct(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_cr(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_cc(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
|
||||
int zgemm_small_matrix_permit(int transa, int transb, BLASLONG m, BLASLONG n, BLASLONG k, double alpha0, double alpha1, double beta0, double beta1);
|
||||
|
||||
int zgemm_small_kernel_nn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_nt(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_nr(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_nc(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
|
||||
int zgemm_small_kernel_tn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_tt(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_tr(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_tc(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
|
||||
int zgemm_small_kernel_rn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_rt(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_rr(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_rc(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
|
||||
int zgemm_small_kernel_cn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_ct(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_cr(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_cc(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
|
||||
int cgemm_small_kernel_b0_nn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_nt(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_nr(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_nc(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
|
||||
int cgemm_small_kernel_b0_tn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_tt(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_tr(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_tc(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
|
||||
int cgemm_small_kernel_b0_rn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_rt(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_rr(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_rc(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
|
||||
int cgemm_small_kernel_b0_cn(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_ct(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_cr(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int cgemm_small_kernel_b0_cc(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
|
||||
int zgemm_small_kernel_b0_nn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_nt(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_nr(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_nc(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
|
||||
int zgemm_small_kernel_b0_tn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_tt(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_tr(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_tc(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
|
||||
int zgemm_small_kernel_b0_rn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_rt(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_rr(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_rc(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
|
||||
int zgemm_small_kernel_b0_cn(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_ct(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_cr(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int zgemm_small_kernel_b0_cc(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
|
||||
#endif
|
||||
|
||||
int cgemm_kernel_n(BLASLONG, BLASLONG, BLASLONG, float, float, float *, float *, float *, BLASLONG);
|
||||
int cgemm_kernel_l(BLASLONG, BLASLONG, BLASLONG, float, float, float *, float *, float *, BLASLONG);
|
||||
int cgemm_kernel_r(BLASLONG, BLASLONG, BLASLONG, float, float, float *, float *, float *, BLASLONG);
|
||||
|
||||
@@ -186,7 +186,7 @@ REALNAME: ;\
|
||||
|
||||
#define BUFFER_SIZE ( 32 << 20)
|
||||
|
||||
#define PAGESIZE (16UL << 1)
|
||||
#define PAGESIZE (16UL << 10)
|
||||
#define FIXED_PAGESIZE (16UL << 10)
|
||||
#define HUGE_PAGESIZE ( 2 << 20)
|
||||
|
||||
|
||||
+120
@@ -644,6 +644,17 @@
|
||||
|
||||
#define GEADD_K DGEADD_K
|
||||
|
||||
#define GEMM_SMALL_MATRIX_PERMIT DGEMM_SMALL_MATRIX_PERMIT
|
||||
|
||||
#define GEMM_SMALL_KERNEL_NN DGEMM_SMALL_KERNEL_NN
|
||||
#define GEMM_SMALL_KERNEL_NT DGEMM_SMALL_KERNEL_NT
|
||||
#define GEMM_SMALL_KERNEL_TN DGEMM_SMALL_KERNEL_TN
|
||||
#define GEMM_SMALL_KERNEL_TT DGEMM_SMALL_KERNEL_TT
|
||||
#define GEMM_SMALL_KERNEL_B0_NN DGEMM_SMALL_KERNEL_B0_NN
|
||||
#define GEMM_SMALL_KERNEL_B0_NT DGEMM_SMALL_KERNEL_B0_NT
|
||||
#define GEMM_SMALL_KERNEL_B0_TN DGEMM_SMALL_KERNEL_B0_TN
|
||||
#define GEMM_SMALL_KERNEL_B0_TT DGEMM_SMALL_KERNEL_B0_TT
|
||||
|
||||
#elif defined(BFLOAT16)
|
||||
|
||||
#define D_TO_BF16_K SBDTOBF16_K
|
||||
@@ -931,6 +942,18 @@
|
||||
|
||||
#define GEADD_K SGEADD_K
|
||||
|
||||
#define GEMM_SMALL_MATRIX_PERMIT SBGEMM_SMALL_MATRIX_PERMIT
|
||||
|
||||
#define GEMM_SMALL_KERNEL_NN SBGEMM_SMALL_KERNEL_NN
|
||||
#define GEMM_SMALL_KERNEL_NT SBGEMM_SMALL_KERNEL_NT
|
||||
#define GEMM_SMALL_KERNEL_TN SBGEMM_SMALL_KERNEL_TN
|
||||
#define GEMM_SMALL_KERNEL_TT SBGEMM_SMALL_KERNEL_TT
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0_NN SBGEMM_SMALL_KERNEL_B0_NN
|
||||
#define GEMM_SMALL_KERNEL_B0_NT SBGEMM_SMALL_KERNEL_B0_NT
|
||||
#define GEMM_SMALL_KERNEL_B0_TN SBGEMM_SMALL_KERNEL_B0_TN
|
||||
#define GEMM_SMALL_KERNEL_B0_TT SBGEMM_SMALL_KERNEL_B0_TT
|
||||
|
||||
#endif
|
||||
|
||||
#else
|
||||
@@ -1236,6 +1259,19 @@
|
||||
#define IMATCOPY_K_RT SIMATCOPY_K_RT
|
||||
|
||||
#define GEADD_K SGEADD_K
|
||||
|
||||
#define GEMM_SMALL_MATRIX_PERMIT SGEMM_SMALL_MATRIX_PERMIT
|
||||
|
||||
#define GEMM_SMALL_KERNEL_NN SGEMM_SMALL_KERNEL_NN
|
||||
#define GEMM_SMALL_KERNEL_NT SGEMM_SMALL_KERNEL_NT
|
||||
#define GEMM_SMALL_KERNEL_TN SGEMM_SMALL_KERNEL_TN
|
||||
#define GEMM_SMALL_KERNEL_TT SGEMM_SMALL_KERNEL_TT
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0_NN SGEMM_SMALL_KERNEL_B0_NN
|
||||
#define GEMM_SMALL_KERNEL_B0_NT SGEMM_SMALL_KERNEL_B0_NT
|
||||
#define GEMM_SMALL_KERNEL_B0_TN SGEMM_SMALL_KERNEL_B0_TN
|
||||
#define GEMM_SMALL_KERNEL_B0_TT SGEMM_SMALL_KERNEL_B0_TT
|
||||
|
||||
#endif
|
||||
#else
|
||||
#ifdef XDOUBLE
|
||||
@@ -2063,6 +2099,48 @@
|
||||
|
||||
#define GEADD_K ZGEADD_K
|
||||
|
||||
#define GEMM_SMALL_MATRIX_PERMIT ZGEMM_SMALL_MATRIX_PERMIT
|
||||
|
||||
#define GEMM_SMALL_KERNEL_NN ZGEMM_SMALL_KERNEL_NN
|
||||
#define GEMM_SMALL_KERNEL_NT ZGEMM_SMALL_KERNEL_NT
|
||||
#define GEMM_SMALL_KERNEL_NR ZGEMM_SMALL_KERNEL_NR
|
||||
#define GEMM_SMALL_KERNEL_NC ZGEMM_SMALL_KERNEL_NC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_TN ZGEMM_SMALL_KERNEL_TN
|
||||
#define GEMM_SMALL_KERNEL_TT ZGEMM_SMALL_KERNEL_TT
|
||||
#define GEMM_SMALL_KERNEL_TR ZGEMM_SMALL_KERNEL_TR
|
||||
#define GEMM_SMALL_KERNEL_TC ZGEMM_SMALL_KERNEL_TC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_RN ZGEMM_SMALL_KERNEL_RN
|
||||
#define GEMM_SMALL_KERNEL_RT ZGEMM_SMALL_KERNEL_RT
|
||||
#define GEMM_SMALL_KERNEL_RR ZGEMM_SMALL_KERNEL_RR
|
||||
#define GEMM_SMALL_KERNEL_RC ZGEMM_SMALL_KERNEL_RC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_CN ZGEMM_SMALL_KERNEL_CN
|
||||
#define GEMM_SMALL_KERNEL_CT ZGEMM_SMALL_KERNEL_CT
|
||||
#define GEMM_SMALL_KERNEL_CR ZGEMM_SMALL_KERNEL_CR
|
||||
#define GEMM_SMALL_KERNEL_CC ZGEMM_SMALL_KERNEL_CC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0_NN ZGEMM_SMALL_KERNEL_B0_NN
|
||||
#define GEMM_SMALL_KERNEL_B0_NT ZGEMM_SMALL_KERNEL_B0_NT
|
||||
#define GEMM_SMALL_KERNEL_B0_NR ZGEMM_SMALL_KERNEL_B0_NR
|
||||
#define GEMM_SMALL_KERNEL_B0_NC ZGEMM_SMALL_KERNEL_B0_NC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0_TN ZGEMM_SMALL_KERNEL_B0_TN
|
||||
#define GEMM_SMALL_KERNEL_B0_TT ZGEMM_SMALL_KERNEL_B0_TT
|
||||
#define GEMM_SMALL_KERNEL_B0_TR ZGEMM_SMALL_KERNEL_B0_TR
|
||||
#define GEMM_SMALL_KERNEL_B0_TC ZGEMM_SMALL_KERNEL_B0_TC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0_RN ZGEMM_SMALL_KERNEL_B0_RN
|
||||
#define GEMM_SMALL_KERNEL_B0_RT ZGEMM_SMALL_KERNEL_B0_RT
|
||||
#define GEMM_SMALL_KERNEL_B0_RR ZGEMM_SMALL_KERNEL_B0_RR
|
||||
#define GEMM_SMALL_KERNEL_B0_RC ZGEMM_SMALL_KERNEL_B0_RC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0_CN ZGEMM_SMALL_KERNEL_B0_CN
|
||||
#define GEMM_SMALL_KERNEL_B0_CT ZGEMM_SMALL_KERNEL_B0_CT
|
||||
#define GEMM_SMALL_KERNEL_B0_CR ZGEMM_SMALL_KERNEL_B0_CR
|
||||
#define GEMM_SMALL_KERNEL_B0_CC ZGEMM_SMALL_KERNEL_B0_CC
|
||||
|
||||
#else
|
||||
|
||||
#define AMAX_K CAMAX_K
|
||||
@@ -2486,6 +2564,48 @@
|
||||
|
||||
#define GEADD_K CGEADD_K
|
||||
|
||||
#define GEMM_SMALL_MATRIX_PERMIT CGEMM_SMALL_MATRIX_PERMIT
|
||||
|
||||
#define GEMM_SMALL_KERNEL_NN CGEMM_SMALL_KERNEL_NN
|
||||
#define GEMM_SMALL_KERNEL_NT CGEMM_SMALL_KERNEL_NT
|
||||
#define GEMM_SMALL_KERNEL_NR CGEMM_SMALL_KERNEL_NR
|
||||
#define GEMM_SMALL_KERNEL_NC CGEMM_SMALL_KERNEL_NC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_TN CGEMM_SMALL_KERNEL_TN
|
||||
#define GEMM_SMALL_KERNEL_TT CGEMM_SMALL_KERNEL_TT
|
||||
#define GEMM_SMALL_KERNEL_TR CGEMM_SMALL_KERNEL_TR
|
||||
#define GEMM_SMALL_KERNEL_TC CGEMM_SMALL_KERNEL_TC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_RN CGEMM_SMALL_KERNEL_RN
|
||||
#define GEMM_SMALL_KERNEL_RT CGEMM_SMALL_KERNEL_RT
|
||||
#define GEMM_SMALL_KERNEL_RR CGEMM_SMALL_KERNEL_RR
|
||||
#define GEMM_SMALL_KERNEL_RC CGEMM_SMALL_KERNEL_RC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_CN CGEMM_SMALL_KERNEL_CN
|
||||
#define GEMM_SMALL_KERNEL_CT CGEMM_SMALL_KERNEL_CT
|
||||
#define GEMM_SMALL_KERNEL_CR CGEMM_SMALL_KERNEL_CR
|
||||
#define GEMM_SMALL_KERNEL_CC CGEMM_SMALL_KERNEL_CC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0_NN CGEMM_SMALL_KERNEL_B0_NN
|
||||
#define GEMM_SMALL_KERNEL_B0_NT CGEMM_SMALL_KERNEL_B0_NT
|
||||
#define GEMM_SMALL_KERNEL_B0_NR CGEMM_SMALL_KERNEL_B0_NR
|
||||
#define GEMM_SMALL_KERNEL_B0_NC CGEMM_SMALL_KERNEL_B0_NC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0_TN CGEMM_SMALL_KERNEL_B0_TN
|
||||
#define GEMM_SMALL_KERNEL_B0_TT CGEMM_SMALL_KERNEL_B0_TT
|
||||
#define GEMM_SMALL_KERNEL_B0_TR CGEMM_SMALL_KERNEL_B0_TR
|
||||
#define GEMM_SMALL_KERNEL_B0_TC CGEMM_SMALL_KERNEL_B0_TC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0_RN CGEMM_SMALL_KERNEL_B0_RN
|
||||
#define GEMM_SMALL_KERNEL_B0_RT CGEMM_SMALL_KERNEL_B0_RT
|
||||
#define GEMM_SMALL_KERNEL_B0_RR CGEMM_SMALL_KERNEL_B0_RR
|
||||
#define GEMM_SMALL_KERNEL_B0_RC CGEMM_SMALL_KERNEL_B0_RC
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0_CN CGEMM_SMALL_KERNEL_B0_CN
|
||||
#define GEMM_SMALL_KERNEL_B0_CT CGEMM_SMALL_KERNEL_B0_CT
|
||||
#define GEMM_SMALL_KERNEL_B0_CR CGEMM_SMALL_KERNEL_B0_CR
|
||||
#define GEMM_SMALL_KERNEL_B0_CC CGEMM_SMALL_KERNEL_B0_CC
|
||||
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
||||
+132
@@ -145,6 +145,19 @@ BLASLONG (*isbmin_k) (BLASLONG, float *, BLASLONG);
|
||||
int (*sbneg_tcopy) (BLASLONG, BLASLONG, float *, BLASLONG, float *);
|
||||
int (*sblaswp_ncopy) (BLASLONG, BLASLONG, BLASLONG, float *, BLASLONG, blasint *, float *);
|
||||
|
||||
#ifdef SMALL_MATRIX_OPT
|
||||
int (*sbgemm_small_matrix_permit)(int transa, int transb, BLASLONG m, BLASLONG n, BLASLONG k, float alpha, float beta);
|
||||
|
||||
int (*sbgemm_small_kernel_nn )(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int (*sbgemm_small_kernel_nt )(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int (*sbgemm_small_kernel_tn )(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int (*sbgemm_small_kernel_tt )(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
|
||||
int (*sbgemm_small_kernel_b0_nn )(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*sbgemm_small_kernel_b0_nt )(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*sbgemm_small_kernel_b0_tn )(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*sbgemm_small_kernel_b0_tt )(BLASLONG m, BLASLONG n, BLASLONG k, bfloat16 * A, BLASLONG lda, float alpha, bfloat16 * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#if defined(BUILD_SINGLE) || defined(BUILD_COMPLEX)
|
||||
@@ -207,6 +220,20 @@ BLASLONG (*ismin_k) (BLASLONG, float *, BLASLONG);
|
||||
int (*sgemm_otcopy )(BLASLONG, BLASLONG, float *, BLASLONG, float *);
|
||||
#endif
|
||||
#ifdef BUILD_SINGLE
|
||||
#ifdef SMALL_MATRIX_OPT
|
||||
int (*sgemm_small_matrix_permit)(int transa, int transb, BLASLONG m, BLASLONG n, BLASLONG k, float alpha, float beta);
|
||||
|
||||
int (*sgemm_small_kernel_nn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int (*sgemm_small_kernel_nt )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int (*sgemm_small_kernel_tn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
int (*sgemm_small_kernel_tt )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float beta, float * C, BLASLONG ldc);
|
||||
|
||||
int (*sgemm_small_kernel_b0_nn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*sgemm_small_kernel_b0_nt )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*sgemm_small_kernel_b0_tn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*sgemm_small_kernel_b0_tt )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
#endif
|
||||
|
||||
int (*strsm_kernel_LN)(BLASLONG, BLASLONG, BLASLONG, float, float *, float *, float *, BLASLONG, BLASLONG);
|
||||
int (*strsm_kernel_LT)(BLASLONG, BLASLONG, BLASLONG, float, float *, float *, float *, BLASLONG, BLASLONG);
|
||||
int (*strsm_kernel_RN)(BLASLONG, BLASLONG, BLASLONG, float, float *, float *, float *, BLASLONG, BLASLONG);
|
||||
@@ -314,6 +341,19 @@ BLASLONG (*idmin_k) (BLASLONG, double *, BLASLONG);
|
||||
int (*dgemm_otcopy )(BLASLONG, BLASLONG, double *, BLASLONG, double *);
|
||||
#endif
|
||||
#ifdef BUILD_DOUBLE
|
||||
#ifdef SMALL_MATRIX_OPT
|
||||
int (*dgemm_small_matrix_permit)(int transa, int transb, BLASLONG m, BLASLONG n, BLASLONG k, double alpha, double beta);
|
||||
|
||||
int (*dgemm_small_kernel_nn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double beta, double * C, BLASLONG ldc);
|
||||
int (*dgemm_small_kernel_nt )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double beta, double * C, BLASLONG ldc);
|
||||
int (*dgemm_small_kernel_tn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double beta, double * C, BLASLONG ldc);
|
||||
int (*dgemm_small_kernel_tt )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double beta, double * C, BLASLONG ldc);
|
||||
|
||||
int (*dgemm_small_kernel_b0_nn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*dgemm_small_kernel_b0_nt )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*dgemm_small_kernel_b0_tn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*dgemm_small_kernel_b0_tt )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
#endif
|
||||
int (*dtrsm_kernel_LN)(BLASLONG, BLASLONG, BLASLONG, double, double *, double *, double *, BLASLONG, BLASLONG);
|
||||
int (*dtrsm_kernel_LT)(BLASLONG, BLASLONG, BLASLONG, double, double *, double *, double *, BLASLONG, BLASLONG);
|
||||
int (*dtrsm_kernel_RN)(BLASLONG, BLASLONG, BLASLONG, double, double *, double *, double *, BLASLONG, BLASLONG);
|
||||
@@ -513,6 +553,50 @@ BLASLONG (*icamin_k)(BLASLONG, float *, BLASLONG);
|
||||
int (*cgemm_oncopy )(BLASLONG, BLASLONG, float *, BLASLONG, float *);
|
||||
int (*cgemm_otcopy )(BLASLONG, BLASLONG, float *, BLASLONG, float *);
|
||||
|
||||
#ifdef SMALL_MATRIX_OPT
|
||||
int (*cgemm_small_matrix_permit)(int transa, int transb, BLASLONG m, BLASLONG n, BLASLONG k, float alpha0, float alpha1, float beta0, float beta1);
|
||||
|
||||
int (*cgemm_small_kernel_nn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_nt )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_nr )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_nc )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
|
||||
int (*cgemm_small_kernel_tn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_tt )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_tr )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_tc )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
|
||||
int (*cgemm_small_kernel_rn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_rt )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_rr )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_rc )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
|
||||
int (*cgemm_small_kernel_cn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_ct )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_cr )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_cc )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float beta0, float beta1, float * C, BLASLONG ldc);
|
||||
|
||||
int (*cgemm_small_kernel_b0_nn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_nt )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_nr )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_nc )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
|
||||
int (*cgemm_small_kernel_b0_tn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_tt )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_tr )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_tc )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
|
||||
int (*cgemm_small_kernel_b0_rn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_rt )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_rr )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_rc )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
|
||||
int (*cgemm_small_kernel_b0_cn )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_ct )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_cr )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
int (*cgemm_small_kernel_b0_cc )(BLASLONG m, BLASLONG n, BLASLONG k, float * A, BLASLONG lda, float alpha0, float alpha1, float * B, BLASLONG ldb, float * C, BLASLONG ldc);
|
||||
#endif
|
||||
|
||||
int (*ctrsm_kernel_LN)(BLASLONG, BLASLONG, BLASLONG, float, float, float *, float *, float *, BLASLONG, BLASLONG);
|
||||
int (*ctrsm_kernel_LT)(BLASLONG, BLASLONG, BLASLONG, float, float, float *, float *, float *, BLASLONG, BLASLONG);
|
||||
int (*ctrsm_kernel_LR)(BLASLONG, BLASLONG, BLASLONG, float, float, float *, float *, float *, BLASLONG, BLASLONG);
|
||||
@@ -679,6 +763,50 @@ BLASLONG (*izamin_k)(BLASLONG, double *, BLASLONG);
|
||||
int (*zgemm_oncopy )(BLASLONG, BLASLONG, double *, BLASLONG, double *);
|
||||
int (*zgemm_otcopy )(BLASLONG, BLASLONG, double *, BLASLONG, double *);
|
||||
|
||||
#ifdef SMALL_MATRIX_OPT
|
||||
int (*zgemm_small_matrix_permit)(int transa, int transb, BLASLONG m, BLASLONG n, BLASLONG k, double alpha0, double alpha1, double beta0, double beta1);
|
||||
|
||||
int (*zgemm_small_kernel_nn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_nt )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_nr )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_nc )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
|
||||
int (*zgemm_small_kernel_tn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_tt )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_tr )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_tc )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
|
||||
int (*zgemm_small_kernel_rn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_rt )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_rr )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_rc )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
|
||||
int (*zgemm_small_kernel_cn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_ct )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_cr )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_cc )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double beta0, double beta1, double * C, BLASLONG ldc);
|
||||
|
||||
int (*zgemm_small_kernel_b0_nn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_nt )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_nr )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_nc )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
|
||||
int (*zgemm_small_kernel_b0_tn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_tt )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_tr )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_tc )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
|
||||
int (*zgemm_small_kernel_b0_rn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_rt )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_rr )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_rc )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
|
||||
int (*zgemm_small_kernel_b0_cn )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_ct )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_cr )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
int (*zgemm_small_kernel_b0_cc )(BLASLONG m, BLASLONG n, BLASLONG k, double * A, BLASLONG lda, double alpha0, double alpha1, double * B, BLASLONG ldb, double * C, BLASLONG ldc);
|
||||
#endif
|
||||
|
||||
int (*ztrsm_kernel_LN)(BLASLONG, BLASLONG, BLASLONG, double, double, double *, double *, double *, BLASLONG, BLASLONG);
|
||||
int (*ztrsm_kernel_LT)(BLASLONG, BLASLONG, BLASLONG, double, double, double *, double *, double *, BLASLONG, BLASLONG);
|
||||
int (*ztrsm_kernel_LR)(BLASLONG, BLASLONG, BLASLONG, double, double, double *, double *, double *, BLASLONG, BLASLONG);
|
||||
@@ -1069,6 +1197,8 @@ BLASLONG (*ixamin_k)(BLASLONG, xdouble *, BLASLONG);
|
||||
|
||||
extern gotoblas_t *gotoblas;
|
||||
|
||||
#define FUNC_OFFSET(func) (size_t)(&((gotoblas_t *)NULL)->func)
|
||||
|
||||
#define DTB_ENTRIES gotoblas -> dtb_entries
|
||||
#define GEMM_OFFSET_A gotoblas -> offsetA
|
||||
#define GEMM_OFFSET_B gotoblas -> offsetB
|
||||
@@ -1174,6 +1304,8 @@ extern gotoblas_t *gotoblas;
|
||||
|
||||
#else
|
||||
|
||||
#define FUNC_OFFSET(func) (size_t)(func)
|
||||
|
||||
#define DTB_ENTRIES DTB_DEFAULT_ENTRIES
|
||||
|
||||
#define GEMM_OFFSET_A GEMM_DEFAULT_OFFSET_A
|
||||
|
||||
+15
@@ -164,6 +164,8 @@
|
||||
|
||||
#define SGEADD_K sgeadd_k
|
||||
|
||||
#define SGEMM_SMALL_MATRIX_PERMIT sgemm_small_matrix_permit
|
||||
|
||||
#else
|
||||
|
||||
#define SAMAX_K gotoblas -> samax_k
|
||||
@@ -299,8 +301,21 @@
|
||||
|
||||
#define SGEADD_K gotoblas -> sgeadd_k
|
||||
|
||||
#define SGEMM_SMALL_MATRIX_PERMIT gotoblas -> sgemm_small_matrix_permit
|
||||
|
||||
#endif
|
||||
|
||||
#define SGEMM_SMALL_KERNEL_NN FUNC_OFFSET(sgemm_small_kernel_nn)
|
||||
#define SGEMM_SMALL_KERNEL_NT FUNC_OFFSET(sgemm_small_kernel_nt)
|
||||
#define SGEMM_SMALL_KERNEL_TN FUNC_OFFSET(sgemm_small_kernel_tn)
|
||||
#define SGEMM_SMALL_KERNEL_TT FUNC_OFFSET(sgemm_small_kernel_tt)
|
||||
|
||||
#define SGEMM_SMALL_KERNEL_B0_NN FUNC_OFFSET(sgemm_small_kernel_b0_nn)
|
||||
#define SGEMM_SMALL_KERNEL_B0_NT FUNC_OFFSET(sgemm_small_kernel_b0_nt)
|
||||
#define SGEMM_SMALL_KERNEL_B0_TN FUNC_OFFSET(sgemm_small_kernel_b0_tn)
|
||||
#define SGEMM_SMALL_KERNEL_B0_TT FUNC_OFFSET(sgemm_small_kernel_b0_tt)
|
||||
|
||||
|
||||
#define SGEMM_NN sgemm_nn
|
||||
#define SGEMM_CN sgemm_tn
|
||||
#define SGEMM_TN sgemm_tn
|
||||
|
||||
+12
@@ -24,6 +24,7 @@
|
||||
#define SBGEMM_BETA sbgemm_beta
|
||||
#define SBGEMM_KERNEL sbgemm_kernel
|
||||
|
||||
#define SBGEMM_SMALL_MATRIX_PERMIT sbgemm_small_matrix_permit
|
||||
#else
|
||||
|
||||
#define SBDOT_K gotoblas -> sbdot_k
|
||||
@@ -41,8 +42,19 @@
|
||||
#define SBGEMM_BETA gotoblas -> sbgemm_beta
|
||||
#define SBGEMM_KERNEL gotoblas -> sbgemm_kernel
|
||||
|
||||
#define SBGEMM_SMALL_MATRIX_PERMIT gotoblas -> sbgemm_small_matrix_permit
|
||||
#endif
|
||||
|
||||
#define SBGEMM_SMALL_KERNEL_NN FUNC_OFFSET(sbgemm_small_kernel_nn)
|
||||
#define SBGEMM_SMALL_KERNEL_NT FUNC_OFFSET(sbgemm_small_kernel_nt)
|
||||
#define SBGEMM_SMALL_KERNEL_TN FUNC_OFFSET(sbgemm_small_kernel_tn)
|
||||
#define SBGEMM_SMALL_KERNEL_TT FUNC_OFFSET(sbgemm_small_kernel_tt)
|
||||
|
||||
#define SBGEMM_SMALL_KERNEL_B0_NN FUNC_OFFSET(sbgemm_small_kernel_b0_nn)
|
||||
#define SBGEMM_SMALL_KERNEL_B0_NT FUNC_OFFSET(sbgemm_small_kernel_b0_nt)
|
||||
#define SBGEMM_SMALL_KERNEL_B0_TN FUNC_OFFSET(sbgemm_small_kernel_b0_tn)
|
||||
#define SBGEMM_SMALL_KERNEL_B0_TT FUNC_OFFSET(sbgemm_small_kernel_b0_tt)
|
||||
|
||||
#define SBGEMM_NN sbgemm_nn
|
||||
#define SBGEMM_CN sbgemm_tn
|
||||
#define SBGEMM_TN sbgemm_tn
|
||||
|
||||
+45
@@ -232,6 +232,8 @@
|
||||
|
||||
#define ZGEADD_K zgeadd_k
|
||||
|
||||
#define ZGEMM_SMALL_MATRIX_PERMIT zgemm_small_matrix_permit
|
||||
|
||||
#else
|
||||
|
||||
#define ZAMAX_K gotoblas -> zamax_k
|
||||
@@ -426,8 +428,51 @@
|
||||
|
||||
#define ZGEADD_K gotoblas -> zgeadd_k
|
||||
|
||||
#define ZGEMM_SMALL_MATRIX_PERMIT gotoblas -> zgemm_small_matrix_permit
|
||||
|
||||
#endif
|
||||
|
||||
#define ZGEMM_SMALL_KERNEL_NN FUNC_OFFSET(zgemm_small_kernel_nn)
|
||||
#define ZGEMM_SMALL_KERNEL_NT FUNC_OFFSET(zgemm_small_kernel_nt)
|
||||
#define ZGEMM_SMALL_KERNEL_NR FUNC_OFFSET(zgemm_small_kernel_nr)
|
||||
#define ZGEMM_SMALL_KERNEL_NC FUNC_OFFSET(zgemm_small_kernel_nc)
|
||||
|
||||
#define ZGEMM_SMALL_KERNEL_TN FUNC_OFFSET(zgemm_small_kernel_tn)
|
||||
#define ZGEMM_SMALL_KERNEL_TT FUNC_OFFSET(zgemm_small_kernel_tt)
|
||||
#define ZGEMM_SMALL_KERNEL_TR FUNC_OFFSET(zgemm_small_kernel_tr)
|
||||
#define ZGEMM_SMALL_KERNEL_TC FUNC_OFFSET(zgemm_small_kernel_tc)
|
||||
|
||||
#define ZGEMM_SMALL_KERNEL_RN FUNC_OFFSET(zgemm_small_kernel_rn)
|
||||
#define ZGEMM_SMALL_KERNEL_RT FUNC_OFFSET(zgemm_small_kernel_rt)
|
||||
#define ZGEMM_SMALL_KERNEL_RR FUNC_OFFSET(zgemm_small_kernel_rr)
|
||||
#define ZGEMM_SMALL_KERNEL_RC FUNC_OFFSET(zgemm_small_kernel_rc)
|
||||
|
||||
#define ZGEMM_SMALL_KERNEL_CN FUNC_OFFSET(zgemm_small_kernel_cn)
|
||||
#define ZGEMM_SMALL_KERNEL_CT FUNC_OFFSET(zgemm_small_kernel_ct)
|
||||
#define ZGEMM_SMALL_KERNEL_CR FUNC_OFFSET(zgemm_small_kernel_cr)
|
||||
#define ZGEMM_SMALL_KERNEL_CC FUNC_OFFSET(zgemm_small_kernel_cc)
|
||||
|
||||
#define ZGEMM_SMALL_KERNEL_B0_NN FUNC_OFFSET(zgemm_small_kernel_b0_nn)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_NT FUNC_OFFSET(zgemm_small_kernel_b0_nt)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_NR FUNC_OFFSET(zgemm_small_kernel_b0_nr)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_NC FUNC_OFFSET(zgemm_small_kernel_b0_nc)
|
||||
|
||||
#define ZGEMM_SMALL_KERNEL_B0_TN FUNC_OFFSET(zgemm_small_kernel_b0_tn)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_TT FUNC_OFFSET(zgemm_small_kernel_b0_tt)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_TR FUNC_OFFSET(zgemm_small_kernel_b0_tr)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_TC FUNC_OFFSET(zgemm_small_kernel_b0_tc)
|
||||
|
||||
#define ZGEMM_SMALL_KERNEL_B0_RN FUNC_OFFSET(zgemm_small_kernel_b0_rn)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_RT FUNC_OFFSET(zgemm_small_kernel_b0_rt)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_RR FUNC_OFFSET(zgemm_small_kernel_b0_rr)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_RC FUNC_OFFSET(zgemm_small_kernel_b0_rc)
|
||||
|
||||
#define ZGEMM_SMALL_KERNEL_B0_CN FUNC_OFFSET(zgemm_small_kernel_b0_cn)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_CT FUNC_OFFSET(zgemm_small_kernel_b0_ct)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_CR FUNC_OFFSET(zgemm_small_kernel_b0_cr)
|
||||
#define ZGEMM_SMALL_KERNEL_B0_CC FUNC_OFFSET(zgemm_small_kernel_b0_cc)
|
||||
|
||||
|
||||
#define ZGEMM_NN zgemm_nn
|
||||
#define ZGEMM_CN zgemm_cn
|
||||
#define ZGEMM_TN zgemm_tn
|
||||
|
||||
@@ -1,13 +1,14 @@
|
||||
include ../Makefile.rule
|
||||
TOPDIR = ..
|
||||
include $(TOPDIR)/Makefile.system
|
||||
|
||||
all :: dgemv_tester dgemm_tester
|
||||
|
||||
dgemv_tester :
|
||||
$(CXX) $(COMMON_OPT) -Wall -Wextra -Wshadow -fopenmp -std=c++11 dgemv_thread_safety.cpp ../libopenblas.a -lpthread -o dgemv_tester
|
||||
$(CXX) $(COMMON_OPT) -Wall -Wextra -Wshadow -fopenmp -std=c++11 dgemv_thread_safety.cpp ../$(LIBNAME) $(EXTRALIB) $(FEXTRALIB) -o dgemv_tester
|
||||
./dgemv_tester
|
||||
|
||||
dgemm_tester : dgemv_tester
|
||||
$(CXX) $(COMMON_OPT) -Wall -Wextra -Wshadow -fopenmp -std=c++11 dgemm_thread_safety.cpp ../libopenblas.a -lpthread -o dgemm_tester
|
||||
$(CXX) $(COMMON_OPT) -Wall -Wextra -Wshadow -fopenmp -std=c++11 dgemm_thread_safety.cpp ../$(LIBNAME) $(EXTRALIB) $(FEXTRALIB) -o dgemm_tester
|
||||
./dgemm_tester
|
||||
|
||||
clean ::
|
||||
|
||||
@@ -120,6 +120,7 @@
|
||||
#define CORE_SKYLAKEX 28
|
||||
#define CORE_DHYANA 29
|
||||
#define CORE_COOPERLAKE 30
|
||||
#define CORE_SAPPHIRERAPIDS 31
|
||||
|
||||
#define HAVE_SSE (1 << 0)
|
||||
#define HAVE_SSE2 (1 << 1)
|
||||
@@ -145,6 +146,7 @@
|
||||
#define HAVE_AVX512VL (1 << 21)
|
||||
#define HAVE_AVX2 (1 << 22)
|
||||
#define HAVE_AVX512BF16 (1 << 23)
|
||||
#define HAVE_AMXBF16 (1 << 24)
|
||||
|
||||
#define CACHE_INFO_L1_I 1
|
||||
#define CACHE_INFO_L1_D 2
|
||||
@@ -222,6 +224,7 @@ typedef struct {
|
||||
#define CPUTYPE_SKYLAKEX 52
|
||||
#define CPUTYPE_DHYANA 53
|
||||
#define CPUTYPE_COOPERLAKE 54
|
||||
#define CPUTYPE_SAPPHIRERAPIDS 55
|
||||
|
||||
#define CPUTYPE_HYGON_UNKNOWN 99
|
||||
|
||||
|
||||
+163
-142
@@ -26,10 +26,12 @@
|
||||
*****************************************************************************/
|
||||
|
||||
#include <string.h>
|
||||
#ifdef OS_DARWIN
|
||||
#ifdef __APPLE__
|
||||
#include <sys/sysctl.h>
|
||||
int32_t value;
|
||||
size_t length=sizeof(value);
|
||||
int64_t value64;
|
||||
size_t length64=sizeof(value64);
|
||||
#endif
|
||||
|
||||
#define CPU_UNKNOWN 0
|
||||
@@ -53,6 +55,8 @@ size_t length=sizeof(value);
|
||||
#define CPU_EMAG8180 10
|
||||
// Apple
|
||||
#define CPU_VORTEX 13
|
||||
// Fujitsu
|
||||
#define CPU_A64FX 15
|
||||
|
||||
static char *cpuname[] = {
|
||||
"UNKNOWN",
|
||||
@@ -69,7 +73,8 @@ static char *cpuname[] = {
|
||||
"NEOVERSEN1",
|
||||
"THUNDERX3T110",
|
||||
"VORTEX",
|
||||
"CORTEXA55"
|
||||
"CORTEXA55",
|
||||
"A64FX"
|
||||
};
|
||||
|
||||
static char *cpuname_lower[] = {
|
||||
@@ -87,7 +92,8 @@ static char *cpuname_lower[] = {
|
||||
"neoversen1",
|
||||
"thunderx3t110",
|
||||
"vortex",
|
||||
"cortexa55"
|
||||
"cortexa55",
|
||||
"a64fx"
|
||||
};
|
||||
|
||||
int get_feature(char *search)
|
||||
@@ -183,6 +189,9 @@ int detect(void)
|
||||
// Ampere
|
||||
else if (strstr(cpu_implementer, "0x50") && strstr(cpu_part, "0x000"))
|
||||
return CPU_EMAG8180;
|
||||
// Fujitsu
|
||||
else if (strstr(cpu_implementer, "0x46") && strstr(cpu_part, "0x001"))
|
||||
return CPU_A64FX;
|
||||
}
|
||||
|
||||
p = (char *) NULL ;
|
||||
@@ -212,9 +221,9 @@ int detect(void)
|
||||
|
||||
}
|
||||
#else
|
||||
#ifdef DARWIN
|
||||
#ifdef __APPLE__
|
||||
sysctlbyname("hw.cpufamily",&value,&length,NULL,0);
|
||||
if (value ==131287967) return CPU_VORTEX;
|
||||
if (value ==131287967|| value == 458787763 ) return CPU_VORTEX;
|
||||
#endif
|
||||
return CPU_ARMV8;
|
||||
#endif
|
||||
@@ -265,7 +274,7 @@ int n=0;
|
||||
|
||||
printf("#define NUM_CORES %d\n",n);
|
||||
#endif
|
||||
#ifdef DARWIN
|
||||
#ifdef __APPLE__
|
||||
sysctlbyname("hw.physicalcpu_max",&value,&length,NULL,0);
|
||||
printf("#define NUM_CORES %d\n",value);
|
||||
#endif
|
||||
@@ -285,154 +294,166 @@ void get_cpuconfig(void)
|
||||
switch (d)
|
||||
{
|
||||
|
||||
case CPU_CORTEXA53:
|
||||
case CPU_CORTEXA55:
|
||||
printf("#define %s\n", cpuname[d]);
|
||||
// Fall-through
|
||||
case CPU_ARMV8:
|
||||
// Minimum parameters for ARMv8 (based on A53)
|
||||
printf("#define L1_DATA_SIZE 32768\n");
|
||||
printf("#define L1_DATA_LINESIZE 64\n");
|
||||
printf("#define L2_SIZE 262144\n");
|
||||
printf("#define L2_LINESIZE 64\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
printf("#define L2_ASSOCIATIVE 4\n");
|
||||
case CPU_CORTEXA53:
|
||||
case CPU_CORTEXA55:
|
||||
printf("#define %s\n", cpuname[d]);
|
||||
// Fall-through
|
||||
case CPU_ARMV8:
|
||||
// Minimum parameters for ARMv8 (based on A53)
|
||||
printf("#define L1_DATA_SIZE 32768\n");
|
||||
printf("#define L1_DATA_LINESIZE 64\n");
|
||||
printf("#define L2_SIZE 262144\n");
|
||||
printf("#define L2_LINESIZE 64\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
printf("#define L2_ASSOCIATIVE 4\n");
|
||||
break;
|
||||
|
||||
case CPU_CORTEXA57:
|
||||
case CPU_CORTEXA72:
|
||||
case CPU_CORTEXA73:
|
||||
case CPU_CORTEXA57:
|
||||
case CPU_CORTEXA72:
|
||||
case CPU_CORTEXA73:
|
||||
// Common minimum settings for these Arm cores
|
||||
// Can change a lot, but we need to be conservative
|
||||
// TODO: detect info from /sys if possible
|
||||
printf("#define %s\n", cpuname[d]);
|
||||
printf("#define L1_CODE_SIZE 49152\n");
|
||||
printf("#define L1_CODE_LINESIZE 64\n");
|
||||
printf("#define L1_CODE_ASSOCIATIVE 3\n");
|
||||
printf("#define L1_DATA_SIZE 32768\n");
|
||||
printf("#define L1_DATA_LINESIZE 64\n");
|
||||
printf("#define L1_DATA_ASSOCIATIVE 2\n");
|
||||
printf("#define L2_SIZE 524288\n");
|
||||
printf("#define L2_LINESIZE 64\n");
|
||||
printf("#define L2_ASSOCIATIVE 16\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
break;
|
||||
case CPU_NEOVERSEN1:
|
||||
printf("#define %s\n", cpuname[d]);
|
||||
printf("#define L1_CODE_SIZE 65536\n");
|
||||
printf("#define L1_CODE_LINESIZE 64\n");
|
||||
printf("#define L1_CODE_ASSOCIATIVE 4\n");
|
||||
printf("#define L1_DATA_SIZE 65536\n");
|
||||
printf("#define L1_DATA_LINESIZE 64\n");
|
||||
printf("#define L1_DATA_ASSOCIATIVE 4\n");
|
||||
printf("#define L2_SIZE 1048576\n");
|
||||
printf("#define L2_LINESIZE 64\n");
|
||||
printf("#define L2_ASSOCIATIVE 16\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
break;
|
||||
printf("#define %s\n", cpuname[d]);
|
||||
printf("#define L1_CODE_SIZE 49152\n");
|
||||
printf("#define L1_CODE_LINESIZE 64\n");
|
||||
printf("#define L1_CODE_ASSOCIATIVE 3\n");
|
||||
printf("#define L1_DATA_SIZE 32768\n");
|
||||
printf("#define L1_DATA_LINESIZE 64\n");
|
||||
printf("#define L1_DATA_ASSOCIATIVE 2\n");
|
||||
printf("#define L2_SIZE 524288\n");
|
||||
printf("#define L2_LINESIZE 64\n");
|
||||
printf("#define L2_ASSOCIATIVE 16\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
break;
|
||||
case CPU_NEOVERSEN1:
|
||||
printf("#define %s\n", cpuname[d]);
|
||||
printf("#define L1_CODE_SIZE 65536\n");
|
||||
printf("#define L1_CODE_LINESIZE 64\n");
|
||||
printf("#define L1_CODE_ASSOCIATIVE 4\n");
|
||||
printf("#define L1_DATA_SIZE 65536\n");
|
||||
printf("#define L1_DATA_LINESIZE 64\n");
|
||||
printf("#define L1_DATA_ASSOCIATIVE 4\n");
|
||||
printf("#define L2_SIZE 1048576\n");
|
||||
printf("#define L2_LINESIZE 64\n");
|
||||
printf("#define L2_ASSOCIATIVE 16\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
break;
|
||||
|
||||
case CPU_FALKOR:
|
||||
printf("#define FALKOR\n");
|
||||
printf("#define L1_CODE_SIZE 65536\n");
|
||||
printf("#define L1_CODE_LINESIZE 64\n");
|
||||
printf("#define L1_DATA_SIZE 32768\n");
|
||||
printf("#define L1_DATA_LINESIZE 128\n");
|
||||
printf("#define L2_SIZE 524288\n");
|
||||
printf("#define L2_LINESIZE 64\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
printf("#define L2_ASSOCIATIVE 16\n");
|
||||
break;
|
||||
case CPU_FALKOR:
|
||||
printf("#define FALKOR\n");
|
||||
printf("#define L1_CODE_SIZE 65536\n");
|
||||
printf("#define L1_CODE_LINESIZE 64\n");
|
||||
printf("#define L1_DATA_SIZE 32768\n");
|
||||
printf("#define L1_DATA_LINESIZE 128\n");
|
||||
printf("#define L2_SIZE 524288\n");
|
||||
printf("#define L2_LINESIZE 64\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
printf("#define L2_ASSOCIATIVE 16\n");
|
||||
break;
|
||||
|
||||
case CPU_THUNDERX:
|
||||
printf("#define THUNDERX\n");
|
||||
printf("#define L1_DATA_SIZE 32768\n");
|
||||
printf("#define L1_DATA_LINESIZE 128\n");
|
||||
printf("#define L2_SIZE 16777216\n");
|
||||
printf("#define L2_LINESIZE 128\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
printf("#define L2_ASSOCIATIVE 16\n");
|
||||
break;
|
||||
case CPU_THUNDERX:
|
||||
printf("#define THUNDERX\n");
|
||||
printf("#define L1_DATA_SIZE 32768\n");
|
||||
printf("#define L1_DATA_LINESIZE 128\n");
|
||||
printf("#define L2_SIZE 16777216\n");
|
||||
printf("#define L2_LINESIZE 128\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
printf("#define L2_ASSOCIATIVE 16\n");
|
||||
break;
|
||||
|
||||
case CPU_THUNDERX2T99:
|
||||
printf("#define THUNDERX2T99 \n");
|
||||
printf("#define L1_CODE_SIZE 32768 \n");
|
||||
printf("#define L1_CODE_LINESIZE 64 \n");
|
||||
printf("#define L1_CODE_ASSOCIATIVE 8 \n");
|
||||
printf("#define L1_DATA_SIZE 32768 \n");
|
||||
printf("#define L1_DATA_LINESIZE 64 \n");
|
||||
printf("#define L1_DATA_ASSOCIATIVE 8 \n");
|
||||
printf("#define L2_SIZE 262144 \n");
|
||||
printf("#define L2_LINESIZE 64 \n");
|
||||
printf("#define L2_ASSOCIATIVE 8 \n");
|
||||
printf("#define L3_SIZE 33554432 \n");
|
||||
printf("#define L3_LINESIZE 64 \n");
|
||||
printf("#define L3_ASSOCIATIVE 32 \n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64 \n");
|
||||
printf("#define DTB_SIZE 4096 \n");
|
||||
break;
|
||||
case CPU_THUNDERX2T99:
|
||||
printf("#define THUNDERX2T99 \n");
|
||||
printf("#define L1_CODE_SIZE 32768 \n");
|
||||
printf("#define L1_CODE_LINESIZE 64 \n");
|
||||
printf("#define L1_CODE_ASSOCIATIVE 8 \n");
|
||||
printf("#define L1_DATA_SIZE 32768 \n");
|
||||
printf("#define L1_DATA_LINESIZE 64 \n");
|
||||
printf("#define L1_DATA_ASSOCIATIVE 8 \n");
|
||||
printf("#define L2_SIZE 262144 \n");
|
||||
printf("#define L2_LINESIZE 64 \n");
|
||||
printf("#define L2_ASSOCIATIVE 8 \n");
|
||||
printf("#define L3_SIZE 33554432 \n");
|
||||
printf("#define L3_LINESIZE 64 \n");
|
||||
printf("#define L3_ASSOCIATIVE 32 \n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64 \n");
|
||||
printf("#define DTB_SIZE 4096 \n");
|
||||
break;
|
||||
|
||||
case CPU_TSV110:
|
||||
printf("#define TSV110 \n");
|
||||
printf("#define L1_CODE_SIZE 65536 \n");
|
||||
printf("#define L1_CODE_LINESIZE 64 \n");
|
||||
printf("#define L1_CODE_ASSOCIATIVE 4 \n");
|
||||
printf("#define L1_DATA_SIZE 65536 \n");
|
||||
printf("#define L1_DATA_LINESIZE 64 \n");
|
||||
printf("#define L1_DATA_ASSOCIATIVE 4 \n");
|
||||
printf("#define L2_SIZE 524228 \n");
|
||||
printf("#define L2_LINESIZE 64 \n");
|
||||
printf("#define L2_ASSOCIATIVE 8 \n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64 \n");
|
||||
printf("#define DTB_SIZE 4096 \n");
|
||||
break;
|
||||
case CPU_TSV110:
|
||||
printf("#define TSV110 \n");
|
||||
printf("#define L1_CODE_SIZE 65536 \n");
|
||||
printf("#define L1_CODE_LINESIZE 64 \n");
|
||||
printf("#define L1_CODE_ASSOCIATIVE 4 \n");
|
||||
printf("#define L1_DATA_SIZE 65536 \n");
|
||||
printf("#define L1_DATA_LINESIZE 64 \n");
|
||||
printf("#define L1_DATA_ASSOCIATIVE 4 \n");
|
||||
printf("#define L2_SIZE 524228 \n");
|
||||
printf("#define L2_LINESIZE 64 \n");
|
||||
printf("#define L2_ASSOCIATIVE 8 \n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64 \n");
|
||||
printf("#define DTB_SIZE 4096 \n");
|
||||
break;
|
||||
|
||||
case CPU_EMAG8180:
|
||||
// Minimum parameters for ARMv8 (based on A53)
|
||||
printf("#define EMAG8180\n");
|
||||
printf("#define L1_CODE_SIZE 32768\n");
|
||||
printf("#define L1_DATA_SIZE 32768\n");
|
||||
printf("#define L1_DATA_LINESIZE 64\n");
|
||||
printf("#define L2_SIZE 262144\n");
|
||||
printf("#define L2_LINESIZE 64\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
break;
|
||||
case CPU_EMAG8180:
|
||||
// Minimum parameters for ARMv8 (based on A53)
|
||||
printf("#define EMAG8180\n");
|
||||
printf("#define L1_CODE_SIZE 32768\n");
|
||||
printf("#define L1_DATA_SIZE 32768\n");
|
||||
printf("#define L1_DATA_LINESIZE 64\n");
|
||||
printf("#define L2_SIZE 262144\n");
|
||||
printf("#define L2_LINESIZE 64\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
break;
|
||||
|
||||
case CPU_THUNDERX3T110:
|
||||
printf("#define THUNDERX3T110 \n");
|
||||
printf("#define L1_CODE_SIZE 65536 \n");
|
||||
printf("#define L1_CODE_LINESIZE 64 \n");
|
||||
printf("#define L1_CODE_ASSOCIATIVE 8 \n");
|
||||
printf("#define L1_DATA_SIZE 32768 \n");
|
||||
printf("#define L1_DATA_LINESIZE 64 \n");
|
||||
printf("#define L1_DATA_ASSOCIATIVE 8 \n");
|
||||
printf("#define L2_SIZE 524288 \n");
|
||||
printf("#define L2_LINESIZE 64 \n");
|
||||
printf("#define L2_ASSOCIATIVE 8 \n");
|
||||
printf("#define L3_SIZE 94371840 \n");
|
||||
printf("#define L3_LINESIZE 64 \n");
|
||||
printf("#define L3_ASSOCIATIVE 32 \n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64 \n");
|
||||
printf("#define DTB_SIZE 4096 \n");
|
||||
break;
|
||||
#ifdef DARWIN
|
||||
case CPU_VORTEX:
|
||||
printf("#define VORTEX \n");
|
||||
sysctlbyname("hw.l1icachesize",&value,&length,NULL,0);
|
||||
printf("#define L1_CODE_SIZE %d \n",value);
|
||||
sysctlbyname("hw.cachelinesize",&value,&length,NULL,0);
|
||||
printf("#define L1_CODE_LINESIZE %d \n",value);
|
||||
sysctlbyname("hw.l1dcachesize",&value,&length,NULL,0);
|
||||
printf("#define L1_DATA_SIZE %d \n",value);
|
||||
sysctlbyname("hw.l2dcachesize",&value,&length,NULL,0);
|
||||
printf("#define L2_SIZE %d \n",value);
|
||||
break;
|
||||
case CPU_THUNDERX3T110:
|
||||
printf("#define THUNDERX3T110 \n");
|
||||
printf("#define L1_CODE_SIZE 65536 \n");
|
||||
printf("#define L1_CODE_LINESIZE 64 \n");
|
||||
printf("#define L1_CODE_ASSOCIATIVE 8 \n");
|
||||
printf("#define L1_DATA_SIZE 32768 \n");
|
||||
printf("#define L1_DATA_LINESIZE 64 \n");
|
||||
printf("#define L1_DATA_ASSOCIATIVE 8 \n");
|
||||
printf("#define L2_SIZE 524288 \n");
|
||||
printf("#define L2_LINESIZE 64 \n");
|
||||
printf("#define L2_ASSOCIATIVE 8 \n");
|
||||
printf("#define L3_SIZE 94371840 \n");
|
||||
printf("#define L3_LINESIZE 64 \n");
|
||||
printf("#define L3_ASSOCIATIVE 32 \n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64 \n");
|
||||
printf("#define DTB_SIZE 4096 \n");
|
||||
break;
|
||||
#ifdef __APPLE__
|
||||
case CPU_VORTEX:
|
||||
printf("#define VORTEX \n");
|
||||
sysctlbyname("hw.l1icachesize",&value64,&length64,NULL,0);
|
||||
printf("#define L1_CODE_SIZE %lld \n",value64);
|
||||
sysctlbyname("hw.cachelinesize",&value64,&length64,NULL,0);
|
||||
printf("#define L1_CODE_LINESIZE %lld \n",value64);
|
||||
sysctlbyname("hw.l1dcachesize",&value64,&length64,NULL,0);
|
||||
printf("#define L1_DATA_SIZE %lld \n",value64);
|
||||
sysctlbyname("hw.l2cachesize",&value64,&length64,NULL,0);
|
||||
printf("#define L2_SIZE %lld \n",value64);
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64 \n");
|
||||
printf("#define DTB_SIZE 4096 \n");
|
||||
break;
|
||||
#endif
|
||||
case CPU_A64FX:
|
||||
printf("#define A64FX\n");
|
||||
printf("#define L1_CODE_SIZE 65535\n");
|
||||
printf("#define L1_DATA_SIZE 65535\n");
|
||||
printf("#define L1_DATA_LINESIZE 256\n");
|
||||
printf("#define L2_SIZE 8388608\n");
|
||||
printf("#define L2_LINESIZE 256\n");
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64\n");
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
break;
|
||||
}
|
||||
get_cpucount();
|
||||
}
|
||||
|
||||
@@ -165,6 +165,7 @@ void get_cpuconfig(void){
|
||||
}else{
|
||||
printf("#define UNKNOWN\n");
|
||||
}
|
||||
if (!get_feature(msa)) printf("#define NO_MSA\n");
|
||||
}
|
||||
|
||||
void get_libname(void){
|
||||
@@ -178,3 +179,38 @@ void get_libname(void){
|
||||
printf("mips\n");
|
||||
}
|
||||
}
|
||||
|
||||
int get_feature(char *search)
|
||||
{
|
||||
|
||||
#ifdef __linux
|
||||
FILE *infile;
|
||||
char buffer[2048], *p,*t;
|
||||
p = (char *) NULL ;
|
||||
|
||||
infile = fopen("/proc/cpuinfo", "r");
|
||||
|
||||
while (fgets(buffer, sizeof(buffer), infile))
|
||||
{
|
||||
|
||||
if (!strncmp("Features", buffer, 8))
|
||||
{
|
||||
p = strchr(buffer, ':') + 2;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
fclose(infile);
|
||||
|
||||
if( p == NULL ) return 0;
|
||||
|
||||
t = strtok(p," ");
|
||||
while( t = strtok(NULL," "))
|
||||
{
|
||||
if (!strcmp(t, search)) { return(1); }
|
||||
}
|
||||
|
||||
#endif
|
||||
return(0);
|
||||
}
|
||||
|
||||
|
||||
+44
-8
@@ -104,17 +104,17 @@ int detect(void){
|
||||
}
|
||||
}
|
||||
fclose(infile);
|
||||
if(p != NULL){
|
||||
if (strstr(p, "Loongson-3A3000") || strstr(p, "Loongson-3B3000")){
|
||||
return CPU_LOONGSON3R3;
|
||||
}else if(strstr(p, "Loongson-3A4000") || strstr(p, "Loongson-3B4000")){
|
||||
return CPU_LOONGSON3R4;
|
||||
} else{
|
||||
return CPU_SICORTEX;
|
||||
if (p != NULL){
|
||||
if (strstr(p, "Loongson-3A3000") || strstr(p, "Loongson-3B3000")){
|
||||
return CPU_LOONGSON3R3;
|
||||
} else if (strstr(p, "Loongson-3A4000") || strstr(p, "Loongson-3B4000")){
|
||||
return CPU_LOONGSON3R4;
|
||||
} else{
|
||||
return CPU_SICORTEX;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
return CPU_UNKNOWN;
|
||||
}
|
||||
}
|
||||
|
||||
char *get_corename(void){
|
||||
@@ -201,6 +201,7 @@ void get_cpuconfig(void){
|
||||
printf("#define DTB_SIZE 4096\n");
|
||||
printf("#define L2_ASSOCIATIVE 8\n");
|
||||
}
|
||||
if (!get_feature(msa)) printf("#define NO_MSA\n");
|
||||
}
|
||||
|
||||
void get_libname(void){
|
||||
@@ -218,3 +219,38 @@ void get_libname(void){
|
||||
printf("mips64\n");
|
||||
}
|
||||
}
|
||||
|
||||
int get_feature(char *search)
|
||||
{
|
||||
|
||||
#ifdef __linux
|
||||
FILE *infile;
|
||||
char buffer[2048], *p,*t;
|
||||
p = (char *) NULL ;
|
||||
|
||||
infile = fopen("/proc/cpuinfo", "r");
|
||||
|
||||
while (fgets(buffer, sizeof(buffer), infile))
|
||||
{
|
||||
|
||||
if (!strncmp("Features", buffer, 8))
|
||||
{
|
||||
p = strchr(buffer, ':') + 2;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
fclose(infile);
|
||||
|
||||
if( p == NULL ) return 0;
|
||||
|
||||
t = strtok(p," ");
|
||||
while( t = strtok(NULL," "))
|
||||
{
|
||||
if (!strcmp(t, search)) { return(1); }
|
||||
}
|
||||
|
||||
#endif
|
||||
return(0);
|
||||
}
|
||||
|
||||
|
||||
+167
-50
@@ -1,3 +1,4 @@
|
||||
//{
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
@@ -266,6 +267,31 @@ int support_avx512_bf16(){
|
||||
#endif
|
||||
}
|
||||
|
||||
#define BIT_AMX_TILE 0x01000000
|
||||
#define BIT_AMX_BF16 0x00400000
|
||||
#define BIT_AMX_ENBD 0x00060000
|
||||
|
||||
int support_amx_bf16() {
|
||||
#if !defined(NO_AVX) && !defined(NO_AVX512)
|
||||
int eax, ebx, ecx, edx;
|
||||
int ret=0;
|
||||
|
||||
if (!support_avx512())
|
||||
return 0;
|
||||
// CPUID.7.0:EDX indicates AMX support
|
||||
cpuid_count(7, 0, &eax, &ebx, &ecx, &edx);
|
||||
if ((edx & BIT_AMX_TILE) && (edx & BIT_AMX_BF16)) {
|
||||
// CPUID.D.0:EAX[17:18] indicates AMX enabled
|
||||
cpuid_count(0xd, 0, &eax, &ebx, &ecx, &edx);
|
||||
if ((eax & BIT_AMX_ENBD) == BIT_AMX_ENBD)
|
||||
ret = 1;
|
||||
}
|
||||
return ret;
|
||||
#else
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
int get_vendor(void){
|
||||
int eax, ebx, ecx, edx;
|
||||
char vendor[13];
|
||||
@@ -353,6 +379,7 @@ int get_cputype(int gettype){
|
||||
if (support_avx2()) feature |= HAVE_AVX2;
|
||||
if (support_avx512()) feature |= HAVE_AVX512VL;
|
||||
if (support_avx512_bf16()) feature |= HAVE_AVX512BF16;
|
||||
if (support_amx_bf16()) feature |= HAVE_AMXBF16;
|
||||
if ((ecx & (1 << 12)) != 0) feature |= HAVE_FMA3;
|
||||
#endif
|
||||
|
||||
@@ -1429,10 +1456,10 @@ int get_cpuname(void){
|
||||
return CPUTYPE_NEHALEM;
|
||||
}
|
||||
break;
|
||||
case 9:
|
||||
case 8:
|
||||
switch (model) {
|
||||
case 12: // Tiger Lake
|
||||
case 13: // Tiger Lake (11th Gen Intel(R) Core(TM) i7-11800H @ 2.30GHz)
|
||||
if(support_avx512())
|
||||
return CPUTYPE_SKYLAKEX;
|
||||
if(support_avx2())
|
||||
@@ -1448,30 +1475,70 @@ int get_cpuname(void){
|
||||
return CPUTYPE_SANDYBRIDGE;
|
||||
else
|
||||
return CPUTYPE_NEHALEM;
|
||||
}
|
||||
case 10: //family 6 exmodel 10
|
||||
switch (model) {
|
||||
case 5: // Comet Lake H and S
|
||||
case 6: // Comet Lake U
|
||||
if(support_avx2())
|
||||
return CPUTYPE_HASWELL;
|
||||
if(support_avx())
|
||||
return CPUTYPE_SANDYBRIDGE;
|
||||
else
|
||||
return CPUTYPE_NEHALEM;
|
||||
case 7: // Rocket Lake
|
||||
if(support_avx512())
|
||||
case 15: // Sapphire Rapids
|
||||
if(support_avx512_bf16())
|
||||
return CPUTYPE_COOPERLAKE;
|
||||
if(support_avx512())
|
||||
return CPUTYPE_SKYLAKEX;
|
||||
if(support_avx2())
|
||||
return CPUTYPE_HASWELL;
|
||||
if(support_avx())
|
||||
return CPUTYPE_SANDYBRIDGE;
|
||||
else
|
||||
return CPUTYPE_NEHALEM;
|
||||
}
|
||||
break;
|
||||
}
|
||||
return CPUTYPE_NEHALEM;
|
||||
}
|
||||
break;
|
||||
case 9:
|
||||
switch (model) {
|
||||
case 7: // Alder Lake desktop
|
||||
case 10: // Alder Lake mobile
|
||||
if(support_avx2())
|
||||
return CPUTYPE_HASWELL;
|
||||
if(support_avx())
|
||||
return CPUTYPE_SANDYBRIDGE;
|
||||
else
|
||||
return CPUTYPE_NEHALEM;
|
||||
case 13: // Ice Lake NNPI
|
||||
if(support_avx512())
|
||||
return CPUTYPE_SKYLAKEX;
|
||||
if(support_avx2())
|
||||
return CPUTYPE_HASWELL;
|
||||
if(support_avx())
|
||||
return CPUTYPE_SANDYBRIDGE;
|
||||
else
|
||||
return CPUTYPE_NEHALEM;
|
||||
case 14: // Kaby Lake and refreshes
|
||||
if(support_avx2())
|
||||
return CPUTYPE_HASWELL;
|
||||
if(support_avx())
|
||||
return CPUTYPE_SANDYBRIDGE;
|
||||
else
|
||||
return CPUTYPE_NEHALEM;
|
||||
}
|
||||
break;
|
||||
case 10: //family 6 exmodel 10
|
||||
switch (model) {
|
||||
case 5: // Comet Lake H and S
|
||||
case 6: // Comet Lake U
|
||||
if(support_avx2())
|
||||
return CPUTYPE_HASWELL;
|
||||
if(support_avx())
|
||||
return CPUTYPE_SANDYBRIDGE;
|
||||
else
|
||||
return CPUTYPE_NEHALEM;
|
||||
case 7: // Rocket Lake
|
||||
if(support_avx512())
|
||||
return CPUTYPE_SKYLAKEX;
|
||||
if(support_avx2())
|
||||
return CPUTYPE_HASWELL;
|
||||
if(support_avx())
|
||||
return CPUTYPE_SANDYBRIDGE;
|
||||
else
|
||||
return CPUTYPE_NEHALEM;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 0x7:
|
||||
return CPUTYPE_ITANIUM;
|
||||
case 0xf:
|
||||
@@ -2042,32 +2109,7 @@ int get_coretype(void){
|
||||
return CORE_NEHALEM;
|
||||
}
|
||||
break;
|
||||
case 10:
|
||||
switch (model) {
|
||||
case 5: // Comet Lake H and S
|
||||
case 6: // Comet Lake U
|
||||
if(support_avx())
|
||||
#ifndef NO_AVX2
|
||||
return CORE_HASWELL;
|
||||
#else
|
||||
return CORE_SANDYBRIDGE;
|
||||
#endif
|
||||
else
|
||||
return CORE_NEHALEM;
|
||||
case 7:// Rocket Lake
|
||||
#ifndef NO_AVX512
|
||||
if(support_avx512())
|
||||
return CORE_SKYLAKEX;
|
||||
#endif
|
||||
#ifndef NO_AVX2
|
||||
if(support_avx2())
|
||||
return CORE_HASWELL;
|
||||
#endif
|
||||
if(support_avx())
|
||||
return CORE_SANDYBRIDGE;
|
||||
else
|
||||
return CORE_NEHALEM;
|
||||
}
|
||||
|
||||
case 5:
|
||||
switch (model) {
|
||||
case 6:
|
||||
@@ -2121,6 +2163,7 @@ int get_coretype(void){
|
||||
return CORE_NEHALEM;
|
||||
}
|
||||
break;
|
||||
|
||||
case 6:
|
||||
if (model == 6)
|
||||
#ifndef NO_AVX512
|
||||
@@ -2135,7 +2178,7 @@ int get_coretype(void){
|
||||
else
|
||||
return CORE_NEHALEM;
|
||||
#endif
|
||||
if (model == 10)
|
||||
if (model == 10 || model == 12)
|
||||
#ifndef NO_AVX512
|
||||
if(support_avx512_bf16())
|
||||
return CORE_COOPERLAKE;
|
||||
@@ -2151,10 +2194,11 @@ int get_coretype(void){
|
||||
return CORE_NEHALEM;
|
||||
#endif
|
||||
break;
|
||||
|
||||
case 7:
|
||||
if (model == 10)
|
||||
return CORE_NEHALEM;
|
||||
if (model == 14)
|
||||
if (model == 13 || model == 14) // Ice Lake
|
||||
#ifndef NO_AVX512
|
||||
return CORE_SKYLAKEX;
|
||||
#else
|
||||
@@ -2168,9 +2212,9 @@ int get_coretype(void){
|
||||
return CORE_NEHALEM;
|
||||
#endif
|
||||
break;
|
||||
case 9:
|
||||
|
||||
case 8:
|
||||
if (model == 12) { // Tiger Lake
|
||||
if (model == 12 || model == 13) { // Tiger Lake
|
||||
if(support_avx512())
|
||||
return CORE_SKYLAKEX;
|
||||
if(support_avx2())
|
||||
@@ -2180,7 +2224,7 @@ int get_coretype(void){
|
||||
else
|
||||
return CORE_NEHALEM;
|
||||
}
|
||||
if (model == 14) { // Kaby Lake
|
||||
if (model == 14) { // Kaby Lake mobile
|
||||
if(support_avx())
|
||||
#ifndef NO_AVX2
|
||||
return CORE_HASWELL;
|
||||
@@ -2190,12 +2234,82 @@ int get_coretype(void){
|
||||
else
|
||||
return CORE_NEHALEM;
|
||||
}
|
||||
}
|
||||
if (model == 15) { // Sapphire Rapids
|
||||
if(support_avx512_bf16())
|
||||
return CPUTYPE_COOPERLAKE;
|
||||
if(support_avx512())
|
||||
return CPUTYPE_SKYLAKEX;
|
||||
if(support_avx2())
|
||||
return CPUTYPE_HASWELL;
|
||||
if(support_avx())
|
||||
return CPUTYPE_SANDYBRIDGE;
|
||||
else
|
||||
return CPUTYPE_NEHALEM;
|
||||
}
|
||||
break;
|
||||
|
||||
case 9:
|
||||
if (model == 7 || model == 10) { // Alder Lake
|
||||
if(support_avx2())
|
||||
return CORE_HASWELL;
|
||||
if(support_avx())
|
||||
return CORE_SANDYBRIDGE;
|
||||
else
|
||||
return CORE_NEHALEM;
|
||||
}
|
||||
if (model == 13) { // Ice Lake NNPI
|
||||
if(support_avx512())
|
||||
return CORE_SKYLAKEX;
|
||||
if(support_avx2())
|
||||
return CORE_HASWELL;
|
||||
if(support_avx())
|
||||
return CORE_SANDYBRIDGE;
|
||||
else
|
||||
return CORE_NEHALEM;
|
||||
}
|
||||
if (model == 14) { // Kaby Lake desktop
|
||||
if(support_avx())
|
||||
#ifndef NO_AVX2
|
||||
return CORE_HASWELL;
|
||||
#else
|
||||
return CORE_SANDYBRIDGE;
|
||||
#endif
|
||||
else
|
||||
return CORE_NEHALEM;
|
||||
}
|
||||
break;
|
||||
|
||||
case 10:
|
||||
switch (model) {
|
||||
case 5: // Comet Lake H and S
|
||||
case 6: // Comet Lake U
|
||||
if(support_avx())
|
||||
#ifndef NO_AVX2
|
||||
return CORE_HASWELL;
|
||||
#else
|
||||
return CORE_SANDYBRIDGE;
|
||||
#endif
|
||||
else
|
||||
return CORE_NEHALEM;
|
||||
case 7:// Rocket Lake
|
||||
#ifndef NO_AVX512
|
||||
if(support_avx512())
|
||||
return CORE_SKYLAKEX;
|
||||
#endif
|
||||
#ifndef NO_AVX2
|
||||
if(support_avx2())
|
||||
return CORE_HASWELL;
|
||||
#endif
|
||||
if(support_avx())
|
||||
return CORE_SANDYBRIDGE;
|
||||
else
|
||||
return CORE_NEHALEM;
|
||||
}
|
||||
|
||||
case 15:
|
||||
if (model <= 0x2) return CORE_NORTHWOOD;
|
||||
else return CORE_PRESCOTT;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2389,6 +2503,7 @@ void get_cpuconfig(void){
|
||||
if (features & HAVE_AVX2 ) printf("#define HAVE_AVX2\n");
|
||||
if (features & HAVE_AVX512VL ) printf("#define HAVE_AVX512VL\n");
|
||||
if (features & HAVE_AVX512BF16 ) printf("#define HAVE_AVX512BF16\n");
|
||||
if (features & HAVE_AMXBF16 ) printf("#define HAVE_AMXBF16\n");
|
||||
if (features & HAVE_3DNOWEX) printf("#define HAVE_3DNOWEX\n");
|
||||
if (features & HAVE_3DNOW) printf("#define HAVE_3DNOW\n");
|
||||
if (features & HAVE_FMA4 ) printf("#define HAVE_FMA4\n");
|
||||
@@ -2460,9 +2575,11 @@ void get_sse(void){
|
||||
if (features & HAVE_AVX2 ) printf("HAVE_AVX2=1\n");
|
||||
if (features & HAVE_AVX512VL ) printf("HAVE_AVX512VL=1\n");
|
||||
if (features & HAVE_AVX512BF16 ) printf("HAVE_AVX512BF16=1\n");
|
||||
if (features & HAVE_AMXBF16 ) printf("HAVE_AMXBF16=1\n");
|
||||
if (features & HAVE_3DNOWEX) printf("HAVE_3DNOWEX=1\n");
|
||||
if (features & HAVE_3DNOW) printf("HAVE_3DNOW=1\n");
|
||||
if (features & HAVE_FMA4 ) printf("HAVE_FMA4=1\n");
|
||||
if (features & HAVE_FMA3 ) printf("HAVE_FMA3=1\n");
|
||||
|
||||
}
|
||||
//}
|
||||
+1
-47
@@ -27,57 +27,11 @@
|
||||
|
||||
#include <string.h>
|
||||
|
||||
#define CPU_GENERIC 0
|
||||
#define CPU_Z13 1
|
||||
#define CPU_Z14 2
|
||||
#define CPU_Z15 3
|
||||
#include "cpuid_zarch.h"
|
||||
|
||||
static char *cpuname[] = {
|
||||
"ZARCH_GENERIC",
|
||||
"Z13",
|
||||
"Z14",
|
||||
"Z15"
|
||||
};
|
||||
|
||||
static char *cpuname_lower[] = {
|
||||
"zarch_generic",
|
||||
"z13",
|
||||
"z14",
|
||||
"z15"
|
||||
};
|
||||
|
||||
int detect(void)
|
||||
{
|
||||
FILE *infile;
|
||||
char buffer[512], *p;
|
||||
|
||||
p = (char *)NULL;
|
||||
infile = fopen("/proc/sysinfo", "r");
|
||||
while (fgets(buffer, sizeof(buffer), infile)){
|
||||
if (!strncmp("Type", buffer, 4)){
|
||||
p = strchr(buffer, ':') + 2;
|
||||
#if 0
|
||||
fprintf(stderr, "%s\n", p);
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
fclose(infile);
|
||||
|
||||
if (strstr(p, "2964")) return CPU_Z13;
|
||||
if (strstr(p, "2965")) return CPU_Z13;
|
||||
if (strstr(p, "3906")) return CPU_Z14;
|
||||
if (strstr(p, "3907")) return CPU_Z14;
|
||||
if (strstr(p, "8561")) return CPU_Z14; // fallback z15 to z14
|
||||
if (strstr(p, "8562")) return CPU_Z14; // fallback z15 to z14
|
||||
|
||||
return CPU_GENERIC;
|
||||
}
|
||||
|
||||
void get_libname(void)
|
||||
{
|
||||
|
||||
int d = detect();
|
||||
printf("%s", cpuname_lower[d]);
|
||||
}
|
||||
|
||||
+101
@@ -0,0 +1,101 @@
|
||||
#include <stdlib.h>
|
||||
|
||||
#define CPU_GENERIC 0
|
||||
#define CPU_Z13 1
|
||||
#define CPU_Z14 2
|
||||
#define CPU_Z15 3
|
||||
|
||||
static char *cpuname[] = {
|
||||
"ZARCH_GENERIC",
|
||||
"Z13",
|
||||
"Z14",
|
||||
"Z15"
|
||||
};
|
||||
|
||||
static char *cpuname_lower[] = {
|
||||
"zarch_generic",
|
||||
"z13",
|
||||
"z14",
|
||||
"z15"
|
||||
};
|
||||
|
||||
// Guard the use of getauxval() on glibc version >= 2.16
|
||||
#ifdef __GLIBC__
|
||||
#include <features.h>
|
||||
#if __GLIBC_PREREQ(2, 16)
|
||||
#include <sys/auxv.h>
|
||||
#define HAVE_GETAUXVAL 1
|
||||
|
||||
static unsigned long get_hwcap(void)
|
||||
{
|
||||
unsigned long hwcap = getauxval(AT_HWCAP);
|
||||
char *maskenv;
|
||||
|
||||
// honor requests for not using specific CPU features in LD_HWCAP_MASK
|
||||
maskenv = getenv("LD_HWCAP_MASK");
|
||||
if (maskenv)
|
||||
hwcap &= strtoul(maskenv, NULL, 0);
|
||||
|
||||
return hwcap;
|
||||
// note that a missing auxval is interpreted as no capabilities
|
||||
// available, which is safe.
|
||||
}
|
||||
|
||||
#else // __GLIBC_PREREQ(2, 16)
|
||||
#warn "Cannot detect SIMD support in Z13 or newer architectures since glibc is older than 2.16"
|
||||
|
||||
static unsigned long get_hwcap(void) {
|
||||
// treat missing support for getauxval() as no capabilities available,
|
||||
// which is safe.
|
||||
return 0;
|
||||
}
|
||||
#endif // __GLIBC_PREREQ(2, 16)
|
||||
#endif // __GLIBC
|
||||
|
||||
static int detect(void)
|
||||
{
|
||||
unsigned long hwcap = get_hwcap();
|
||||
|
||||
// Choose the architecture level for optimized kernels based on hardware
|
||||
// capability bits (just like glibc chooses optimized implementations).
|
||||
//
|
||||
// The hardware capability bits that are used here indicate both
|
||||
// hardware support for a particular ISA extension and the presence of
|
||||
// software support to enable its use. For example, when HWCAP_S390_VX
|
||||
// is set then both the CPU can execute SIMD instructions and the Linux
|
||||
// kernel can manage applications using the vector registers and SIMD
|
||||
// instructions.
|
||||
//
|
||||
// See glibc's sysdeps/s390/dl-procinfo.h for an overview (also in
|
||||
// sysdeps/unix/sysv/linux/s390/bits/hwcap.h) of the defined hardware
|
||||
// capability bits. They are derived from the information that the
|
||||
// "store facility list (extended)" instructions provide.
|
||||
// (https://sourceware.org/git/?p=glibc.git;a=blob_plain;f=sysdeps/s390/dl-procinfo.h;hb=HEAD)
|
||||
//
|
||||
// currently used:
|
||||
// HWCAP_S390_VX - vector facility for z/Architecture (introduced with
|
||||
// IBM z13), enables level CPU_Z13 (SIMD)
|
||||
// HWCAP_S390_VXE - vector enhancements facility 1 (introduced with IBM
|
||||
// z14), together with VX enables level CPU_Z14
|
||||
// (single-precision SIMD instructions)
|
||||
//
|
||||
// When you add optimized kernels that make use of other ISA extensions
|
||||
// (e.g., for exploiting the vector-enhancements facility 2 that was introduced
|
||||
// with IBM z15), then add a new architecture level (e.g., CPU_Z15) and gate
|
||||
// it on the hwcap that represents it here (e.g., HWCAP_S390_VXRS_EXT2
|
||||
// for the z15 vector enhancements).
|
||||
//
|
||||
// To learn the value of hwcaps on a given system, set the environment
|
||||
// variable LD_SHOW_AUXV and let ld.so dump it (e.g., by running
|
||||
// LD_SHOW_AUXV=1 /bin/true).
|
||||
// Also, the init function for dynamic arch support will print hwcaps
|
||||
// when OPENBLAS_VERBOSE is set to 2 or higher.
|
||||
if ((hwcap & HWCAP_S390_VX) && (hwcap & HWCAP_S390_VXE))
|
||||
return CPU_Z14;
|
||||
|
||||
if (hwcap & HWCAP_S390_VX)
|
||||
return CPU_Z13;
|
||||
|
||||
return CPU_GENERIC;
|
||||
}
|
||||
|
||||
@@ -84,7 +84,7 @@ OS_AIX
|
||||
OS_OSF
|
||||
#endif
|
||||
|
||||
#if defined(__WIN32) || defined(__WIN64) || defined(__WINNT)
|
||||
#if defined(__WIN32) || defined(__WIN64) || defined(_WIN32) || defined(_WIN64) || defined(__WINNT)
|
||||
OS_WINNT
|
||||
#endif
|
||||
|
||||
@@ -141,7 +141,7 @@ ARCH_SPARC
|
||||
ARCH_IA64
|
||||
#endif
|
||||
|
||||
#if defined(__LP64) || defined(__LP64__) || defined(__ptr64) || defined(__x86_64__) || defined(__amd64__) || defined(__64BIT__)
|
||||
#if defined(__LP64) || defined(__LP64__) || defined(__ptr64) || defined(__x86_64__) || defined(__amd64__) || defined(__64BIT__) || defined(__aarch64__)
|
||||
BINARY_64
|
||||
#endif
|
||||
|
||||
|
||||
@@ -81,6 +81,7 @@ foreach (float_type ${FLOAT_TYPES})
|
||||
GenerateNamedObjects("gbmv_thread.c" "TRANSA" "gbmv_thread_t" false "" "" false ${float_type})
|
||||
endif ()
|
||||
|
||||
# special defines for complex
|
||||
if (${float_type} STREQUAL "COMPLEX" OR ${float_type} STREQUAL "ZCOMPLEX")
|
||||
|
||||
foreach (u_source ${U_SOURCES})
|
||||
@@ -197,6 +198,13 @@ foreach (float_type ${FLOAT_TYPES})
|
||||
endif ()
|
||||
endforeach ()
|
||||
|
||||
if (BUILD_BFLOAT16)
|
||||
if (USE_THREAD)
|
||||
GenerateNamedObjects("sbgemv_thread.c" "" "gemv_thread_n" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("sbgemv_thread.c" "TRANSA" "gemv_thread_t" false "" "" false "BFLOAT16")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if ( BUILD_COMPLEX AND NOT BUILD_SINGLE)
|
||||
if (USE_THREAD)
|
||||
GenerateNamedObjects("gemv_thread.c" "" "gemv_thread_n" false "" "" false "SINGLE")
|
||||
|
||||
@@ -12,6 +12,12 @@ foreach (GEMM_DEFINE ${GEMM_DEFINES})
|
||||
if (USE_THREAD AND NOT USE_SIMPLE_THREADED_LEVEL3)
|
||||
GenerateNamedObjects("gemm.c" "${GEMM_DEFINE};THREADED_LEVEL3" "gemm_thread_${GEMM_DEFINE_LC}" 0)
|
||||
endif ()
|
||||
if (BUILD_BFLOAT16)
|
||||
GenerateNamedObjects("gemm.c" "${GEMM_DEFINE}" "gemm_${GEMM_DEFINE_LC}" 0 "" "" false "BFLOAT16")
|
||||
if (USE_THREAD AND NOT USE_SIMPLE_THREADED_LEVEL3)
|
||||
GenerateNamedObjects("gemm.c" "${GEMM_DEFINE};THREADED_LEVEL3" "gemm_thread_${GEMM_DEFINE_LC}" 0 "" "" false "BFLOAT16")
|
||||
endif ()
|
||||
endif ()
|
||||
endforeach ()
|
||||
|
||||
if ( BUILD_COMPLEX16 AND NOT BUILD_DOUBLE)
|
||||
|
||||
@@ -333,7 +333,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n,
|
||||
#else
|
||||
for(jjs = js; jjs < js + min_j; jjs += min_jj){
|
||||
min_jj = min_j + js - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
|
||||
@@ -367,7 +367,7 @@ static int inner_thread(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n,
|
||||
/* Split local region of B into parts */
|
||||
for(jjs = js; jjs < MIN(n_to, js + div_n); jjs += min_jj){
|
||||
min_jj = MIN(n_to, js + div_n) - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
|
||||
@@ -138,7 +138,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
|
||||
for(jjs = js; jjs < js + min_j; jjs += min_jj){
|
||||
min_jj = min_j + js - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
@@ -215,7 +215,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
|
||||
for(jjs = js; jjs < js + min_j; jjs += min_jj){
|
||||
min_jj = min_j + js - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
@@ -320,7 +320,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
|
||||
for(jjs = js; jjs < js + min_j; jjs += min_jj){
|
||||
min_jj = min_j + js - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
@@ -399,7 +399,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
|
||||
for(jjs = js; jjs < js + min_j; jjs += min_jj){
|
||||
min_jj = min_j + js - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
|
||||
@@ -122,7 +122,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
|
||||
for(jjs = 0; jjs < ls - js; jjs += min_jj){
|
||||
min_jj = ls - js - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
@@ -146,7 +146,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
|
||||
for(jjs = 0; jjs < min_l; jjs += min_jj){
|
||||
min_jj = min_l - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
@@ -203,7 +203,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
|
||||
for(jjs = js; jjs < js + min_j; jjs += min_jj){
|
||||
min_jj = min_j + js - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
@@ -258,7 +258,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
|
||||
for(jjs = 0; jjs < min_l; jjs += min_jj){
|
||||
min_jj = min_l - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
@@ -283,7 +283,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
|
||||
for(jjs = 0; jjs < js - ls - min_l; jjs += min_jj){
|
||||
min_jj = js - ls - min_l - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
@@ -344,7 +344,7 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
|
||||
for(jjs = js; jjs < js + min_j; jjs += min_jj){
|
||||
min_jj = min_j + js - jjs;
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
#if defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
/* the current AVX512 s/d/c/z GEMM kernel requires n>=6*GEMM_UNROLL_N to achieve the best performance */
|
||||
if (min_jj >= 6*GEMM_UNROLL_N) min_jj = 6*GEMM_UNROLL_N;
|
||||
#else
|
||||
|
||||
@@ -49,6 +49,8 @@ GenerateNamedObjects("openblas_get_config.c;openblas_get_parallel.c" "" "" 0 ""
|
||||
if (DYNAMIC_ARCH)
|
||||
if (ARM64)
|
||||
list(APPEND COMMON_SOURCES dynamic_arm64.c)
|
||||
elseif (POWER)
|
||||
list(APPEND COMMON_SOURCES dynamic_power.c)
|
||||
else ()
|
||||
list(APPEND COMMON_SOURCES dynamic.c)
|
||||
endif ()
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
#include <stdlib.h>
|
||||
#include "common.h"
|
||||
|
||||
#if defined(OS_CYGWIN_NT) && !defined(unlikely)
|
||||
#if !defined(unlikely)
|
||||
#ifdef __GNUC__
|
||||
#define unlikely(x) __builtin_expect(!!(x), 0)
|
||||
#else
|
||||
@@ -391,8 +391,9 @@ int blas_thread_init(void){
|
||||
|
||||
int exec_blas_async(BLASLONG pos, blas_queue_t *queue){
|
||||
|
||||
#if defined(SMP_SERVER) && defined(OS_CYGWIN_NT)
|
||||
#if defined(SMP_SERVER)
|
||||
// Handle lazy re-init of the thread-pool after a POSIX fork
|
||||
// on Cygwin or as delayed init when a static library is used
|
||||
if (unlikely(blas_server_avail == 0)) blas_thread_init();
|
||||
#endif
|
||||
|
||||
|
||||
+55
-6
@@ -624,7 +624,7 @@ static gotoblas_t *get_coretype(void){
|
||||
return &gotoblas_NEHALEM;
|
||||
}
|
||||
}
|
||||
if (model == 10) {
|
||||
if (model == 10 || model == 12){
|
||||
// Ice Lake SP
|
||||
if(support_avx512_bf16())
|
||||
return &gotoblas_COOPERLAKE;
|
||||
@@ -639,12 +639,12 @@ static gotoblas_t *get_coretype(void){
|
||||
openblas_warning(FALLBACK_VERBOSE, NEHALEM_FALLBACK);
|
||||
return &gotoblas_NEHALEM;
|
||||
}
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
case 7:
|
||||
if (model == 10) // Goldmont Plus
|
||||
return &gotoblas_NEHALEM;
|
||||
if (model == 14) {
|
||||
if (model == 13 || model == 14) {
|
||||
// Ice Lake
|
||||
if (support_avx512())
|
||||
return &gotoblas_SKYLAKEX;
|
||||
@@ -661,9 +661,8 @@ static gotoblas_t *get_coretype(void){
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
case 9:
|
||||
case 8:
|
||||
if (model == 12) { // Tiger Lake
|
||||
if (model == 12 || model == 13) { // Tiger Lake
|
||||
if (support_avx512())
|
||||
return &gotoblas_SKYLAKEX;
|
||||
if(support_avx2()){
|
||||
@@ -689,6 +688,50 @@ static gotoblas_t *get_coretype(void){
|
||||
return &gotoblas_NEHALEM; //OS doesn't support AVX. Use old kernels.
|
||||
}
|
||||
}
|
||||
if (model == 15){ // Sapphire Rapids
|
||||
if(support_avx512_bf16())
|
||||
return &gotoblas_COOPERLAKE;
|
||||
if (support_avx512())
|
||||
return &gotoblas_SKYLAKEX;
|
||||
if(support_avx2())
|
||||
return &gotoblas_HASWELL;
|
||||
if(support_avx()) {
|
||||
openblas_warning(FALLBACK_VERBOSE, SANDYBRIDGE_FALLBACK);
|
||||
return &gotoblas_SANDYBRIDGE;
|
||||
} else {
|
||||
openblas_warning(FALLBACK_VERBOSE, NEHALEM_FALLBACK);
|
||||
return &gotoblas_NEHALEM;
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
|
||||
|
||||
case 9:
|
||||
if (model == 7 || model == 10) { // Alder Lake
|
||||
if(support_avx2()){
|
||||
openblas_warning(FALLBACK_VERBOSE, HASWELL_FALLBACK);
|
||||
return &gotoblas_HASWELL;
|
||||
}
|
||||
if(support_avx()) {
|
||||
openblas_warning(FALLBACK_VERBOSE, SANDYBRIDGE_FALLBACK);
|
||||
return &gotoblas_SANDYBRIDGE;
|
||||
} else {
|
||||
openblas_warning(FALLBACK_VERBOSE, NEHALEM_FALLBACK);
|
||||
return &gotoblas_NEHALEM;
|
||||
}
|
||||
}
|
||||
if (model == 14 ) { // Kaby Lake, Coffee Lake
|
||||
if(support_avx2())
|
||||
return &gotoblas_HASWELL;
|
||||
if(support_avx()) {
|
||||
openblas_warning(FALLBACK_VERBOSE, SANDYBRIDGE_FALLBACK);
|
||||
return &gotoblas_SANDYBRIDGE;
|
||||
} else {
|
||||
openblas_warning(FALLBACK_VERBOSE, NEHALEM_FALLBACK);
|
||||
return &gotoblas_NEHALEM; //OS doesn't support AVX. Use old kernels.
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
case 10:
|
||||
if (model == 5 || model == 6) {
|
||||
if(support_avx2())
|
||||
@@ -1018,7 +1061,13 @@ void gotoblas_dynamic_init(void) {
|
||||
#ifdef ARCH_X86
|
||||
if (gotoblas == NULL) gotoblas = &gotoblas_KATMAI;
|
||||
#else
|
||||
if (gotoblas == NULL) gotoblas = &gotoblas_PRESCOTT;
|
||||
if (gotoblas == NULL) {
|
||||
if (support_avx512_bf16()) gotoblas = &gotoblas_COOPERLAKE;
|
||||
else if (support_avx512()) gotoblas = &gotoblas_SKYLAKEX;
|
||||
else if (support_avx2()) gotoblas = &gotoblas_HASWELL;
|
||||
else if (support_avx()) gotoblas = &gotoblas_SANDYBRIDGE;
|
||||
else gotoblas = &gotoblas_PRESCOTT;
|
||||
}
|
||||
/* sanity check, if 64bit pointer we can't have a 32 bit cpu */
|
||||
if (sizeof(void*) == 8) {
|
||||
if (gotoblas == &gotoblas_KATMAI ||
|
||||
|
||||
@@ -6,10 +6,6 @@ extern gotoblas_t gotoblas_POWER8;
|
||||
#if (!defined __GNUC__) || ( __GNUC__ >= 6)
|
||||
extern gotoblas_t gotoblas_POWER9;
|
||||
#endif
|
||||
//#if (!defined __GNUC__) || ( __GNUC__ >= 11) \
|
||||
// || (__GNUC__ == 10 && __GNUC_MINOR__ >= 2)
|
||||
//#define HAVE_P10_SUPPORT 1
|
||||
//#endif
|
||||
#ifdef HAVE_P10_SUPPORT
|
||||
extern gotoblas_t gotoblas_POWER10;
|
||||
#endif
|
||||
|
||||
@@ -1,38 +1,7 @@
|
||||
#include "common.h"
|
||||
#include "cpuid_zarch.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
// Guard the use of getauxval() on glibc version >= 2.16
|
||||
#ifdef __GLIBC__
|
||||
#include <features.h>
|
||||
#if __GLIBC_PREREQ(2, 16)
|
||||
#include <sys/auxv.h>
|
||||
#define HAVE_GETAUXVAL 1
|
||||
|
||||
static unsigned long get_hwcap(void)
|
||||
{
|
||||
unsigned long hwcap = getauxval(AT_HWCAP);
|
||||
char *maskenv;
|
||||
|
||||
// honor requests for not using specific CPU features in LD_HWCAP_MASK
|
||||
maskenv = getenv("LD_HWCAP_MASK");
|
||||
if (maskenv)
|
||||
hwcap &= strtoul(maskenv, NULL, 0);
|
||||
|
||||
return hwcap;
|
||||
// note that a missing auxval is interpreted as no capabilities
|
||||
// available, which is safe.
|
||||
}
|
||||
|
||||
#else // __GLIBC_PREREQ(2, 16)
|
||||
#warn "Cannot detect SIMD support in Z13 or newer architectures since glibc is older than 2.16"
|
||||
|
||||
static unsigned long get_hwcap(void) {
|
||||
// treat missing support for getauxval() as no capabilities available,
|
||||
// which is safe.
|
||||
return 0;
|
||||
}
|
||||
#endif // __GLIBC_PREREQ(2, 16)
|
||||
#endif // __GLIBC
|
||||
|
||||
extern gotoblas_t gotoblas_ZARCH_GENERIC;
|
||||
#ifdef DYN_Z13
|
||||
@@ -44,25 +13,19 @@ extern gotoblas_t gotoblas_Z14;
|
||||
|
||||
#define NUM_CORETYPES 4
|
||||
|
||||
extern int openblas_verbose();
|
||||
extern void openblas_warning(int verbose, const char* msg);
|
||||
|
||||
static char* corename[] = {
|
||||
"unknown",
|
||||
"Z13",
|
||||
"Z14",
|
||||
"ZARCH_GENERIC",
|
||||
};
|
||||
|
||||
char* gotoblas_corename(void) {
|
||||
#ifdef DYN_Z13
|
||||
if (gotoblas == &gotoblas_Z13) return corename[1];
|
||||
if (gotoblas == &gotoblas_Z13) return cpuname[CPU_Z13];
|
||||
#endif
|
||||
#ifdef DYN_Z14
|
||||
if (gotoblas == &gotoblas_Z14) return corename[2];
|
||||
if (gotoblas == &gotoblas_Z14) return cpuname[CPU_Z14];
|
||||
#endif
|
||||
if (gotoblas == &gotoblas_ZARCH_GENERIC) return corename[3];
|
||||
if (gotoblas == &gotoblas_ZARCH_GENERIC) return cpuname[CPU_GENERIC];
|
||||
|
||||
return corename[0];
|
||||
return "unknown";
|
||||
}
|
||||
|
||||
#ifndef HWCAP_S390_VXE
|
||||
@@ -79,25 +42,28 @@ char* gotoblas_corename(void) {
|
||||
*/
|
||||
static gotoblas_t* get_coretype(void) {
|
||||
|
||||
unsigned long hwcap __attribute__((unused)) = get_hwcap();
|
||||
int cpu = detect();
|
||||
|
||||
#ifdef DYN_Z14
|
||||
switch(cpu) {
|
||||
// z14 and z15 systems: exploit Vector Facility (SIMD) and
|
||||
// Vector-Enhancements Facility 1 (float SIMD instructions), if present.
|
||||
if ((hwcap & HWCAP_S390_VX) && (hwcap & HWCAP_S390_VXE))
|
||||
case CPU_Z14:
|
||||
#ifdef DYN_Z14
|
||||
return &gotoblas_Z14;
|
||||
#endif
|
||||
|
||||
#ifdef DYN_Z13
|
||||
// z13: Vector Facility (SIMD for double)
|
||||
if (hwcap & HWCAP_S390_VX)
|
||||
case CPU_Z13:
|
||||
#ifdef DYN_Z13
|
||||
return &gotoblas_Z13;
|
||||
#endif
|
||||
|
||||
default:
|
||||
// fallback in case of missing compiler support, systems before z13, or
|
||||
// when the OS does not advertise support for the Vector Facility (e.g.,
|
||||
// missing support in the OS kernel)
|
||||
return &gotoblas_ZARCH_GENERIC;
|
||||
return &gotoblas_ZARCH_GENERIC;
|
||||
}
|
||||
}
|
||||
|
||||
static gotoblas_t* force_coretype(char* coretype) {
|
||||
@@ -108,28 +74,28 @@ static gotoblas_t* force_coretype(char* coretype) {
|
||||
|
||||
for (i = 0; i < NUM_CORETYPES; i++)
|
||||
{
|
||||
if (!strncasecmp(coretype, corename[i], 20))
|
||||
if (!strncasecmp(coretype, cpuname[i], 20))
|
||||
{
|
||||
found = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (found == 1) {
|
||||
if (found == CPU_Z13) {
|
||||
#ifdef DYN_Z13
|
||||
return &gotoblas_Z13;
|
||||
#else
|
||||
openblas_warning(1, "Z13 support not compiled in");
|
||||
return NULL;
|
||||
#endif
|
||||
} else if (found == 2) {
|
||||
} else if (found == CPU_Z14) {
|
||||
#ifdef DYN_Z14
|
||||
return &gotoblas_Z14;
|
||||
#else
|
||||
openblas_warning(1, "Z14 support not compiled in");
|
||||
return NULL;
|
||||
#endif
|
||||
} else if (found == 3) {
|
||||
} else if (found == CPU_GENERIC) {
|
||||
return &gotoblas_ZARCH_GENERIC;
|
||||
}
|
||||
|
||||
@@ -155,6 +121,11 @@ void gotoblas_dynamic_init(void) {
|
||||
else
|
||||
{
|
||||
gotoblas = get_coretype();
|
||||
if (openblas_verbose() >= 2) {
|
||||
snprintf(coremsg, sizeof(coremsg), "Choosing kernels based on getauxval(AT_HWCAP)=0x%lx\n",
|
||||
getauxval(AT_HWCAP));
|
||||
openblas_warning(2, coremsg);
|
||||
}
|
||||
}
|
||||
|
||||
if (gotoblas == NULL)
|
||||
@@ -165,9 +136,11 @@ void gotoblas_dynamic_init(void) {
|
||||
}
|
||||
|
||||
if (gotoblas && gotoblas->init) {
|
||||
strncpy(coren, gotoblas_corename(), 20);
|
||||
sprintf(coremsg, "Core: %s\n", coren);
|
||||
openblas_warning(2, coremsg);
|
||||
if (openblas_verbose() >= 2) {
|
||||
strncpy(coren, gotoblas_corename(), 20);
|
||||
sprintf(coremsg, "Core: %s\n", coren);
|
||||
openblas_warning(2, coremsg);
|
||||
}
|
||||
gotoblas->init();
|
||||
}
|
||||
else {
|
||||
|
||||
+236
-5
@@ -73,6 +73,16 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
#include "common.h"
|
||||
|
||||
#ifndef likely
|
||||
#ifdef __GNUC__
|
||||
#define likely(x) __builtin_expect(!!(x), 1)
|
||||
#define unlikely(x) __builtin_expect(!!(x), 0)
|
||||
#else
|
||||
#define likely(x) (x)
|
||||
#define unlikely(x) (x)
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#if defined(USE_TLS) && defined(SMP)
|
||||
#define COMPILE_TLS
|
||||
|
||||
@@ -236,6 +246,14 @@ int get_num_procs(void) {
|
||||
#endif
|
||||
|
||||
if (!nums) nums = sysconf(_SC_NPROCESSORS_CONF);
|
||||
|
||||
#if defined(USE_OPENMP)
|
||||
#if _OPENMP >= 201511
|
||||
nums = omp_get_num_places();
|
||||
#endif
|
||||
return nums;
|
||||
#endif
|
||||
|
||||
#if !defined(OS_LINUX)
|
||||
return nums;
|
||||
#endif
|
||||
@@ -1796,10 +1814,19 @@ int get_num_procs(void) {
|
||||
#endif
|
||||
|
||||
if (!nums) nums = sysconf(_SC_NPROCESSORS_CONF);
|
||||
|
||||
#if defined(USE_OPENMP)
|
||||
/* if (omp_get_proc_bind() != omp_proc_bind_false) */
|
||||
#if _OPENMP >= 201511
|
||||
nums = omp_get_num_places();
|
||||
#endif
|
||||
return nums;
|
||||
#endif
|
||||
|
||||
#if !defined(OS_LINUX)
|
||||
return nums;
|
||||
#endif
|
||||
|
||||
|
||||
#if !defined(__GLIBC_PREREQ)
|
||||
return nums;
|
||||
#else
|
||||
@@ -2060,6 +2087,7 @@ struct release_t {
|
||||
int hugetlb_allocated = 0;
|
||||
|
||||
static struct release_t release_info[NUM_BUFFERS];
|
||||
static struct release_t *new_release_info;
|
||||
static int release_pos = 0;
|
||||
|
||||
#if defined(OS_LINUX) && !defined(NO_WARMUP)
|
||||
@@ -2110,8 +2138,13 @@ static void *alloc_mmap(void *address){
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
LOCK_COMMAND(&alloc_lock);
|
||||
#endif
|
||||
if (likely(release_pos < NUM_BUFFERS)) {
|
||||
release_info[release_pos].address = map_address;
|
||||
release_info[release_pos].func = alloc_mmap_free;
|
||||
} else {
|
||||
new_release_info[release_pos-NUM_BUFFERS].address = map_address;
|
||||
new_release_info[release_pos-NUM_BUFFERS].func = alloc_mmap_free;
|
||||
}
|
||||
release_pos ++;
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
UNLOCK_COMMAND(&alloc_lock);
|
||||
@@ -2274,8 +2307,13 @@ static void *alloc_mmap(void *address){
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
LOCK_COMMAND(&alloc_lock);
|
||||
#endif
|
||||
if (likely(release_pos < NUM_BUFFERS)) {
|
||||
release_info[release_pos].address = map_address;
|
||||
release_info[release_pos].func = alloc_mmap_free;
|
||||
} else {
|
||||
new_release_info[release_pos-NUM_BUFFERS].address = map_address;
|
||||
new_release_info[release_pos-NUM_BUFFERS].func = alloc_mmap_free;
|
||||
}
|
||||
release_pos ++;
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
UNLOCK_COMMAND(&alloc_lock);
|
||||
@@ -2307,8 +2345,13 @@ static void *alloc_malloc(void *address){
|
||||
if (map_address == (void *)NULL) map_address = (void *)-1;
|
||||
|
||||
if (map_address != (void *)-1) {
|
||||
if (likely(release_pos < NUM_BUFFERS)) {
|
||||
release_info[release_pos].address = map_address;
|
||||
release_info[release_pos].func = alloc_malloc_free;
|
||||
} else {
|
||||
new_release_info[release_pos-NUM_BUFFERS].address = map_address;
|
||||
new_release_info[release_pos-NUM_BUFFERS].func = alloc_malloc_free;
|
||||
}
|
||||
release_pos ++;
|
||||
}
|
||||
|
||||
@@ -2341,8 +2384,13 @@ static void *alloc_qalloc(void *address){
|
||||
if (map_address == (void *)NULL) map_address = (void *)-1;
|
||||
|
||||
if (map_address != (void *)-1) {
|
||||
if (likely(release_pos < NUM_BUFFERS)) {
|
||||
release_info[release_pos].address = map_address;
|
||||
release_info[release_pos].func = alloc_qalloc_free;
|
||||
} else {
|
||||
new_release_info[release_pos-NUM_BUFFERS].address = map_address;
|
||||
new_release_info[release_pos-NUM_BUFFERS].func = alloc_qalloc_free;
|
||||
}
|
||||
release_pos ++;
|
||||
}
|
||||
|
||||
@@ -2370,8 +2418,13 @@ static void *alloc_windows(void *address){
|
||||
if (map_address == (void *)NULL) map_address = (void *)-1;
|
||||
|
||||
if (map_address != (void *)-1) {
|
||||
if (likely(release_pos < NUM_BUFFERS)) {
|
||||
release_info[release_pos].address = map_address;
|
||||
release_info[release_pos].func = alloc_windows_free;
|
||||
} else {
|
||||
new_release_info[release_pos-NUM_BUFFERS].address = map_address;
|
||||
new_release_info[release_pos-NUM_BUFFERS].func = alloc_windows_free;
|
||||
}
|
||||
release_pos ++;
|
||||
}
|
||||
|
||||
@@ -2414,9 +2467,15 @@ static void *alloc_devicedirver(void *address){
|
||||
fd, 0);
|
||||
|
||||
if (map_address != (void *)-1) {
|
||||
if (likely(release_pos < NUM_BUFFERS)) {
|
||||
release_info[release_pos].address = map_address;
|
||||
release_info[release_pos].attr = fd;
|
||||
release_info[release_pos].func = alloc_devicedirver_free;
|
||||
} else {
|
||||
new_release_info[release_pos-NUM_BUFFERS].address = map_address;
|
||||
new_release_info[release_pos-NUM_BUFFERS].attr = fd;
|
||||
new_release_info[release_pos-NUM_BUFFERS].func = alloc_devicedirver_free;
|
||||
}
|
||||
release_pos ++;
|
||||
}
|
||||
|
||||
@@ -2450,9 +2509,15 @@ static void *alloc_shm(void *address){
|
||||
|
||||
shmctl(shmid, IPC_RMID, 0);
|
||||
|
||||
if (likely(release_pos < NUM_BUFFERS)) {
|
||||
release_info[release_pos].address = map_address;
|
||||
release_info[release_pos].attr = shmid;
|
||||
release_info[release_pos].func = alloc_shm_free;
|
||||
} else {
|
||||
new_release_info[release_pos-NUM_BUFFERS].address = map_address;
|
||||
new_release_info[release_pos-NUM_BUFFERS].attr = shmid;
|
||||
new_release_info[release_pos-NUM_BUFFERS].func = alloc_shm_free;
|
||||
}
|
||||
release_pos ++;
|
||||
}
|
||||
|
||||
@@ -2556,8 +2621,13 @@ static void *alloc_hugetlb(void *address){
|
||||
#endif
|
||||
|
||||
if (map_address != (void *)-1){
|
||||
if (likely(release_pos < NUM_BUFFERS)) {
|
||||
release_info[release_pos].address = map_address;
|
||||
release_info[release_pos].func = alloc_hugetlb_free;
|
||||
} else {
|
||||
new_release_info[release_pos-NUM_BUFFERS].address = map_address;
|
||||
new_release_info[release_pos-NUM_BUFFERS].func = alloc_hugetlb_free;
|
||||
}
|
||||
release_pos ++;
|
||||
}
|
||||
|
||||
@@ -2604,9 +2674,15 @@ static void *alloc_hugetlbfile(void *address){
|
||||
fd, 0);
|
||||
|
||||
if (map_address != (void *)-1) {
|
||||
if (likely(release_pos < NUM_BUFFERS)) {
|
||||
release_info[release_pos].address = map_address;
|
||||
release_info[release_pos].attr = fd;
|
||||
release_info[release_pos].func = alloc_hugetlbfile_free;
|
||||
} else {
|
||||
new_release_info[release_pos-NUM_BUFFERS].address = map_address;
|
||||
new_release_info[release_pos-NUM_BUFFERS].attr = fd;
|
||||
new_release_info[release_pos-NUM_BUFFERS].func = alloc_hugetlbfile_free;
|
||||
}
|
||||
release_pos ++;
|
||||
}
|
||||
|
||||
@@ -2636,8 +2712,25 @@ static volatile struct {
|
||||
|
||||
} memory[NUM_BUFFERS];
|
||||
|
||||
static int memory_initialized = 0;
|
||||
struct newmemstruct
|
||||
{
|
||||
BLASULONG lock;
|
||||
void *addr;
|
||||
#if defined(WHEREAMI) && !defined(USE_OPENMP)
|
||||
int pos;
|
||||
#endif
|
||||
int used;
|
||||
#ifndef __64BIT__
|
||||
char dummy[48];
|
||||
#else
|
||||
char dummy[40];
|
||||
#endif
|
||||
|
||||
};
|
||||
static volatile struct newmemstruct *newmemory;
|
||||
|
||||
static int memory_initialized = 0;
|
||||
static int memory_overflowed = 0;
|
||||
/* Memory allocation routine */
|
||||
/* procpos ... indicates where it comes from */
|
||||
/* 0 : Level 3 functions */
|
||||
@@ -2646,6 +2739,8 @@ static int memory_initialized = 0;
|
||||
|
||||
void *blas_memory_alloc(int procpos){
|
||||
|
||||
int i;
|
||||
|
||||
int position;
|
||||
#if defined(WHEREAMI) && !defined(USE_OPENMP)
|
||||
int mypos = 0;
|
||||
@@ -2776,6 +2871,25 @@ void *blas_memory_alloc(int procpos){
|
||||
position ++;
|
||||
|
||||
} while (position < NUM_BUFFERS);
|
||||
|
||||
if (memory_overflowed) {
|
||||
|
||||
do {
|
||||
RMB;
|
||||
#if defined(USE_OPENMP)
|
||||
if (!newmemory[position-NUM_BUFFERS].used) {
|
||||
blas_lock(&newmemory[position-NUM_BUFFERS].lock);
|
||||
#endif
|
||||
if (!newmemory[position-NUM_BUFFERS].used) goto allocation2;
|
||||
|
||||
#if defined(USE_OPENMP)
|
||||
blas_unlock(&newmemory[position-NUM_BUFFERS].lock);
|
||||
}
|
||||
#endif
|
||||
position ++;
|
||||
|
||||
} while (position < 512+NUM_BUFFERS);
|
||||
}
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
UNLOCK_COMMAND(&alloc_lock);
|
||||
#endif
|
||||
@@ -2803,7 +2917,7 @@ void *blas_memory_alloc(int procpos){
|
||||
|
||||
func = &memoryalloc[0];
|
||||
|
||||
while ((func != NULL) && (map_address == (void *) -1)) {
|
||||
while ((*func != NULL) && (map_address == (void *) -1)) {
|
||||
|
||||
map_address = (*func)((void *)base_address);
|
||||
|
||||
@@ -2883,6 +2997,96 @@ void *blas_memory_alloc(int procpos){
|
||||
return (void *)memory[position].addr;
|
||||
|
||||
error:
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
LOCK_COMMAND(&alloc_lock);
|
||||
#endif
|
||||
if (memory_overflowed) goto terminate;
|
||||
fprintf(stderr,"OpenBLAS warning: precompiled NUM_THREADS exceeded, adding auxiliary array for thread metadata.\n");
|
||||
memory_overflowed=1;
|
||||
new_release_info = (struct release_t*) malloc(512*sizeof(struct release_t));
|
||||
newmemory = (struct newmemstruct*) malloc(512*sizeof(struct newmemstruct));
|
||||
for (i = 0; i < 512; i++) {
|
||||
newmemory[i].addr = (void *)0;
|
||||
#if defined(WHEREAMI) && !defined(USE_OPENMP)
|
||||
newmemory[i].pos = -1;
|
||||
#endif
|
||||
newmemory[i].used = 0;
|
||||
newmemory[i].lock = 0;
|
||||
}
|
||||
|
||||
allocation2:
|
||||
newmemory[position-NUM_BUFFERS].used = 1;
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
UNLOCK_COMMAND(&alloc_lock);
|
||||
#else
|
||||
blas_unlock(&newmemory[position-NUM_BUFFERS].lock);
|
||||
#endif
|
||||
do {
|
||||
#ifdef DEBUG
|
||||
printf("Allocation Start : %lx\n", base_address);
|
||||
#endif
|
||||
|
||||
map_address = (void *)-1;
|
||||
|
||||
func = &memoryalloc[0];
|
||||
|
||||
while ((*func != NULL) && (map_address == (void *) -1)) {
|
||||
|
||||
map_address = (*func)((void *)base_address);
|
||||
|
||||
#ifdef ALLOC_DEVICEDRIVER
|
||||
if ((*func == alloc_devicedirver) && (map_address == (void *)-1)) {
|
||||
fprintf(stderr, "OpenBLAS Warning ... Physically contiguous allocation was failed.\n");
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ALLOC_HUGETLBFILE
|
||||
if ((*func == alloc_hugetlbfile) && (map_address == (void *)-1)) {
|
||||
#ifndef OS_WINDOWS
|
||||
fprintf(stderr, "OpenBLAS Warning ... HugeTLB(File) allocation was failed.\n");
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
#if (defined ALLOC_SHM) && (defined OS_LINUX || defined OS_AIX || defined __sun__ || defined OS_WINDOWS)
|
||||
if ((*func == alloc_hugetlb) && (map_address != (void *)-1)) hugetlb_allocated = 1;
|
||||
#endif
|
||||
|
||||
func ++;
|
||||
}
|
||||
|
||||
#ifdef DEBUG
|
||||
printf(" Success -> %08lx\n", map_address);
|
||||
#endif
|
||||
if (((BLASLONG) map_address) == -1) base_address = 0UL;
|
||||
|
||||
if (base_address) base_address += BUFFER_SIZE + FIXED_PAGESIZE;
|
||||
|
||||
} while ((BLASLONG)map_address == -1);
|
||||
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
LOCK_COMMAND(&alloc_lock);
|
||||
#endif
|
||||
newmemory[position-NUM_BUFFERS].addr = map_address;
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
UNLOCK_COMMAND(&alloc_lock);
|
||||
#endif
|
||||
|
||||
#ifdef DEBUG
|
||||
printf(" Mapping Succeeded. %p(%d)\n", (void *)newmemory[position-NUM_BUFFERS].addr, position);
|
||||
#endif
|
||||
|
||||
#if defined(WHEREAMI) && !defined(USE_OPENMP)
|
||||
|
||||
if (newmemory[position-NUM_BUFFERS].pos == -1) newmemory[position-NUM_BUFFERS].pos = mypos;
|
||||
|
||||
#endif
|
||||
return (void *)newmemory[position-NUM_BUFFERS].addr;
|
||||
|
||||
terminate:
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
UNLOCK_COMMAND(&alloc_lock);
|
||||
#endif
|
||||
printf("OpenBLAS : Program is Terminated. Because you tried to allocate too many memory regions.\n");
|
||||
printf("This library was built to support a maximum of %d threads - either rebuild OpenBLAS\n", NUM_BUFFERS);
|
||||
printf("with a larger NUM_THREADS value or set the environment variable OPENBLAS_NUM_THREADS to\n");
|
||||
@@ -2907,13 +3111,28 @@ void blas_memory_free(void *free_area){
|
||||
while ((position < NUM_BUFFERS) && (memory[position].addr != free_area))
|
||||
position++;
|
||||
|
||||
if (position >= NUM_BUFFERS) goto error;
|
||||
if (position >= NUM_BUFFERS && !memory_overflowed) goto error;
|
||||
|
||||
#ifdef DEBUG
|
||||
if (memory[position].addr != free_area) goto error;
|
||||
printf(" Position : %d\n", position);
|
||||
#endif
|
||||
if (unlikely(memory_overflowed && position >= NUM_BUFFERS)) {
|
||||
while ((position < NUM_BUFFERS+512) && (newmemory[position-NUM_BUFFERS].addr != free_area))
|
||||
position++;
|
||||
// arm: ensure all writes are finished before other thread takes this memory
|
||||
WMB;
|
||||
|
||||
newmemory[position].used = 0;
|
||||
#if (defined(SMP) || defined(USE_LOCKING)) && !defined(USE_OPENMP)
|
||||
UNLOCK_COMMAND(&alloc_lock);
|
||||
#endif
|
||||
|
||||
#ifdef DEBUG
|
||||
printf("Unmap from overflow area succeeded.\n\n");
|
||||
#endif
|
||||
return;
|
||||
} else {
|
||||
// arm: ensure all writes are finished before other thread takes this memory
|
||||
WMB;
|
||||
|
||||
@@ -2927,7 +3146,7 @@ void blas_memory_free(void *free_area){
|
||||
#endif
|
||||
|
||||
return;
|
||||
|
||||
}
|
||||
error:
|
||||
printf("BLAS : Bad memory unallocation! : %4d %p\n", position, free_area);
|
||||
|
||||
@@ -2962,7 +3181,10 @@ void blas_shutdown(void){
|
||||
LOCK_COMMAND(&alloc_lock);
|
||||
|
||||
for (pos = 0; pos < release_pos; pos ++) {
|
||||
if (likely(pos < NUM_BUFFERS))
|
||||
release_info[pos].func(&release_info[pos]);
|
||||
else
|
||||
new_release_info[pos-NUM_BUFFERS].func(&new_release_info[pos-NUM_BUFFERS]);
|
||||
}
|
||||
|
||||
#ifdef SEEK_ADDRESS
|
||||
@@ -2979,6 +3201,15 @@ void blas_shutdown(void){
|
||||
#endif
|
||||
memory[pos].lock = 0;
|
||||
}
|
||||
if (memory_overflowed)
|
||||
for (pos = 0; pos < 512; pos ++){
|
||||
newmemory[pos].addr = (void *)0;
|
||||
newmemory[pos].used = 0;
|
||||
#if defined(WHEREAMI) && !defined(USE_OPENMP)
|
||||
newmemory[pos].pos = -1;
|
||||
#endif
|
||||
newmemory[pos].lock = 0;
|
||||
}
|
||||
|
||||
UNLOCK_COMMAND(&alloc_lock);
|
||||
|
||||
|
||||
@@ -183,7 +183,7 @@ int get_L2_size(void){
|
||||
defined(CORE_PRESCOTT) || defined(CORE_CORE2) || defined(PENRYN) || defined(DUNNINGTON) || \
|
||||
defined(CORE_NEHALEM) || defined(CORE_SANDYBRIDGE) || defined(ATOM) || defined(GENERIC) || \
|
||||
defined(PILEDRIVER) || defined(HASWELL) || defined(STEAMROLLER) || defined(EXCAVATOR) || \
|
||||
defined(ZEN) || defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
defined(ZEN) || defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
|
||||
cpuid(0x80000006, &eax, &ebx, &ecx, &edx);
|
||||
|
||||
@@ -269,7 +269,7 @@ void blas_set_parameter(void){
|
||||
int factor;
|
||||
#if defined(BULLDOZER) || defined(PILEDRIVER) || defined(SANDYBRIDGE) || defined(NEHALEM) || \
|
||||
defined(HASWELL) || defined(STEAMROLLER) || defined(EXCAVATOR) || defined(ZEN) || \
|
||||
defined(SKYLAKEX) || defined(COOPERLAKE)
|
||||
defined(SKYLAKEX) || defined(COOPERLAKE) || defined(SAPPHIRERAPIDS)
|
||||
int size = 16;
|
||||
#else
|
||||
int size = get_L2_size();
|
||||
@@ -524,6 +524,9 @@ void blas_set_parameter(void){
|
||||
xgemm_p = ((xgemm_p + XGEMM_UNROLL_M - 1)/XGEMM_UNROLL_M) * XGEMM_UNROLL_M;
|
||||
#endif
|
||||
|
||||
#ifdef BUILD_BFLOAT16
|
||||
sbgemm_r = (((BUFFER_SIZE - ((SBGEMM_P * SBGEMM_Q * 4 + GEMM_OFFSET_A + GEMM_ALIGN) & ~GEMM_ALIGN)) / (SBGEMM_Q * 4)) - 15) & ~15;
|
||||
#endif
|
||||
sgemm_r = (((BUFFER_SIZE - ((SGEMM_P * SGEMM_Q * 4 + GEMM_OFFSET_A + GEMM_ALIGN) & ~GEMM_ALIGN)) / (SGEMM_Q * 4)) - 15) & ~15;
|
||||
dgemm_r = (((BUFFER_SIZE - ((DGEMM_P * DGEMM_Q * 8 + GEMM_OFFSET_A + GEMM_ALIGN) & ~GEMM_ALIGN)) / (DGEMM_Q * 8)) - 15) & ~15;
|
||||
cgemm_r = (((BUFFER_SIZE - ((CGEMM_P * CGEMM_Q * 8 + GEMM_OFFSET_A + GEMM_ALIGN) & ~GEMM_ALIGN)) / (CGEMM_Q * 8)) - 15) & ~15;
|
||||
@@ -629,7 +632,9 @@ void blas_set_parameter(void){
|
||||
xgemm_p = 16 * (size + 1);
|
||||
#endif
|
||||
|
||||
#ifdef BUILD_BFLOAT16
|
||||
sbgemm_r = (((BUFFER_SIZE - ((SBGEMM_P * SBGEMM_Q * 4 + GEMM_OFFSET_A + GEMM_ALIGN) & ~GEMM_ALIGN)) / (SBGEMM_Q * 4)) - 15) & ~15;
|
||||
#endif
|
||||
sgemm_r = (((BUFFER_SIZE - ((SGEMM_P * SGEMM_Q * 4 + GEMM_OFFSET_A + GEMM_ALIGN) & ~GEMM_ALIGN)) / (SGEMM_Q * 4)) - 15) & ~15;
|
||||
dgemm_r = (((BUFFER_SIZE - ((DGEMM_P * DGEMM_Q * 8 + GEMM_OFFSET_A + GEMM_ALIGN) & ~GEMM_ALIGN)) / (DGEMM_Q * 8)) - 15) & ~15;
|
||||
cgemm_r = (((BUFFER_SIZE - ((CGEMM_P * CGEMM_Q * 8 + GEMM_OFFSET_A + GEMM_ALIGN) & ~GEMM_ALIGN)) / (CGEMM_Q * 8)) - 15) & ~15;
|
||||
|
||||
@@ -313,6 +313,16 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define FORCE
|
||||
#define FORCE_INTEL
|
||||
#define ARCHITECTURE "X86"
|
||||
#ifdef NO_AVX
|
||||
#define SUBARCHITECTURE "NEHALEM"
|
||||
#define ARCHCONFIG "-DNEHALEM " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2"
|
||||
#define LIBNAME "nehalem"
|
||||
#define CORENAME "NEHALEM"
|
||||
#else
|
||||
#define SUBARCHITECTURE "SANDYBRIDGE"
|
||||
#define ARCHCONFIG "-DSANDYBRIDGE " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
@@ -322,12 +332,23 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define LIBNAME "sandybridge"
|
||||
#define CORENAME "SANDYBRIDGE"
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_HASWELL
|
||||
#define FORCE
|
||||
#define FORCE_INTEL
|
||||
#define ARCHITECTURE "X86"
|
||||
#ifdef NO_AVX2
|
||||
#ifdef NO_AVX
|
||||
#define SUBARCHITECTURE "NEHALEM"
|
||||
#define ARCHCONFIG "-DNEHALEM " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2"
|
||||
#define LIBNAME "nehalem"
|
||||
#define CORENAME "NEHALEM"
|
||||
#else
|
||||
#define SUBARCHITECTURE "SANDYBRIDGE"
|
||||
#define ARCHCONFIG "-DSANDYBRIDGE " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
@@ -336,6 +357,7 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2 -DHAVE_AVX"
|
||||
#define LIBNAME "sandybridge"
|
||||
#define CORENAME "SANDYBRIDGE"
|
||||
#endif
|
||||
#else
|
||||
#define SUBARCHITECTURE "HASWELL"
|
||||
#define ARCHCONFIG "-DHASWELL " \
|
||||
@@ -350,10 +372,31 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_SKYLAKEX
|
||||
#ifdef NO_AVX512
|
||||
#define FORCE
|
||||
#define FORCE_INTEL
|
||||
#define ARCHITECTURE "X86"
|
||||
#ifdef NO_AVX512
|
||||
#ifdef NO_AVX2
|
||||
#ifdef NO_AVX
|
||||
#define SUBARCHITECTURE "NEHALEM"
|
||||
#define ARCHCONFIG "-DNEHALEM " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2"
|
||||
#define LIBNAME "nehalem"
|
||||
#define CORENAME "NEHALEM"
|
||||
#else
|
||||
#define SUBARCHITECTURE "SANDYBRIDGE"
|
||||
#define ARCHCONFIG "-DSANDYBRIDGE " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2 -DHAVE_AVX"
|
||||
#define LIBNAME "sandybridge"
|
||||
#define CORENAME "SANDYBRIDGE"
|
||||
#endif
|
||||
#else
|
||||
#define SUBARCHITECTURE "HASWELL"
|
||||
#define ARCHCONFIG "-DHASWELL " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
@@ -363,10 +406,8 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
"-DHAVE_AVX2 -DHAVE_FMA3 -DFMA3"
|
||||
#define LIBNAME "haswell"
|
||||
#define CORENAME "HASWELL"
|
||||
#endif
|
||||
#else
|
||||
#define FORCE
|
||||
#define FORCE_INTEL
|
||||
#define ARCHITECTURE "X86"
|
||||
#define SUBARCHITECTURE "SKYLAKEX"
|
||||
#define ARCHCONFIG "-DSKYLAKEX " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
@@ -380,10 +421,31 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_COOPERLAKE
|
||||
#ifdef NO_AVX512
|
||||
#define FORCE
|
||||
#define FORCE_INTEL
|
||||
#define ARCHITECTURE "X86"
|
||||
#ifdef NO_AVX512
|
||||
#ifdef NO_AVX2
|
||||
#ifdef NO_AVX
|
||||
#define SUBARCHITECTURE "NEHALEM"
|
||||
#define ARCHCONFIG "-DNEHALEM " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2"
|
||||
#define LIBNAME "nehalem"
|
||||
#define CORENAME "NEHALEM"
|
||||
#else
|
||||
#define SUBARCHITECTURE "SANDYBRIDGE"
|
||||
#define ARCHCONFIG "-DSANDYBRIDGE " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2 -DHAVE_AVX"
|
||||
#define LIBNAME "sandybridge"
|
||||
#define CORENAME "SANDYBRIDGE"
|
||||
#endif
|
||||
#else
|
||||
#define SUBARCHITECTURE "HASWELL"
|
||||
#define ARCHCONFIG "-DHASWELL " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
@@ -393,10 +455,8 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
"-DHAVE_AVX2 -DHAVE_FMA3 -DFMA3"
|
||||
#define LIBNAME "haswell"
|
||||
#define CORENAME "HASWELL"
|
||||
#endif
|
||||
#else
|
||||
#define FORCE
|
||||
#define FORCE_INTEL
|
||||
#define ARCHITECTURE "X86"
|
||||
#define SUBARCHITECTURE "COOPERLAKE"
|
||||
#define ARCHCONFIG "-DCOOPERLAKE " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
@@ -409,6 +469,55 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_SAPPHIRERAPIDS
|
||||
#define FORCE
|
||||
#define FORCE_INTEL
|
||||
#define ARCHITECTURE "X86"
|
||||
#ifdef NO_AVX512
|
||||
#ifdef NO_AVX2
|
||||
#ifdef NO_AVX
|
||||
#define SUBARCHITECTURE "NEHALEM"
|
||||
#define ARCHCONFIG "-DNEHALEM " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2"
|
||||
#define LIBNAME "nehalem"
|
||||
#define CORENAME "NEHALEM"
|
||||
#else
|
||||
#define SUBARCHITECTURE "SANDYBRIDGE"
|
||||
#define ARCHCONFIG "-DSANDYBRIDGE " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2 -DHAVE_AVX"
|
||||
#define LIBNAME "sandybridge"
|
||||
#define CORENAME "SANDYBRIDGE"
|
||||
#endif
|
||||
#else
|
||||
#define SUBARCHITECTURE "HASWELL"
|
||||
#define ARCHCONFIG "-DHASWELL " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2 -DHAVE_AVX " \
|
||||
"-DHAVE_AVX2 -DHAVE_FMA3 -DFMA3"
|
||||
#define LIBNAME "haswell"
|
||||
#define CORENAME "HASWELL"
|
||||
#endif
|
||||
#else
|
||||
#define SUBARCHITECTURE "SAPPHIRERAPIDS"
|
||||
#define ARCHCONFIG "-DSAPPHIRERAPIDS " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2 -DHAVE_AVX " \
|
||||
"-DHAVE_AVX2 -DHAVE_FMA3 -DFMA3 -DHAVE_AVX512VL -DHAVE_AVX512BF16 -march=sapphirerapids"
|
||||
#define LIBNAME "sapphirerapids"
|
||||
#define CORENAME "SAPPHIRERAPIDS"
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_ATOM
|
||||
#define FORCE
|
||||
#define FORCE_INTEL
|
||||
@@ -564,6 +673,16 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define FORCE_INTEL
|
||||
#define ARCHITECTURE "X86"
|
||||
#ifdef NO_AVX2
|
||||
#ifdef NO_AVX
|
||||
#define SUBARCHITECTURE "NEHALEM"
|
||||
#define ARCHCONFIG "-DNEHALEM " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2"
|
||||
#define LIBNAME "nehalem"
|
||||
#define CORENAME "NEHALEM"
|
||||
#else
|
||||
#define SUBARCHITECTURE "SANDYBRIDGE"
|
||||
#define ARCHCONFIG "-DSANDYBRIDGE " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
@@ -572,6 +691,7 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
"-DHAVE_CMOV -DHAVE_MMX -DHAVE_SSE -DHAVE_SSE2 -DHAVE_SSE3 -DHAVE_SSSE3 -DHAVE_SSE4_1 -DHAVE_SSE4_2 -DHAVE_AVX"
|
||||
#define LIBNAME "sandybridge"
|
||||
#define CORENAME "SANDYBRIDGE"
|
||||
#endif
|
||||
#else
|
||||
#define SUBARCHITECTURE "ZEN"
|
||||
#define ARCHCONFIG "-DZEN " \
|
||||
@@ -893,7 +1013,7 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define ARCHCONFIG "-DP5600 " \
|
||||
"-DL1_DATA_SIZE=65536 -DL1_DATA_LINESIZE=32 " \
|
||||
"-DL2_SIZE=1048576 -DL2_LINESIZE=32 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=8 "
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=8 -DNO_MSA"
|
||||
#define LIBNAME "p5600"
|
||||
#define CORENAME "P5600"
|
||||
#else
|
||||
@@ -907,7 +1027,7 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define ARCHCONFIG "-DMIPS1004K " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=32 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=32 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=8 "
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=8 -DNO_MSA"
|
||||
#define LIBNAME "mips1004K"
|
||||
#define CORENAME "MIPS1004K"
|
||||
#else
|
||||
@@ -921,7 +1041,7 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define ARCHCONFIG "-DMIPS24K " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=32 " \
|
||||
"-DL2_SIZE=32768 -DL2_LINESIZE=32 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=8 "
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=8 -DNO_MSA"
|
||||
#define LIBNAME "mips24K"
|
||||
#define CORENAME "MIPS24K"
|
||||
#else
|
||||
@@ -1078,6 +1198,20 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#else
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_ARMV8SVE
|
||||
#define FORCE
|
||||
#define ARCHITECTURE "ARM64"
|
||||
#define SUBARCHITECTURE "ARMV8SVE"
|
||||
#define SUBDIRNAME "arm64"
|
||||
#define ARCHCONFIG "-DARMV8SVE " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=32 " \
|
||||
"-DHAVE_VFPV4 -DHAVE_VFPV3 -DHAVE_VFP -DHAVE_NEON -DHAVE_SVE -DARMV8"
|
||||
#define LIBNAME "armv8sve"
|
||||
#define CORENAME "ARMV8SVE"
|
||||
#endif
|
||||
|
||||
|
||||
#ifdef FORCE_ARMV8
|
||||
#define FORCE
|
||||
@@ -1304,6 +1438,24 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define CORENAME "VORTEX"
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_A64FX
|
||||
#define ARMV8
|
||||
#define FORCE
|
||||
#define ARCHITECTURE "ARM64"
|
||||
#define SUBARCHITECTURE "A64FX"
|
||||
#define SUBDIRNAME "arm64"
|
||||
#define ARCHCONFIG "-DA64FX " \
|
||||
"-DL1_CODE_SIZE=65536 -DL1_CODE_LINESIZE=256 -DL1_CODE_ASSOCIATIVE=8 " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=256 -DL1_DATA_ASSOCIATIVE=8 " \
|
||||
"-DL2_SIZE=8388608 -DL2_LINESIZE=256 -DL2_ASSOCIATIVE=8 " \
|
||||
"-DL3_SIZE=0 -DL3_LINESIZE=0 -DL3_ASSOCIATIVE=0 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 " \
|
||||
"-DHAVE_VFPV4 -DHAVE_VFPV3 -DHAVE_VFP -DHAVE_NEON -DHAVE_SVE -DARMV8"
|
||||
#define LIBNAME "a64fx"
|
||||
#define CORENAME "A64FX"
|
||||
#else
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_ZARCH_GENERIC
|
||||
#define FORCE
|
||||
#define ARCHITECTURE "ZARCH"
|
||||
|
||||
@@ -82,6 +82,7 @@ foreach (CBLAS_FLAG ${CBLAS_FLAGS})
|
||||
GenerateNamedObjects("${BLAS3_SOURCES}" "" "" ${CBLAS_FLAG} "" "" false ${DISABLE_COMPLEX})
|
||||
GenerateNamedObjects("${BLAS3_MANGLED_SOURCES}" "" "" ${CBLAS_FLAG} "" "" false ${MANGLE_COMPLEX})
|
||||
|
||||
GenerateNamedObjects("xerbla.c" "" "xerbla" ${CBLAS_FLAG} "" "" true)
|
||||
#sdsdot, dsdot
|
||||
if (BUILD_SINGLE OR BUILD_DOUBLE)
|
||||
GenerateNamedObjects("sdsdot.c" "" "sdsdot" ${CBLAS_FLAG} "" "" true "SINGLE")
|
||||
@@ -104,6 +105,15 @@ endif ()
|
||||
GenerateNamedObjects("imax.c" "USE_ABS;USE_MIN" "i*amin" ${CBLAS_FLAG})
|
||||
GenerateNamedObjects("imax.c" "USE_MIN" "i*min" ${CBLAS_FLAG})
|
||||
|
||||
if (BUILD_BFLOAT16)
|
||||
GenerateNamedObjects("bf16dot.c" "" "sbdot" ${CBLAS_FLAG} "" "" true "BFLOAT16")
|
||||
GenerateNamedObjects("gemm.c" "" "sbgemm" ${CBLAS_FLAG} "" "" true "BFLOAT16")
|
||||
GenerateNamedObjects("sbgemv.c" "" "sbgemv" ${CBLAS_FLAG} "" "" true "BFLOAT16")
|
||||
GenerateNamedObjects("tobf16.c" "SINGLE_PREC" "sbstobf16" ${CBLAS_FLAG} "" "" true "BFLOAT16")
|
||||
GenerateNamedObjects("tobf16.c" "DOUBLE_PREC" "sbdtobf16" ${CBLAS_FLAG} "" "" true "BFLOAT16")
|
||||
GenerateNamedObjects("bf16to.c" "SINGLE_PREC" "sbf16tos" ${CBLAS_FLAG} "" "" true "BFLOAT16")
|
||||
GenerateNamedObjects("bf16to.c" "DOUBLE_PREC" "dbf16tod" ${CBLAS_FLAG} "" "" true "BFLOAT16")
|
||||
endif ()
|
||||
|
||||
# complex-specific sources
|
||||
foreach (float_type ${FLOAT_TYPES})
|
||||
|
||||
+74
-3
@@ -105,6 +105,55 @@ static int (*gemm[])(blas_arg_t *, BLASLONG *, BLASLONG *, IFLOAT *, IFLOAT *, B
|
||||
#endif
|
||||
};
|
||||
|
||||
#if defined(SMALL_MATRIX_OPT) && !defined(GEMM3M) && !defined(XDOUBLE)
|
||||
#define USE_SMALL_MATRIX_OPT 1
|
||||
#else
|
||||
#define USE_SMALL_MATRIX_OPT 0
|
||||
#endif
|
||||
|
||||
#if USE_SMALL_MATRIX_OPT
|
||||
#ifndef DYNAMIC_ARCH
|
||||
#define SMALL_KERNEL_ADDR(table, idx) ((void *)(table[idx]))
|
||||
#else
|
||||
#define SMALL_KERNEL_ADDR(table, idx) ((void *)(*(uintptr_t *)((char *)gotoblas + (size_t)(table[idx]))))
|
||||
#endif
|
||||
|
||||
|
||||
#ifndef COMPLEX
|
||||
static size_t gemm_small_kernel[] = {
|
||||
GEMM_SMALL_KERNEL_NN, GEMM_SMALL_KERNEL_TN, 0, 0,
|
||||
GEMM_SMALL_KERNEL_NT, GEMM_SMALL_KERNEL_TT, 0, 0,
|
||||
};
|
||||
|
||||
|
||||
static size_t gemm_small_kernel_b0[] = {
|
||||
GEMM_SMALL_KERNEL_B0_NN, GEMM_SMALL_KERNEL_B0_TN, 0, 0,
|
||||
GEMM_SMALL_KERNEL_B0_NT, GEMM_SMALL_KERNEL_B0_TT, 0, 0,
|
||||
};
|
||||
|
||||
#define GEMM_SMALL_KERNEL_B0(idx) (int (*)(BLASLONG, BLASLONG, BLASLONG, IFLOAT *, BLASLONG, FLOAT, IFLOAT *, BLASLONG, FLOAT *, BLASLONG)) SMALL_KERNEL_ADDR(gemm_small_kernel_b0, (idx))
|
||||
#define GEMM_SMALL_KERNEL(idx) (int (*)(BLASLONG, BLASLONG, BLASLONG, IFLOAT *, BLASLONG, FLOAT, IFLOAT *, BLASLONG, FLOAT, FLOAT *, BLASLONG)) SMALL_KERNEL_ADDR(gemm_small_kernel, (idx))
|
||||
#else
|
||||
|
||||
static size_t zgemm_small_kernel[] = {
|
||||
GEMM_SMALL_KERNEL_NN, GEMM_SMALL_KERNEL_TN, GEMM_SMALL_KERNEL_RN, GEMM_SMALL_KERNEL_CN,
|
||||
GEMM_SMALL_KERNEL_NT, GEMM_SMALL_KERNEL_TT, GEMM_SMALL_KERNEL_RT, GEMM_SMALL_KERNEL_CT,
|
||||
GEMM_SMALL_KERNEL_NR, GEMM_SMALL_KERNEL_TR, GEMM_SMALL_KERNEL_RR, GEMM_SMALL_KERNEL_CR,
|
||||
GEMM_SMALL_KERNEL_NC, GEMM_SMALL_KERNEL_TC, GEMM_SMALL_KERNEL_RC, GEMM_SMALL_KERNEL_CC,
|
||||
};
|
||||
|
||||
static size_t zgemm_small_kernel_b0[] = {
|
||||
GEMM_SMALL_KERNEL_B0_NN, GEMM_SMALL_KERNEL_B0_TN, GEMM_SMALL_KERNEL_B0_RN, GEMM_SMALL_KERNEL_B0_CN,
|
||||
GEMM_SMALL_KERNEL_B0_NT, GEMM_SMALL_KERNEL_B0_TT, GEMM_SMALL_KERNEL_B0_RT, GEMM_SMALL_KERNEL_B0_CT,
|
||||
GEMM_SMALL_KERNEL_B0_NR, GEMM_SMALL_KERNEL_B0_TR, GEMM_SMALL_KERNEL_B0_RR, GEMM_SMALL_KERNEL_B0_CR,
|
||||
GEMM_SMALL_KERNEL_B0_NC, GEMM_SMALL_KERNEL_B0_TC, GEMM_SMALL_KERNEL_B0_RC, GEMM_SMALL_KERNEL_B0_CC,
|
||||
};
|
||||
|
||||
#define ZGEMM_SMALL_KERNEL(idx) (int (*)(BLASLONG, BLASLONG, BLASLONG, FLOAT *, BLASLONG, FLOAT , FLOAT, FLOAT *, BLASLONG, FLOAT , FLOAT, FLOAT *, BLASLONG)) SMALL_KERNEL_ADDR(zgemm_small_kernel, (idx))
|
||||
#define ZGEMM_SMALL_KERNEL_B0(idx) (int (*)(BLASLONG, BLASLONG, BLASLONG, FLOAT *, BLASLONG, FLOAT , FLOAT, FLOAT *, BLASLONG, FLOAT *, BLASLONG)) SMALL_KERNEL_ADDR(zgemm_small_kernel_b0, (idx))
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifndef CBLAS
|
||||
|
||||
void NAME(char *TRANSA, char *TRANSB,
|
||||
@@ -224,8 +273,8 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_TRANSPOSE TransA, enum CBLAS_TRANS
|
||||
blasint m, blasint n, blasint k,
|
||||
#ifndef COMPLEX
|
||||
FLOAT alpha,
|
||||
FLOAT *a, blasint lda,
|
||||
FLOAT *b, blasint ldb,
|
||||
IFLOAT *a, blasint lda,
|
||||
IFLOAT *b, blasint ldb,
|
||||
FLOAT beta,
|
||||
FLOAT *c, blasint ldc) {
|
||||
#else
|
||||
@@ -277,7 +326,7 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_TRANSPOSE TransA, enum CBLAS_TRANS
|
||||
|
||||
PRINT_DEBUG_CNAME;
|
||||
|
||||
#if !defined(COMPLEX) && !defined(DOUBLE) && defined(USE_SGEMM_KERNEL_DIRECT)
|
||||
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && defined(USE_SGEMM_KERNEL_DIRECT)
|
||||
#ifdef DYNAMIC_ARCH
|
||||
if (support_avx512() )
|
||||
#endif
|
||||
@@ -417,6 +466,28 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_TRANSPOSE TransA, enum CBLAS_TRANS
|
||||
|
||||
FUNCTION_PROFILE_START();
|
||||
|
||||
#if USE_SMALL_MATRIX_OPT
|
||||
#if !defined(COMPLEX)
|
||||
if(GEMM_SMALL_MATRIX_PERMIT(transa, transb, args.m, args.n, args.k, *(FLOAT *)(args.alpha), *(FLOAT *)(args.beta))){
|
||||
if(*(FLOAT *)(args.beta) == 0.0){
|
||||
(GEMM_SMALL_KERNEL_B0((transb << 2) | transa))(args.m, args.n, args.k, args.a, args.lda, *(FLOAT *)(args.alpha), args.b, args.ldb, args.c, args.ldc);
|
||||
}else{
|
||||
(GEMM_SMALL_KERNEL((transb << 2) | transa))(args.m, args.n, args.k, args.a, args.lda, *(FLOAT *)(args.alpha), args.b, args.ldb, *(FLOAT *)(args.beta), args.c, args.ldc);
|
||||
}
|
||||
return;
|
||||
}
|
||||
#else
|
||||
if(GEMM_SMALL_MATRIX_PERMIT(transa, transb, args.m, args.n, args.k, alpha[0], alpha[1], beta[0], beta[1])){
|
||||
if(beta[0] == 0.0 && beta[1] == 0.0){
|
||||
(ZGEMM_SMALL_KERNEL_B0((transb << 2) | transa))(args.m, args.n, args.k, args.a, args.lda, alpha[0], alpha[1], args.b, args.ldb, args.c, args.ldc);
|
||||
}else{
|
||||
(ZGEMM_SMALL_KERNEL((transb << 2) | transa))(args.m, args.n, args.k, args.a, args.lda, alpha[0], alpha[1], args.b, args.ldb, beta[0], beta[1], args.c, args.ldc);
|
||||
}
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
buffer = (XFLOAT *)blas_memory_alloc(0);
|
||||
|
||||
sa = (XFLOAT *)((BLASLONG)buffer +GEMM_OFFSET_A);
|
||||
|
||||
@@ -188,12 +188,6 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_UPLO Uplo,
|
||||
|
||||
if (n == 0) return;
|
||||
|
||||
if (incx == 1 && trans == 0 && n < 50) {
|
||||
buffer = NULL;
|
||||
(trsv[(trans<<2) | (uplo<<1) | unit])(n, a, lda, x, incx, buffer);
|
||||
return;
|
||||
}
|
||||
|
||||
IDEBUG_START;
|
||||
|
||||
FUNCTION_PROFILE_START();
|
||||
|
||||
+7
-1
@@ -42,14 +42,20 @@
|
||||
#include "functable.h"
|
||||
#endif
|
||||
|
||||
#ifndef CBLAS
|
||||
void NAME(blasint *N, FLOAT *x, blasint *INCX, FLOAT *y, blasint *INCY, FLOAT *C, FLOAT *S){
|
||||
|
||||
BLASLONG n = *N;
|
||||
BLASLONG incx = *INCX;
|
||||
BLASLONG incy = *INCY;
|
||||
FLOAT c = *C;
|
||||
FLOAT s = *S;
|
||||
|
||||
#else
|
||||
void CNAME(blasint n, void *VX, blasint incx, void *VY, blasint incy, FLOAT c, FLOAT s) {
|
||||
FLOAT *x = (FLOAT*) VX;
|
||||
FLOAT *y = (FLOAT*) VY;
|
||||
#endif /* CBLAS */
|
||||
|
||||
PRINT_DEBUG_NAME;
|
||||
|
||||
if (n <= 0) return;
|
||||
|
||||
@@ -4,8 +4,16 @@
|
||||
#include "functable.h"
|
||||
#endif
|
||||
|
||||
#ifndef CBLAS
|
||||
void NAME(FLOAT *DA, FLOAT *DB, FLOAT *C, FLOAT *S){
|
||||
|
||||
#else
|
||||
void CNAME(void *VDA, void *VDB, FLOAT *C, void *VS) {
|
||||
FLOAT *DA = (FLOAT*) VDA;
|
||||
FLOAT *DB = (FLOAT*) VDB;
|
||||
FLOAT *S = (FLOAT*) VS;
|
||||
#endif /* CBLAS */
|
||||
|
||||
#if defined(__i386__) || defined(__x86_64__) || defined(__ia64__) || defined(_M_X64) || defined(_M_IX86)
|
||||
|
||||
long double da_r = *(DA + 0);
|
||||
|
||||
+1
-2
@@ -119,7 +119,7 @@ void NAME(char *UPLO, blasint *N, FLOAT *ALPHA,
|
||||
void CNAME(enum CBLAS_ORDER order, enum CBLAS_UPLO Uplo, int n, FLOAT alpha, FLOAT *x, int incx, FLOAT *a, int lda) {
|
||||
|
||||
FLOAT *buffer;
|
||||
int trans, uplo;
|
||||
int uplo;
|
||||
blasint info;
|
||||
FLOAT * ALPHA = α
|
||||
FLOAT alpha_r = ALPHA[0];
|
||||
@@ -130,7 +130,6 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_UPLO Uplo, int n, FLOAT alpha, FLO
|
||||
|
||||
PRINT_DEBUG_CNAME;
|
||||
|
||||
trans = -1;
|
||||
uplo = -1;
|
||||
info = 0;
|
||||
|
||||
|
||||
@@ -199,12 +199,6 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_UPLO Uplo,
|
||||
|
||||
if (n == 0) return;
|
||||
|
||||
if (incx == 1 && trans == 0 && n < 50) {
|
||||
buffer = NULL;
|
||||
(trsv[(trans<<2) | (uplo<<1) | unit])(n, a, lda, x, incx, buffer);
|
||||
return;
|
||||
}
|
||||
|
||||
IDEBUG_START;
|
||||
|
||||
FUNCTION_PROFILE_START();
|
||||
|
||||
+221
-38
@@ -9,11 +9,11 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
if (${DYNAMIC_ARCH})
|
||||
include("${PROJECT_SOURCE_DIR}/cmake/system.cmake")
|
||||
endif ()
|
||||
ParseMakefileVars("${KERNELDIR}/KERNEL")
|
||||
ParseMakefileVars("${KERNELDIR}/KERNEL.${TARGET_CORE}")
|
||||
SetDefaultL1()
|
||||
SetDefaultL2()
|
||||
SetDefaultL3()
|
||||
ParseMakefileVars("${KERNELDIR}/KERNEL")
|
||||
ParseMakefileVars("${KERNELDIR}/KERNEL.${TARGET_CORE}")
|
||||
|
||||
set(KERNEL_INTERFACE common_level1.h common_level2.h common_level3.h)
|
||||
if(NOT NO_LAPACK)
|
||||
@@ -91,6 +91,15 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
GenerateNamedObjects("${KERNELDIR}/${DSDOTKERNEL}" "DSDOT" "d*dot_k" false "" "" false "SINGLE")
|
||||
GenerateNamedObjects("${KERNELDIR}/${DSDOTKERNEL}" "DSDOT" "dsdot_k" false "" "" false "SINGLE")
|
||||
|
||||
# sbdot
|
||||
if (BUILD_BFLOAT16)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBDOTKERNEL}" "SBDOT" "dot_k" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${BF16TOKERNEL}" "SINGLE" "f16tos_k" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${BF16TOKERNEL}" "DOUBLE" "bf16tod_k" false "" "" false "DOUBLE")
|
||||
GenerateNamedObjects("${KERNELDIR}/${TOBF16KERNEL}" "SINGLE" "stobf16_k" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${TOBF16KERNEL}" "DOUBLE" "dtobf16_k" false "" "" false "BFLOAT16")
|
||||
endif()
|
||||
|
||||
if ((BUILD_COMPLEX OR BUILD_DOUBLE) AND NOT BUILD_SINGLE)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SAMAXKERNEL}" "USE_ABS" "amax_k" false "" "" false "SINGLE")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SAMINKERNEL}" "USE_ABS;USE_MIN" "amin_k" false "" "" false "SINGLE")
|
||||
@@ -149,9 +158,6 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
GenerateNamedObjects("generic/ger.c" "" "ger_k" false "" "" "" 3)
|
||||
foreach (float_type ${FLOAT_TYPES})
|
||||
string(SUBSTRING ${float_type} 0 1 float_char)
|
||||
if (${float_type} STREQUAL "BFLOAT16")
|
||||
set (float_char "SB")
|
||||
endif ()
|
||||
if (${float_type} STREQUAL "COMPLEX" OR ${float_type} STREQUAL "ZCOMPLEX")
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GERUKERNEL}" "" "geru_k" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GERCKERNEL}" "CONJ" "gerc_k" false "" "" false ${float_type})
|
||||
@@ -185,10 +191,14 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMVNKERNEL}" "" "gemv_n" false "" "" false "SINGLE")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMVTKERNEL}" "TRANS" "gemv_t" false "" "" false "SINGLE")
|
||||
endif ()
|
||||
if (BUILD_BFLOAT16)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMVNKERNEL}" "" "gemv_n" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMVTKERNEL}" "" "gemv_t" false "" "" false "BFLOAT16")
|
||||
endif ()
|
||||
# Makefile.L3
|
||||
set(USE_TRMM false)
|
||||
string(TOUPPER ${TARGET_CORE} UC_TARGET_CORE)
|
||||
if (ARM OR ARM64 OR (UC_TARGET_CORE MATCHES LONGSOON3B) OR (UC_TARGET_CORE MATCHES GENERIC) OR (UC_TARGET_CORE MATCHES HASWELL) OR (UC_TARGET_CORE MATCHES ZEN) OR (UC_TARGET_CORE MATCHES SKYLAKEX) OR (UC_TARGET_CORE MATCHES COOPERLAKE))
|
||||
if (ARM OR ARM64 OR (UC_TARGET_CORE MATCHES LONGSOON3B) OR (UC_TARGET_CORE MATCHES GENERIC) OR (UC_TARGET_CORE MATCHES HASWELL) OR (UC_TARGET_CORE MATCHES ZEN) OR (UC_TARGET_CORE MATCHES SKYLAKEX) OR (UC_TARGET_CORE MATCHES COOPERLAKE) OR (UC_TARGET_CORE MATCHES SAPPHIRERAPIDS))
|
||||
set(USE_TRMM true)
|
||||
endif ()
|
||||
if (ZARCH OR (UC_TARGET_CORE MATCHES POWER8) OR (UC_TARGET_CORE MATCHES POWER9) OR (UC_TARGET_CORE MATCHES POWER10))
|
||||
@@ -209,15 +219,8 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTPERFORMANT}" "" "gemm_direct_performant" false "" "" false SINGLE)
|
||||
endif()
|
||||
|
||||
foreach (float_type SINGLE DOUBLE BFLOAT16)
|
||||
foreach (float_type SINGLE DOUBLE)
|
||||
string(SUBSTRING ${float_type} 0 1 float_char)
|
||||
if (${float_type} STREQUAL "BFLOAT16")
|
||||
if (NOT ${BUILD_BFLOAT16})
|
||||
continue ()
|
||||
else ()
|
||||
set (float_char "SB")
|
||||
endif ()
|
||||
endif ()
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMMKERNEL}" "" "gemm_kernel" false "" "" false ${float_type})
|
||||
endforeach()
|
||||
if (BUILD_COMPLEX16 AND NOT BUILD_DOUBLE)
|
||||
@@ -253,11 +256,24 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMM_BETA}" "" "gemm_beta" false "" "" false "SINGLE")
|
||||
endif ()
|
||||
|
||||
if (BUILD_BFLOAT16)
|
||||
if (SBGEMMINCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMMINCOPY}" "" "${SBGEMMINCOPYOBJ}" false "" "" true "BFLOAT16")
|
||||
endif ()
|
||||
if (SBGEMMITCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMMITCOPY}" "" "${SBGEMMITCOPYOBJ}" false "" "" true "BFLOAT16")
|
||||
endif ()
|
||||
if (SBGEMMONCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMMONCOPY}" "" "${SBGEMMONCOPYOBJ}" false "" "" true "BFLOAT16")
|
||||
endif ()
|
||||
if (SBGEMMOTCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMMOTCOPY}" "" "${SBGEMMOTCOPYOBJ}" false "" "" true "BFLOAT16")
|
||||
endif ()
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMMKERNEL}" "" "gemm_kernel" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMM_BETA}" "" "gemm_beta" false "" "" false "BFLOAT16")
|
||||
endif ()
|
||||
foreach (float_type ${FLOAT_TYPES})
|
||||
string(SUBSTRING ${float_type} 0 1 float_char)
|
||||
if (${float_type} STREQUAL "BFLOAT16")
|
||||
set (float_char "SB")
|
||||
endif ()
|
||||
if (${float_char}GEMMINCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMMINCOPY}" "${float_type}" "${${float_char}GEMMINCOPYOBJ}" false "" "" true ${float_type})
|
||||
endif ()
|
||||
@@ -402,32 +418,50 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
GenerateCombinationObjects("${KERNELDIR}/${TRMM_KERNEL}" "LEFT;TRANSA" "R;N" "TRMMKERNEL" 2 "trmm_kernel" false ${float_type})
|
||||
|
||||
# symm for s and d
|
||||
if (NOT DEFINED ${float_char}SYMMUCOPY_M)
|
||||
set(SYMMUCOPY_M "generic/symm_ucopy_${${float_char}GEMM_UNROLL_M}.c")
|
||||
set(SYMMLCOPY_M "generic/symm_lcopy_${${float_char}GEMM_UNROLL_M}.c")
|
||||
else ()
|
||||
set(SYMMUCOPY_M "${KERNELDIR}/${${float_char}SYMMUCOPY_M}")
|
||||
set(SYMMLCOPY_M "${KERNELDIR}/${${float_char}SYMMLCOPY_M}")
|
||||
endif()
|
||||
GenerateNamedObjects("generic/symm_ucopy_${${float_char}GEMM_UNROLL_N}.c" "OUTER" "symm_outcopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/symm_ucopy_${${float_char}GEMM_UNROLL_M}.c" "" "symm_iutcopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects(${SYMMUCOPY_M} "" "symm_iutcopy" false "" "" false ${float_type})
|
||||
|
||||
GenerateNamedObjects("generic/symm_lcopy_${${float_char}GEMM_UNROLL_N}.c" "LOWER;OUTER" "symm_oltcopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/symm_lcopy_${${float_char}GEMM_UNROLL_M}.c" "LOWER" "symm_iltcopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects(${SYMMLCOPY_M} "LOWER" "symm_iltcopy" false "" "" false ${float_type})
|
||||
|
||||
# These don't use a scheme that is easy to iterate over - the filenames have part of the DEFINE codes in them, for UPPER/TRANS but not for UNIT/OUTER. Also TRANS is not passed in as a define.
|
||||
# Could simplify it a bit by pairing up by -UUNIT/-DUNIT.
|
||||
|
||||
GenerateNamedObjects("generic/trmm_uncopy_${${float_char}GEMM_UNROLL_M}.c" "UNIT" "trmm_iunucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_uncopy_${${float_char}GEMM_UNROLL_M}.c" "" "trmm_iunncopy" false "" "" false ${float_type})
|
||||
if (NOT DEFINED ${float_char}TRMMUNCOPY_M)
|
||||
set(TRMMUNCOPY_M "generic/trmm_uncopy_${${float_char}GEMM_UNROLL_M}.c")
|
||||
set(TRMMLNCOPY_M "generic/trmm_lncopy_${${float_char}GEMM_UNROLL_M}.c")
|
||||
set(TRMMUTCOPY_M "generic/trmm_utcopy_${${float_char}GEMM_UNROLL_M}.c")
|
||||
set(TRMMLTCOPY_M "generic/trmm_ltcopy_${${float_char}GEMM_UNROLL_M}.c")
|
||||
else ()
|
||||
set(TRMMUNCOPY_M "${KERNELDIR}/${${float_char}TRMMUNCOPY_M}")
|
||||
set(TRMMLNCOPY_M "${KERNELDIR}/${${float_char}TRMMLNCOPY_M}")
|
||||
set(TRMMUTCOPY_M "${KERNELDIR}/${${float_char}TRMMUTCOPY_M}")
|
||||
set(TRMMLTCOPY_M "${KERNELDIR}/${${float_char}TRMMLTCOPY_M}")
|
||||
endif ()
|
||||
GenerateNamedObjects(${TRMMUNCOPY_M} "UNIT" "trmm_iunucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects(${TRMMUNCOPY_M} "" "trmm_iunncopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_uncopy_${${float_char}GEMM_UNROLL_N}.c" "OUTER;UNIT" "trmm_ounucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_uncopy_${${float_char}GEMM_UNROLL_N}.c" "OUTER" "trmm_ounncopy" false "" "" false ${float_type})
|
||||
|
||||
GenerateNamedObjects("generic/trmm_lncopy_${${float_char}GEMM_UNROLL_M}.c" "LOWER;UNIT" "trmm_ilnucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_lncopy_${${float_char}GEMM_UNROLL_M}.c" "LOWER" "trmm_ilnncopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects(${TRMMLNCOPY_M} "LOWER;UNIT" "trmm_ilnucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects(${TRMMLNCOPY_M} "LOWER" "trmm_ilnncopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_lncopy_${${float_char}GEMM_UNROLL_N}.c" "OUTER;LOWER;UNIT" "trmm_olnucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_lncopy_${${float_char}GEMM_UNROLL_N}.c" "OUTER;LOWER" "trmm_olnncopy" false "" "" false ${float_type})
|
||||
|
||||
GenerateNamedObjects("generic/trmm_utcopy_${${float_char}GEMM_UNROLL_M}.c" "UNIT" "trmm_iutucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_utcopy_${${float_char}GEMM_UNROLL_M}.c" "" "trmm_iutncopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects(${TRMMUTCOPY_M} "UNIT" "trmm_iutucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects(${TRMMUTCOPY_M} "" "trmm_iutncopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_utcopy_${${float_char}GEMM_UNROLL_N}.c" "OUTER;UNIT" "trmm_outucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_utcopy_${${float_char}GEMM_UNROLL_N}.c" "OUTER" "trmm_outncopy" false "" "" false ${float_type})
|
||||
|
||||
GenerateNamedObjects("generic/trmm_ltcopy_${${float_char}GEMM_UNROLL_M}.c" "LOWER;UNIT" "trmm_iltucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_ltcopy_${${float_char}GEMM_UNROLL_M}.c" "LOWER" "trmm_iltncopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects(${TRMMLTCOPY_M} "LOWER;UNIT" "trmm_iltucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects(${TRMMLTCOPY_M} "LOWER" "trmm_iltncopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_ltcopy_${${float_char}GEMM_UNROLL_N}.c" "OUTER;LOWER;UNIT" "trmm_oltucopy" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("generic/trmm_ltcopy_${${float_char}GEMM_UNROLL_N}.c" "OUTER;LOWER" "trmm_oltncopy" false "" "" false ${float_type})
|
||||
|
||||
@@ -458,7 +492,155 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}TRSMKERNEL_RN}" "UPPER;RN;TRSMKERNEL" "trsm_kernel_RN" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}TRSMKERNEL_RT}" "RT;TRSMKERNEL" "trsm_kernel_RT" false "" "" false ${float_type})
|
||||
|
||||
if (NOT DEFINED ${float_char}GEMM_SMALL_M_PERMIT)
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
set(${float_char}GEMM_SMALL_M_PERMIT ../generic/zgemm_small_matrix_permit.c)
|
||||
else ()
|
||||
set(${float_char}GEMM_SMALL_M_PERMIT ../generic/gemm_small_matrix_permit.c)
|
||||
endif ()
|
||||
endif ()
|
||||
if (NOT DEFINED ${float_char}GEMM_SMALL_K_NN)
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
set(${float_char}GEMM_SMALL_K_NN ../generic/zgemm_small_matrix_kernel_nn.c)
|
||||
else ()
|
||||
set(${float_char}GEMM_SMALL_K_NN ../generic/gemm_small_matrix_kernel_nn.c)
|
||||
endif ()
|
||||
endif ()
|
||||
if (NOT DEFINED ${float_char}GEMM_SMALL_K_NT)
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
set(${float_char}GEMM_SMALL_K_NT ../generic/zgemm_small_matrix_kernel_nt.c)
|
||||
else ()
|
||||
set(${float_char}GEMM_SMALL_K_NT ../generic/gemm_small_matrix_kernel_nt.c)
|
||||
endif ()
|
||||
endif ()
|
||||
if (NOT DEFINED ${float_char}GEMM_SMALL_K_TN)
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
set(${float_char}GEMM_SMALL_K_TN ../generic/zgemm_small_matrix_kernel_tn.c)
|
||||
else ()
|
||||
set(${float_char}GEMM_SMALL_K_TN ../generic/gemm_small_matrix_kernel_tn.c)
|
||||
endif ()
|
||||
endif ()
|
||||
if (NOT DEFINED ${float_char}GEMM_SMALL_K_TT)
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
set(${float_char}GEMM_SMALL_K_TT ../generic/zgemm_small_matrix_kernel_tt.c)
|
||||
else ()
|
||||
set(${float_char}GEMM_SMALL_K_TT ../generic/gemm_small_matrix_kernel_tt.c)
|
||||
endif ()
|
||||
endif ()
|
||||
if (NOT DEFINED ${float_char}GEMM_SMALL_K_B0_NN)
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
set(${float_char}GEMM_SMALL_K_B0_NN ../generic/zgemm_small_matrix_kernel_nn.c)
|
||||
else ()
|
||||
set(${float_char}GEMM_SMALL_K_B0_NN ../generic/gemm_small_matrix_kernel_nn.c)
|
||||
endif ()
|
||||
endif ()
|
||||
if (NOT DEFINED ${float_char}GEMM_SMALL_K_B0_NT)
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
set(${float_char}GEMM_SMALL_K_B0_NT ../generic/zgemm_small_matrix_kernel_nt.c)
|
||||
else ()
|
||||
set(${float_char}GEMM_SMALL_K_B0_NT ../generic/gemm_small_matrix_kernel_nt.c)
|
||||
endif ()
|
||||
endif ()
|
||||
if (NOT DEFINED ${float_char}GEMM_SMALL_K_B0_TN)
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
set(${float_char}GEMM_SMALL_K_B0_TN ../generic/zgemm_small_matrix_kernel_tn.c)
|
||||
else ()
|
||||
set(${float_char}GEMM_SMALL_K_B0_TN ../generic/gemm_small_matrix_kernel_tn.c)
|
||||
endif ()
|
||||
endif ()
|
||||
if (NOT DEFINED ${float_char}GEMM_SMALL_K_B0_TT)
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
set(${float_char}GEMM_SMALL_K_B0_TT ../generic/zgemm_small_matrix_kernel_tt.c)
|
||||
else ()
|
||||
set(${float_char}GEMM_SMALL_K_B0_TT ../generic/gemm_small_matrix_kernel_tt.c)
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (SMALL_MATRIX_OPT)
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_M_PERMIT}" "" "gemm_small_matrix_permit" false "" "" false ${float_type})
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_NN}" "NN" "gemm_small_kernel_nn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_NN}" "NR" "gemm_small_kernel_nr" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_NN}" "RN" "gemm_small_kernel_rn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_NN}" "RR" "gemm_small_kernel_rr" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_NT}" "NT" "gemm_small_kernel_nt" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_NT}" "NC" "gemm_small_kernel_nc" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_NT}" "RT" "gemm_small_kernel_rt" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_NT}" "RC" "gemm_small_kernel_rc" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_TN}" "TN" "gemm_small_kernel_tn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_TN}" "TR" "gemm_small_kernel_tr" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_TN}" "CN" "gemm_small_kernel_cn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_TN}" "CR" "gemm_small_kernel_cr" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_TT}" "TT" "gemm_small_kernel_tt" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_TT}" "TC" "gemm_small_kernel_tc" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_TT}" "CT" "gemm_small_kernel_ct" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_TT}" "CC" "gemm_small_kernel_cc" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_NN}" "NN;B0" "gemm_small_kernel_b0_nn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_NN}" "NR;B0" "gemm_small_kernel_b0_nr" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_NN}" "RN;B0" "gemm_small_kernel_b0_rn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_NN}" "RR;B0" "gemm_small_kernel_b0_rr" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_NT}" "NT;B0" "gemm_small_kernel_b0_nt" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_NT}" "NC;B0" "gemm_small_kernel_b0_nc" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_NT}" "RT;B0" "gemm_small_kernel_b0_rt" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_NT}" "RC;B0" "gemm_small_kernel_b0_rc" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_TN}" "TN;B0" "gemm_small_kernel_b0_tn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_TN}" "TR;B0" "gemm_small_kernel_b0_tr" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_TN}" "CN;B0" "gemm_small_kernel_b0_cn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_TN}" "CR;B0" "gemm_small_kernel_b0_cr" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_TT}" "TT;B0" "gemm_small_kernel_b0_tt" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_TT}" "TC;B0" "gemm_small_kernel_b0_tc" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_TT}" "CT;B0" "gemm_small_kernel_b0_ct" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_TT}" "CC;B0" "gemm_small_kernel_b0_cc" false "" "" false ${float_type})
|
||||
|
||||
else ()
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_NN}" "" "gemm_small_kernel_nn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_NT}" "" "gemm_small_kernel_nt" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_TN}" "" "gemm_small_kernel_tn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_TT}" "" "gemm_small_kernel_tt" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_NN}" "B0" "gemm_small_kernel_b0_nn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_NT}" "B0" "gemm_small_kernel_b0_nt" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_TN}" "B0" "gemm_small_kernel_b0_tn" false "" "" false ${float_type})
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMM_SMALL_K_B0_TT}" "B0" "gemm_small_kernel_b0_tt" false "" "" false ${float_type})
|
||||
endif ()
|
||||
if (BUILD_BFLOAT16)
|
||||
if (NOT DEFINED SBGEMM_SMALL_M_PERMIT)
|
||||
set(SBGEMM_SMALL_M_PERMIT ../generic/gemm_small_matrix_permit.c)
|
||||
endif ()
|
||||
if (NOT DEFINED SBGEMM_SMALL_K_NN)
|
||||
set(SBGEMM_SMALL_K_NN ../generic/gemm_small_matrix_kernel_nn.c)
|
||||
endif ()
|
||||
if (NOT DEFINED SBGEMM_SMALL_K_NT)
|
||||
set(SBGEMM_SMALL_K_NT ../generic/gemm_small_matrix_kernel_nt.c)
|
||||
endif ()
|
||||
if (NOT DEFINED SBGEMM_SMALL_K_TN)
|
||||
set(SBGEMM_SMALL_K_TN ../generic/gemm_small_matrix_kernel_tn.c)
|
||||
endif ()
|
||||
if (NOT DEFINED SBGEMM_SMALL_K_TT)
|
||||
set(SBGEMM_SMALL_K_TT ../generic/gemm_small_matrix_kernel_tt.c)
|
||||
endif ()
|
||||
if (NOT DEFINED SBGEMM_SMALL_K_B0_NN)
|
||||
set(SBGEMM_SMALL_K_B0_NN ../generic/gemm_small_matrix_kernel_nn.c)
|
||||
endif ()
|
||||
if (NOT DEFINED SBGEMM_SMALL_K_B0_NT)
|
||||
set(SBGEMM_SMALL_K_B0_NT ../generic/gemm_small_matrix_kernel_nt.c)
|
||||
endif ()
|
||||
if (NOT DEFINED SBGEMM_SMALL_K_B0_TN)
|
||||
set(SBGEMM_SMALL_K_B0_TN ../generic/gemm_small_matrix_kernel_tn.c)
|
||||
endif ()
|
||||
if (NOT DEFINED SBGEMM_SMALL_K_B0_TT)
|
||||
set($SBGEMM_SMALL_K_B0_TT ../generic/gemm_small_matrix_kernel_tt.c)
|
||||
endif ()
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMM_SMALL_M_PERMIT}" "" "gemm_small_matrix_permit" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMM_SMALL_K_NN}" "" "gemm_small_kernel_nn" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMM_SMALL_K_NT}" "" "gemm_small_kernel_nt" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMM_SMALL_K_TN}" "" "gemm_small_kernel_tn" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMM_SMALL_K_TT}" "" "gemm_small_kernel_tt" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMM_SMALL_K_B0_NN}" "B0" "gemm_small_kernel_b0_nn" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMM_SMALL_K_B0_NT}" "B0" "gemm_small_kernel_b0_nt" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMM_SMALL_K_B0_TN}" "B0" "gemm_small_kernel_b0_tn" false "" "" false "BFLOAT16")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SBGEMM_SMALL_K_B0_TT}" "B0" "gemm_small_kernel_b0_tt" false "" "" false "BFLOAT16")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (NOT DEFINED ${float_char}OMATCOPY_CN)
|
||||
if (${float_char} STREQUAL "Z" OR ${float_char} STREQUAL "C")
|
||||
@@ -592,6 +774,7 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
#geadd
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEADD_KERNEL}" "" "geadd_k" false "" "" false ${float_type})
|
||||
endforeach ()
|
||||
|
||||
if (BUILD_DOUBLE AND NOT BUILD_SINGLE)
|
||||
GenerateNamedObjects("${KERNELDIR}/${STRSMKERNEL_LN}" "UPPER;LN;TRSMKERNEL" "trsm_kernel_LN" false "" "" false "SINGLE")
|
||||
GenerateNamedObjects("${KERNELDIR}/${STRSMKERNEL_LT}" "LT;TRSMKERNEL" "trsm_kernel_LT" false "" "" false "SINGLE")
|
||||
@@ -730,22 +913,22 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
GenerateNamedObjects("generic/trsm_ltcopy_${SGEMM_UNROLL_N}.c" "OUTER;LOWER" "trsm_oltncopy" false "" ${TSUFFIX} false "SINGLE")
|
||||
|
||||
if (SGEMMINCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMINCOPY}" "SINGLE" "${SGEMMINCOPYOBJ}" false "" "" true "SINGLE")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMINCOPY}" "SINGLE" "${SGEMMINCOPYOBJ}" false "" "" true "SINGLE")
|
||||
endif ()
|
||||
if (SGEMMITCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMITCOPY}" "SINGLE" "${SGEMMITCOPYOBJ}" false "" "" true "SINGLE")
|
||||
endif ()
|
||||
if (SGEMMONCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMONCOPY}" "SINGLE" "${SGEMMONCOPYOBJ}" false "" "" true "SINGLE")
|
||||
endif ()
|
||||
if (SGEMMOTCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMOTCOPY}" "SINGLE" "${SGEMMOTCOPYOBJ}" false "" "" true "SINGLE")
|
||||
if (SGEMMITCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMITCOPY}" "SINGLE" "${SGEMMITCOPYOBJ}" false "" "" true "SINGLE")
|
||||
endif ()
|
||||
if (SGEMMONCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMONCOPY}" "SINGLE" "${SGEMMONCOPYOBJ}" false "" "" true "SINGLE")
|
||||
endif ()
|
||||
if (SGEMMOTCOPY)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMOTCOPY}" "SINGLE" "${SGEMMOTCOPYOBJ}" false "" "" true "SINGLE")
|
||||
endif ()
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMVNKERNEL}" "" "gemv_n" false "" "" false "SINGLE")
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMVTKERNEL}" "TRANS" "gemv_t" false "" "" false "SINGLE")
|
||||
endif ()
|
||||
|
||||
if (BUILD_COMPLEX16 AND NOT BUILD_DOUBLE)
|
||||
|
||||
if (BUILD_COMPLEX16 AND NOT BUILD_DOUBLE)
|
||||
GenerateNamedObjects("generic/neg_tcopy_${DGEMM_UNROLL_M}.c" "" "neg_tcopy" false "" ${TSUFFIX} false "DOUBLE")
|
||||
GenerateNamedObjects("generic/laswp_ncopy_${DGEMM_UNROLL_N}.c" "" "laswp_ncopy" false "" ${TSUFFIX} false "DOUBLE")
|
||||
endif ()
|
||||
|
||||
+16
-1
@@ -31,7 +31,22 @@ ifdef NO_AVX2
|
||||
endif
|
||||
|
||||
ifdef TARGET_CORE
|
||||
ifeq ($(TARGET_CORE), COOPERLAKE)
|
||||
ifeq ($(TARGET_CORE), SAPPHIRERAPIDS)
|
||||
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE)
|
||||
ifeq ($(GCCVERSIONGTEQ10), 1)
|
||||
override CFLAGS += -march=sapphirerapids
|
||||
else
|
||||
override CFLAGS += -march=skylake-avx512 -mavx512f
|
||||
endif
|
||||
ifeq ($(OSNAME), CYGWIN_NT)
|
||||
override CFLAGS += -fno-asynchronous-unwind-tables
|
||||
endif
|
||||
ifeq ($(OSNAME), WINNT)
|
||||
ifeq ($(C_COMPILER), GCC)
|
||||
override CFLAGS += -fno-asynchronous-unwind-tables
|
||||
endif
|
||||
endif
|
||||
else ifeq ($(TARGET_CORE), COOPERLAKE)
|
||||
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE)
|
||||
ifeq ($(GCCVERSIONGTEQ10), 1)
|
||||
override CFLAGS += -march=cooperlake
|
||||
|
||||
@@ -47,6 +47,10 @@ ifeq ($(CORE), COOPERLAKE)
|
||||
USE_TRMM = 1
|
||||
endif
|
||||
|
||||
ifeq ($(CORE), SAPPHIRERAPIDS)
|
||||
USE_TRMM = 1
|
||||
endif
|
||||
|
||||
ifeq ($(CORE), ZEN)
|
||||
USE_TRMM = 1
|
||||
endif
|
||||
@@ -447,6 +451,72 @@ XBLASOBJS += \
|
||||
|
||||
endif
|
||||
|
||||
###### BLAS small matrix optimization #####
|
||||
ifeq ($(SMALL_MATRIX_OPT), 1)
|
||||
|
||||
ifeq ($(BUILD_BFLOAT16),1)
|
||||
SBBLASOBJS += \
|
||||
sbgemm_small_matrix_permit$(TSUFFIX).$(SUFFIX) \
|
||||
sbgemm_small_kernel_nn$(TSUFFIX).$(SUFFIX) sbgemm_small_kernel_nt$(TSUFFIX).$(SUFFIX) \
|
||||
sbgemm_small_kernel_tn$(TSUFFIX).$(SUFFIX) sbgemm_small_kernel_tt$(TSUFFIX).$(SUFFIX) \
|
||||
sbgemm_small_kernel_b0_nn$(TSUFFIX).$(SUFFIX) sbgemm_small_kernel_b0_nt$(TSUFFIX).$(SUFFIX) \
|
||||
sbgemm_small_kernel_b0_tn$(TSUFFIX).$(SUFFIX) sbgemm_small_kernel_b0_tt$(TSUFFIX).$(SUFFIX)
|
||||
endif
|
||||
|
||||
SBLASOBJS += \
|
||||
sgemm_small_matrix_permit$(TSUFFIX).$(SUFFIX) \
|
||||
sgemm_small_kernel_nn$(TSUFFIX).$(SUFFIX) sgemm_small_kernel_nt$(TSUFFIX).$(SUFFIX) \
|
||||
sgemm_small_kernel_tn$(TSUFFIX).$(SUFFIX) sgemm_small_kernel_tt$(TSUFFIX).$(SUFFIX) \
|
||||
sgemm_small_kernel_b0_nn$(TSUFFIX).$(SUFFIX) sgemm_small_kernel_b0_nt$(TSUFFIX).$(SUFFIX) \
|
||||
sgemm_small_kernel_b0_tn$(TSUFFIX).$(SUFFIX) sgemm_small_kernel_b0_tt$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
DBLASOBJS += \
|
||||
dgemm_small_matrix_permit$(TSUFFIX).$(SUFFIX) \
|
||||
dgemm_small_kernel_nn$(TSUFFIX).$(SUFFIX) dgemm_small_kernel_nt$(TSUFFIX).$(SUFFIX) \
|
||||
dgemm_small_kernel_tn$(TSUFFIX).$(SUFFIX) dgemm_small_kernel_tt$(TSUFFIX).$(SUFFIX) \
|
||||
dgemm_small_kernel_b0_nn$(TSUFFIX).$(SUFFIX) dgemm_small_kernel_b0_nt$(TSUFFIX).$(SUFFIX) \
|
||||
dgemm_small_kernel_b0_tn$(TSUFFIX).$(SUFFIX) dgemm_small_kernel_b0_tt$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
CBLASOBJS += \
|
||||
cgemm_small_matrix_permit$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_nn$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_nt$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_nr$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_nc$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_tn$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_tt$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_tr$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_tc$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_rn$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_rt$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_rr$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_rc$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_cn$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_ct$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_cr$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_cc$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_b0_nn$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_b0_nt$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_b0_nr$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_b0_nc$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_b0_tn$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_b0_tt$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_b0_tr$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_b0_tc$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_b0_rn$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_b0_rt$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_b0_rr$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_b0_rc$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_b0_cn$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_b0_ct$(TSUFFIX).$(SUFFIX) \
|
||||
cgemm_small_kernel_b0_cr$(TSUFFIX).$(SUFFIX) cgemm_small_kernel_b0_cc$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
ZBLASOBJS += \
|
||||
zgemm_small_matrix_permit$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_nn$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_nt$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_nr$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_nc$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_tn$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_tt$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_tr$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_tc$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_rn$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_rt$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_rr$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_rc$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_cn$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_ct$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_cr$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_cc$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_b0_nn$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_b0_nt$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_b0_nr$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_b0_nc$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_b0_tn$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_b0_tt$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_b0_tr$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_b0_tc$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_b0_rn$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_b0_rt$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_b0_rr$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_b0_rc$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_b0_cn$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_b0_ct$(TSUFFIX).$(SUFFIX) \
|
||||
zgemm_small_kernel_b0_cr$(TSUFFIX).$(SUFFIX) zgemm_small_kernel_b0_cc$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
endif
|
||||
|
||||
###### BLAS extensions #####
|
||||
|
||||
ifeq ($(BUILD_SINGLE),1)
|
||||
@@ -1413,29 +1483,61 @@ $(KDIR)xtrsm_kernel_RC$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(XTRSMKERNEL_RT) $(XT
|
||||
$(CC) -c $(CFLAGS) -DTRSMKERNEL -DCOMPLEX -DXDOUBLE -UUPPER -DRT -DCONJ $< -o $@
|
||||
|
||||
|
||||
ifdef STRMMUNCOPY_M
|
||||
$(KDIR)strmm_iunucopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(STRMMUNCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -ULOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)strmm_iunncopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(STRMMUNCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -ULOWER -UUNIT $< -o $@
|
||||
else
|
||||
$(KDIR)strmm_iunucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_uncopy_$(SGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -ULOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)strmm_iunncopy$(TSUFFIX).$(SUFFIX) : generic/trmm_uncopy_$(SGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -ULOWER -UUNIT $< -o $@
|
||||
endif
|
||||
|
||||
ifdef STRMMLNCOPY_M
|
||||
$(KDIR)strmm_ilnucopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(STRMMLNCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -DLOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)strmm_ilnncopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(STRMMLNCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -DLOWER -UUNIT $< -o $@
|
||||
else
|
||||
$(KDIR)strmm_ilnucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_lncopy_$(SGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -DLOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)strmm_ilnncopy$(TSUFFIX).$(SUFFIX) : generic/trmm_lncopy_$(SGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -DLOWER -UUNIT $< -o $@
|
||||
endif
|
||||
|
||||
ifdef STRMMUTCOPY_M
|
||||
$(KDIR)strmm_iutucopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(STRMMUTCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -ULOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)strmm_iutncopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(STRMMUTCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -ULOWER -UUNIT $< -o $@
|
||||
else
|
||||
$(KDIR)strmm_iutucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_utcopy_$(SGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -ULOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)strmm_iutncopy$(TSUFFIX).$(SUFFIX) : generic/trmm_utcopy_$(SGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -ULOWER -UUNIT $< -o $@
|
||||
endif
|
||||
|
||||
ifdef STRMMLTCOPY_M
|
||||
$(KDIR)strmm_iltucopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(STRMMLTCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -DLOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)strmm_iltncopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(STRMMLTCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -DLOWER -UUNIT $< -o $@
|
||||
else
|
||||
$(KDIR)strmm_iltucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_ltcopy_$(SGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -DLOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)strmm_iltncopy$(TSUFFIX).$(SUFFIX) : generic/trmm_ltcopy_$(SGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -DLOWER -UUNIT $< -o $@
|
||||
endif
|
||||
|
||||
$(KDIR)strmm_ounucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_uncopy_$(SGEMM_UNROLL_N).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -DOUTER -ULOWER -DUNIT $< -o $@
|
||||
@@ -1461,29 +1563,61 @@ $(KDIR)strmm_oltucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_ltcopy_$(SGEMM_UNROLL_N
|
||||
$(KDIR)strmm_oltncopy$(TSUFFIX).$(SUFFIX) : generic/trmm_ltcopy_$(SGEMM_UNROLL_N).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -DOUTER -DLOWER -UUNIT $< -o $@
|
||||
|
||||
ifdef DTRMMUNCOPY_M
|
||||
$(KDIR)dtrmm_iunucopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DTRMMUNCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -ULOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)dtrmm_iunncopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DTRMMUNCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -ULOWER -UUNIT $< -o $@
|
||||
else
|
||||
$(KDIR)dtrmm_iunucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_uncopy_$(DGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -ULOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)dtrmm_iunncopy$(TSUFFIX).$(SUFFIX) : generic/trmm_uncopy_$(DGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -ULOWER -UUNIT $< -o $@
|
||||
endif
|
||||
|
||||
ifdef DTRMMLNCOPY_M
|
||||
$(KDIR)dtrmm_ilnucopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DTRMMLNCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -DLOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)dtrmm_ilnncopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DTRMMLNCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -DLOWER -UUNIT $< -o $@
|
||||
else
|
||||
$(KDIR)dtrmm_ilnucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_lncopy_$(DGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -DLOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)dtrmm_ilnncopy$(TSUFFIX).$(SUFFIX) : generic/trmm_lncopy_$(DGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -DLOWER -UUNIT $< -o $@
|
||||
endif
|
||||
|
||||
ifdef DTRMMUTCOPY_M
|
||||
$(KDIR)dtrmm_iutucopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DTRMMUTCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -ULOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)dtrmm_iutncopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DTRMMUTCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -ULOWER -UUNIT $< -o $@
|
||||
else
|
||||
$(KDIR)dtrmm_iutucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_utcopy_$(DGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -ULOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)dtrmm_iutncopy$(TSUFFIX).$(SUFFIX) : generic/trmm_utcopy_$(DGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -ULOWER -UUNIT $< -o $@
|
||||
endif
|
||||
|
||||
ifdef DTRMMLTCOPY_M
|
||||
$(KDIR)dtrmm_iltucopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DTRMMLTCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -DLOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)dtrmm_iltncopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DTRMMLTCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -DLOWER -UUNIT $< -o $@
|
||||
else
|
||||
$(KDIR)dtrmm_iltucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_ltcopy_$(DGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -DLOWER -DUNIT $< -o $@
|
||||
|
||||
$(KDIR)dtrmm_iltncopy$(TSUFFIX).$(SUFFIX) : generic/trmm_ltcopy_$(DGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -DLOWER -UUNIT $< -o $@
|
||||
endif
|
||||
|
||||
$(KDIR)dtrmm_ounucopy$(TSUFFIX).$(SUFFIX) : generic/trmm_uncopy_$(DGEMM_UNROLL_N).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -DOUTER -ULOWER -DUNIT $< -o $@
|
||||
@@ -1707,11 +1841,21 @@ $(KDIR)ssymm_outcopy$(TSUFFIX).$(SUFFIX) : generic/symm_ucopy_$(SGEMM_UNROLL_N).
|
||||
$(KDIR)ssymm_oltcopy$(TSUFFIX).$(SUFFIX) : generic/symm_lcopy_$(SGEMM_UNROLL_N).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -DOUTER -DLOWER $< -o $@
|
||||
|
||||
ifdef SSYMMUCOPY_M
|
||||
$(KDIR)ssymm_iutcopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYMMUCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -ULOWER $< -o $@
|
||||
else
|
||||
$(KDIR)ssymm_iutcopy$(TSUFFIX).$(SUFFIX) : generic/symm_ucopy_$(SGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -ULOWER $< -o $@
|
||||
endif
|
||||
|
||||
ifdef SSYMMLCOPY_M
|
||||
$(KDIR)ssymm_iltcopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYMMLCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -DLOWER $< -o $@
|
||||
else
|
||||
$(KDIR)ssymm_iltcopy$(TSUFFIX).$(SUFFIX) : generic/symm_lcopy_$(SGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -UDOUBLE -UCOMPLEX -UOUTER -DLOWER $< -o $@
|
||||
endif
|
||||
|
||||
$(KDIR)dsymm_outcopy$(TSUFFIX).$(SUFFIX) : generic/symm_ucopy_$(DGEMM_UNROLL_N).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -DOUTER -ULOWER $< -o $@
|
||||
@@ -1719,11 +1863,21 @@ $(KDIR)dsymm_outcopy$(TSUFFIX).$(SUFFIX) : generic/symm_ucopy_$(DGEMM_UNROLL_N).
|
||||
$(KDIR)dsymm_oltcopy$(TSUFFIX).$(SUFFIX) : generic/symm_lcopy_$(DGEMM_UNROLL_N).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -DOUTER -DLOWER $< -o $@
|
||||
|
||||
ifdef DSYMMUCOPY_M
|
||||
$(KDIR)dsymm_iutcopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DSYMMUCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -ULOWER $< -o $@
|
||||
else
|
||||
$(KDIR)dsymm_iutcopy$(TSUFFIX).$(SUFFIX) : generic/symm_ucopy_$(DGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -ULOWER $< -o $@
|
||||
endif
|
||||
|
||||
ifdef DSYMMLCOPY_M
|
||||
$(KDIR)dsymm_iltcopy$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DSYMMLCOPY_M)
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -DLOWER $< -o $@
|
||||
else
|
||||
$(KDIR)dsymm_iltcopy$(TSUFFIX).$(SUFFIX) : generic/symm_lcopy_$(DGEMM_UNROLL_M).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DDOUBLE -UCOMPLEX -UOUTER -DLOWER $< -o $@
|
||||
endif
|
||||
|
||||
$(KDIR)qsymm_outcopy$(TSUFFIX).$(SUFFIX) : generic/symm_ucopy_$(QGEMM_UNROLL_N).c
|
||||
$(CC) -c $(CFLAGS) $(NO_UNINITIALIZED_WARN) -DXDOUBLE -UCOMPLEX -DOUTER -ULOWER $< -o $@
|
||||
@@ -4237,3 +4391,469 @@ endif
|
||||
$(KDIR)zgeadd_k$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEADD_K)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -UROWM $< -o $@
|
||||
|
||||
|
||||
|
||||
###### BLAS small matrix optimization #####
|
||||
|
||||
ifndef DGEMM_SMALL_M_PERMIT
|
||||
DGEMM_SMALL_M_PERMIT = ../generic/gemm_small_matrix_permit.c
|
||||
endif
|
||||
|
||||
ifndef DGEMM_SMALL_K_NN
|
||||
DGEMM_SMALL_K_NN = ../generic/gemm_small_matrix_kernel_nn.c
|
||||
endif
|
||||
|
||||
ifndef DGEMM_SMALL_K_NT
|
||||
DGEMM_SMALL_K_NT = ../generic/gemm_small_matrix_kernel_nt.c
|
||||
endif
|
||||
|
||||
ifndef DGEMM_SMALL_K_TN
|
||||
DGEMM_SMALL_K_TN = ../generic/gemm_small_matrix_kernel_tn.c
|
||||
endif
|
||||
|
||||
ifndef DGEMM_SMALL_K_TT
|
||||
DGEMM_SMALL_K_TT = ../generic/gemm_small_matrix_kernel_tt.c
|
||||
endif
|
||||
|
||||
$(KDIR)dgemm_small_matrix_permit$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DGEMM_SMALL_M_PERMIT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)dgemm_small_kernel_nn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)dgemm_small_kernel_nt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)dgemm_small_kernel_tn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)dgemm_small_kernel_tt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
ifndef DGEMM_SMALL_K_B0_NN
|
||||
DGEMM_SMALL_K_B0_NN = ../generic/gemm_small_matrix_kernel_nn.c
|
||||
endif
|
||||
|
||||
ifndef DGEMM_SMALL_K_B0_NT
|
||||
DGEMM_SMALL_K_B0_NT = ../generic/gemm_small_matrix_kernel_nt.c
|
||||
endif
|
||||
|
||||
ifndef DGEMM_SMALL_K_B0_TN
|
||||
DGEMM_SMALL_K_B0_TN = ../generic/gemm_small_matrix_kernel_tn.c
|
||||
endif
|
||||
|
||||
ifndef DGEMM_SMALL_K_B0_TT
|
||||
DGEMM_SMALL_K_B0_TT = ../generic/gemm_small_matrix_kernel_tt.c
|
||||
endif
|
||||
|
||||
$(KDIR)dgemm_small_kernel_b0_nn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
$(KDIR)dgemm_small_kernel_b0_nt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
$(KDIR)dgemm_small_kernel_b0_tn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
$(KDIR)dgemm_small_kernel_b0_tt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(DGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
ifndef SGEMM_SMALL_M_PERMIT
|
||||
SGEMM_SMALL_M_PERMIT = ../generic/gemm_small_matrix_permit.c
|
||||
endif
|
||||
|
||||
ifndef SGEMM_SMALL_K_NN
|
||||
SGEMM_SMALL_K_NN = ../generic/gemm_small_matrix_kernel_nn.c
|
||||
endif
|
||||
|
||||
ifndef SGEMM_SMALL_K_NT
|
||||
SGEMM_SMALL_K_NT = ../generic/gemm_small_matrix_kernel_nt.c
|
||||
endif
|
||||
|
||||
ifndef SGEMM_SMALL_K_TN
|
||||
SGEMM_SMALL_K_TN = ../generic/gemm_small_matrix_kernel_tn.c
|
||||
endif
|
||||
|
||||
ifndef SGEMM_SMALL_K_TT
|
||||
SGEMM_SMALL_K_TT = ../generic/gemm_small_matrix_kernel_tt.c
|
||||
endif
|
||||
|
||||
$(KDIR)sgemm_small_matrix_permit$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMM_SMALL_M_PERMIT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)sgemm_small_kernel_nn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)sgemm_small_kernel_nt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)sgemm_small_kernel_tn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)sgemm_small_kernel_tt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
ifndef SGEMM_SMALL_K_B0_NN
|
||||
SGEMM_SMALL_K_B0_NN = ../generic/gemm_small_matrix_kernel_nn.c
|
||||
endif
|
||||
|
||||
ifndef SGEMM_SMALL_K_B0_NT
|
||||
SGEMM_SMALL_K_B0_NT = ../generic/gemm_small_matrix_kernel_nt.c
|
||||
endif
|
||||
|
||||
ifndef SGEMM_SMALL_K_B0_TN
|
||||
SGEMM_SMALL_K_B0_TN = ../generic/gemm_small_matrix_kernel_tn.c
|
||||
endif
|
||||
|
||||
ifndef SGEMM_SMALL_K_B0_TT
|
||||
SGEMM_SMALL_K_B0_TT = ../generic/gemm_small_matrix_kernel_tt.c
|
||||
endif
|
||||
|
||||
$(KDIR)sgemm_small_kernel_b0_nn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
$(KDIR)sgemm_small_kernel_b0_nt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
$(KDIR)sgemm_small_kernel_b0_tn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
$(KDIR)sgemm_small_kernel_b0_tt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
|
||||
ifeq ($(BUILD_BFLOAT16), 1)
|
||||
ifndef SBGEMM_SMALL_M_PERMIT
|
||||
SBGEMM_SMALL_M_PERMIT = ../generic/gemm_small_matrix_permit.c
|
||||
endif
|
||||
|
||||
ifndef SBGEMM_SMALL_K_NN
|
||||
SBGEMM_SMALL_K_NN = ../generic/gemm_small_matrix_kernel_nn.c
|
||||
endif
|
||||
|
||||
ifndef SBGEMM_SMALL_K_NT
|
||||
SBGEMM_SMALL_K_NT = ../generic/gemm_small_matrix_kernel_nt.c
|
||||
endif
|
||||
|
||||
ifndef SBGEMM_SMALL_K_TN
|
||||
SBGEMM_SMALL_K_TN = ../generic/gemm_small_matrix_kernel_tn.c
|
||||
endif
|
||||
|
||||
ifndef SBGEMM_SMALL_K_TT
|
||||
SBGEMM_SMALL_K_TT = ../generic/gemm_small_matrix_kernel_tt.c
|
||||
endif
|
||||
|
||||
$(KDIR)sbgemm_small_matrix_permit$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SBGEMM_SMALL_M_PERMIT)
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -UDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)sbgemm_small_kernel_nn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SBGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -UDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)sbgemm_small_kernel_nt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SBGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -UDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)sbgemm_small_kernel_tn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SBGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -UDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)sbgemm_small_kernel_tt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SBGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -UDOUBLE -UCOMPLEX $< -o $@
|
||||
|
||||
ifndef SBGEMM_SMALL_K_B0_NN
|
||||
SBGEMM_SMALL_K_B0_NN = ../generic/gemm_small_matrix_kernel_nn.c
|
||||
endif
|
||||
|
||||
ifndef SBGEMM_SMALL_K_B0_NT
|
||||
SBGEMM_SMALL_K_B0_NT = ../generic/gemm_small_matrix_kernel_nt.c
|
||||
endif
|
||||
|
||||
ifndef SBGEMM_SMALL_K_B0_TN
|
||||
SBGEMM_SMALL_K_B0_TN = ../generic/gemm_small_matrix_kernel_tn.c
|
||||
endif
|
||||
|
||||
ifndef SBGEMM_SMALL_K_B0_TT
|
||||
SBGEMM_SMALL_K_B0_TT = ../generic/gemm_small_matrix_kernel_tt.c
|
||||
endif
|
||||
|
||||
$(KDIR)sbgemm_small_kernel_b0_nn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SBGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -UDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
$(KDIR)sbgemm_small_kernel_b0_nt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SBGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -UDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
$(KDIR)sbgemm_small_kernel_b0_tn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SBGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -UDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
|
||||
$(KDIR)sbgemm_small_kernel_b0_tt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SBGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -UDOUBLE -UCOMPLEX -DB0 $< -o $@
|
||||
endif
|
||||
|
||||
ifndef CGEMM_SMALL_M_PERMIT
|
||||
CGEMM_SMALL_M_PERMIT = ../generic/zgemm_small_matrix_permit.c
|
||||
endif
|
||||
|
||||
ifndef CGEMM_SMALL_K_NN
|
||||
CGEMM_SMALL_K_NN = ../generic/zgemm_small_matrix_kernel_nn.c
|
||||
endif
|
||||
|
||||
ifndef CGEMM_SMALL_K_NT
|
||||
CGEMM_SMALL_K_NT = ../generic/zgemm_small_matrix_kernel_nt.c
|
||||
endif
|
||||
|
||||
ifndef CGEMM_SMALL_K_TN
|
||||
CGEMM_SMALL_K_TN = ../generic/zgemm_small_matrix_kernel_tn.c
|
||||
endif
|
||||
|
||||
ifndef CGEMM_SMALL_K_TT
|
||||
CGEMM_SMALL_K_TT = ../generic/zgemm_small_matrix_kernel_tt.c
|
||||
endif
|
||||
|
||||
$(KDIR)cgemm_small_matrix_permit$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_M_PERMIT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_nn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DNN $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_nr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DNR $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_rn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DRN $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_rr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DRR $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_nt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DNT $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_nc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DNC $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_rt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DRT $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_rc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DRC=RC $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_tn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DTN $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_tr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DTR $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_cn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DCN $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_cr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DCR=CR $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_tt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DTT $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_tc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DTC $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_ct$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DCT $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_cc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DCC $< -o $@
|
||||
|
||||
ifndef CGEMM_SMALL_K_B0_NN
|
||||
CGEMM_SMALL_K_B0_NN = ../generic/zgemm_small_matrix_kernel_nn.c
|
||||
endif
|
||||
|
||||
ifndef CGEMM_SMALL_K_B0_NT
|
||||
CGEMM_SMALL_K_B0_NT = ../generic/zgemm_small_matrix_kernel_nt.c
|
||||
endif
|
||||
|
||||
ifndef CGEMM_SMALL_K_B0_TN
|
||||
CGEMM_SMALL_K_B0_TN = ../generic/zgemm_small_matrix_kernel_tn.c
|
||||
endif
|
||||
|
||||
ifndef CGEMM_SMALL_K_B0_TT
|
||||
CGEMM_SMALL_K_B0_TT = ../generic/zgemm_small_matrix_kernel_tt.c
|
||||
endif
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_nn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DNN -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_nr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DNR -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_rn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DRN -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_rr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DRR -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_nt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DNT -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_nc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DNC -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_rt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DRT -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_rc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DRC=RC -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_tn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DTN -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_tr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DTR -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_cn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DCN -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_cr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DCR=CR -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_tt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DTT -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_tc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DTC -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_ct$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DCT -DB0 $< -o $@
|
||||
|
||||
$(KDIR)cgemm_small_kernel_b0_cc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(CGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -DCOMPLEX -DCC -DB0 $< -o $@
|
||||
|
||||
ifndef ZGEMM_SMALL_M_PERMIT
|
||||
ZGEMM_SMALL_M_PERMIT = ../generic/zgemm_small_matrix_permit.c
|
||||
endif
|
||||
|
||||
ifndef ZGEMM_SMALL_K_NN
|
||||
ZGEMM_SMALL_K_NN = ../generic/zgemm_small_matrix_kernel_nn.c
|
||||
endif
|
||||
|
||||
ifndef ZGEMM_SMALL_K_NT
|
||||
ZGEMM_SMALL_K_NT = ../generic/zgemm_small_matrix_kernel_nt.c
|
||||
endif
|
||||
|
||||
ifndef ZGEMM_SMALL_K_TN
|
||||
ZGEMM_SMALL_K_TN = ../generic/zgemm_small_matrix_kernel_tn.c
|
||||
endif
|
||||
|
||||
ifndef ZGEMM_SMALL_K_TT
|
||||
ZGEMM_SMALL_K_TT = ../generic/zgemm_small_matrix_kernel_tt.c
|
||||
endif
|
||||
|
||||
$(KDIR)zgemm_small_matrix_permit$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_M_PERMIT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX $< -o $@
|
||||
|
||||
|
||||
$(KDIR)zgemm_small_kernel_nn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DNN $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_nr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DNR $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_rn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DRN $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_rr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_NN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DRR $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_nt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DNT $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_nc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DNC $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_rt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DRT $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_rc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_NT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DRC=RC $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_tn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DTN $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_tr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DTR $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_cn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DCN $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_cr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_TN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DCR=CR $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_tt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DTT $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_tc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DTC $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_ct$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DCT $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_cc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_TT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DCC $< -o $@
|
||||
|
||||
ifndef ZGEMM_SMALL_K_B0_NN
|
||||
ZGEMM_SMALL_K_B0_NN = ../generic/zgemm_small_matrix_kernel_nn.c
|
||||
endif
|
||||
|
||||
ifndef ZGEMM_SMALL_K_B0_NT
|
||||
ZGEMM_SMALL_K_B0_NT = ../generic/zgemm_small_matrix_kernel_nt.c
|
||||
endif
|
||||
|
||||
ifndef ZGEMM_SMALL_K_B0_TN
|
||||
ZGEMM_SMALL_K_B0_TN = ../generic/zgemm_small_matrix_kernel_tn.c
|
||||
endif
|
||||
|
||||
ifndef ZGEMM_SMALL_K_B0_TT
|
||||
ZGEMM_SMALL_K_B0_TT = ../generic/zgemm_small_matrix_kernel_tt.c
|
||||
endif
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_nn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DNN -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_nr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DNR -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_rn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DRN -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_rr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_NN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DRR -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_nt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DNT -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_nc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DNC -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_rt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DRT -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_rc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_NT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DRC=RC -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_tn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DTN -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_tr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DTR -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_cn$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DCN -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_cr$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_TN)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DCR=CR -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_tt$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DTT -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_tc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DTC -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_ct$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DCT -DB0 $< -o $@
|
||||
|
||||
$(KDIR)zgemm_small_kernel_b0_cc$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(ZGEMM_SMALL_K_B0_TT)
|
||||
$(CC) $(CFLAGS) -c -DDOUBLE -DCOMPLEX -DCC -DB0 $< -o $@
|
||||
|
||||
@@ -0,0 +1,183 @@
|
||||
SAMINKERNEL = ../arm/amin.c
|
||||
DAMINKERNEL = ../arm/amin.c
|
||||
CAMINKERNEL = ../arm/zamin.c
|
||||
ZAMINKERNEL = ../arm/zamin.c
|
||||
|
||||
SMAXKERNEL = ../arm/max.c
|
||||
DMAXKERNEL = ../arm/max.c
|
||||
|
||||
SMINKERNEL = ../arm/min.c
|
||||
DMINKERNEL = ../arm/min.c
|
||||
|
||||
ISAMINKERNEL = ../arm/iamin.c
|
||||
IDAMINKERNEL = ../arm/iamin.c
|
||||
ICAMINKERNEL = ../arm/izamin.c
|
||||
IZAMINKERNEL = ../arm/izamin.c
|
||||
|
||||
ISMAXKERNEL = ../arm/imax.c
|
||||
IDMAXKERNEL = ../arm/imax.c
|
||||
|
||||
ISMINKERNEL = ../arm/imin.c
|
||||
IDMINKERNEL = ../arm/imin.c
|
||||
|
||||
STRSMKERNEL_LN = ../generic/trsm_kernel_LN.c
|
||||
STRSMKERNEL_LT = ../generic/trsm_kernel_LT.c
|
||||
STRSMKERNEL_RN = ../generic/trsm_kernel_RN.c
|
||||
STRSMKERNEL_RT = ../generic/trsm_kernel_RT.c
|
||||
|
||||
DTRSMKERNEL_LN = ../generic/trsm_kernel_LN.c
|
||||
DTRSMKERNEL_LT = ../generic/trsm_kernel_LT.c
|
||||
DTRSMKERNEL_RN = ../generic/trsm_kernel_RN.c
|
||||
DTRSMKERNEL_RT = ../generic/trsm_kernel_RT.c
|
||||
|
||||
CTRSMKERNEL_LN = ../generic/trsm_kernel_LN.c
|
||||
CTRSMKERNEL_LT = ../generic/trsm_kernel_LT.c
|
||||
CTRSMKERNEL_RN = ../generic/trsm_kernel_RN.c
|
||||
CTRSMKERNEL_RT = ../generic/trsm_kernel_RT.c
|
||||
|
||||
ZTRSMKERNEL_LN = ../generic/trsm_kernel_LN.c
|
||||
ZTRSMKERNEL_LT = ../generic/trsm_kernel_LT.c
|
||||
ZTRSMKERNEL_RN = ../generic/trsm_kernel_RN.c
|
||||
ZTRSMKERNEL_RT = ../generic/trsm_kernel_RT.c
|
||||
|
||||
SAMAXKERNEL = amax.S
|
||||
DAMAXKERNEL = amax.S
|
||||
CAMAXKERNEL = zamax.S
|
||||
ZAMAXKERNEL = zamax.S
|
||||
|
||||
SAXPYKERNEL = axpy.S
|
||||
DAXPYKERNEL = axpy.S
|
||||
CAXPYKERNEL = zaxpy.S
|
||||
ZAXPYKERNEL = zaxpy.S
|
||||
|
||||
SROTKERNEL = rot.S
|
||||
DROTKERNEL = rot.S
|
||||
CROTKERNEL = zrot.S
|
||||
ZROTKERNEL = zrot.S
|
||||
|
||||
SSCALKERNEL = scal.S
|
||||
DSCALKERNEL = scal.S
|
||||
CSCALKERNEL = zscal.S
|
||||
ZSCALKERNEL = zscal.S
|
||||
|
||||
SGEMVNKERNEL = gemv_n.S
|
||||
DGEMVNKERNEL = gemv_n.S
|
||||
CGEMVNKERNEL = zgemv_n.S
|
||||
ZGEMVNKERNEL = zgemv_n.S
|
||||
|
||||
SGEMVTKERNEL = gemv_t.S
|
||||
DGEMVTKERNEL = gemv_t.S
|
||||
CGEMVTKERNEL = zgemv_t.S
|
||||
ZGEMVTKERNEL = zgemv_t.S
|
||||
|
||||
|
||||
SASUMKERNEL = asum.S
|
||||
DASUMKERNEL = asum.S
|
||||
CASUMKERNEL = casum.S
|
||||
ZASUMKERNEL = zasum.S
|
||||
|
||||
SCOPYKERNEL = copy.S
|
||||
DCOPYKERNEL = copy.S
|
||||
CCOPYKERNEL = copy.S
|
||||
ZCOPYKERNEL = copy.S
|
||||
|
||||
SSWAPKERNEL = swap.S
|
||||
DSWAPKERNEL = swap.S
|
||||
CSWAPKERNEL = swap.S
|
||||
ZSWAPKERNEL = swap.S
|
||||
|
||||
ISAMAXKERNEL = iamax.S
|
||||
IDAMAXKERNEL = iamax.S
|
||||
ICAMAXKERNEL = izamax.S
|
||||
IZAMAXKERNEL = izamax.S
|
||||
|
||||
SNRM2KERNEL = nrm2.S
|
||||
DNRM2KERNEL = nrm2.S
|
||||
CNRM2KERNEL = znrm2.S
|
||||
ZNRM2KERNEL = znrm2.S
|
||||
|
||||
DDOTKERNEL = dot.S
|
||||
ifneq ($(C_COMPILER), PGI)
|
||||
SDOTKERNEL = ../generic/dot.c
|
||||
else
|
||||
SDOTKERNEL = dot.S
|
||||
endif
|
||||
ifneq ($(C_COMPILER), PGI)
|
||||
CDOTKERNEL = zdot.S
|
||||
ZDOTKERNEL = zdot.S
|
||||
else
|
||||
CDOTKERNEL = ../arm/zdot.c
|
||||
ZDOTKERNEL = ../arm/zdot.c
|
||||
endif
|
||||
DSDOTKERNEL = dot.S
|
||||
|
||||
DGEMM_BETA = dgemm_beta.S
|
||||
SGEMM_BETA = sgemm_beta.S
|
||||
|
||||
SGEMMKERNEL = sgemm_kernel_sve_v2x$(SGEMM_UNROLL_N).S
|
||||
STRMMKERNEL = strmm_kernel_sve_v1x$(SGEMM_UNROLL_N).S
|
||||
|
||||
SGEMMINCOPY = sgemm_ncopy_sve_v1.c
|
||||
SGEMMITCOPY = sgemm_tcopy_sve_v1.c
|
||||
SGEMMONCOPY = sgemm_ncopy_$(DGEMM_UNROLL_N).S
|
||||
SGEMMOTCOPY = sgemm_tcopy_$(DGEMM_UNROLL_N).S
|
||||
|
||||
SGEMMINCOPYOBJ = sgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
SGEMMITCOPYOBJ = sgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
SGEMMONCOPYOBJ = sgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
SGEMMOTCOPYOBJ = sgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
STRMMUNCOPY_M = trmm_uncopy_sve_v1.c
|
||||
STRMMLNCOPY_M = trmm_lncopy_sve_v1.c
|
||||
STRMMUTCOPY_M = trmm_utcopy_sve_v1.c
|
||||
STRMMLTCOPY_M = trmm_ltcopy_sve_v1.c
|
||||
|
||||
SSYMMUCOPY_M = symm_ucopy_sve.c
|
||||
SSYMMLCOPY_M = symm_lcopy_sve.c
|
||||
|
||||
DGEMMKERNEL = dgemm_kernel_sve_v2x$(DGEMM_UNROLL_N).S
|
||||
DTRMMKERNEL = dtrmm_kernel_sve_v1x$(DGEMM_UNROLL_N).S
|
||||
|
||||
DGEMMINCOPY = dgemm_ncopy_sve_v1.c
|
||||
DGEMMITCOPY = dgemm_tcopy_sve_v1.c
|
||||
DGEMMONCOPY = dgemm_ncopy_$(DGEMM_UNROLL_N).S
|
||||
DGEMMOTCOPY = dgemm_tcopy_$(DGEMM_UNROLL_N).S
|
||||
|
||||
DGEMMINCOPYOBJ = dgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
DGEMMITCOPYOBJ = dgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
DGEMMONCOPYOBJ = dgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
DGEMMOTCOPYOBJ = dgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
DTRMMUNCOPY_M = trmm_uncopy_sve_v1.c
|
||||
DTRMMLNCOPY_M = trmm_lncopy_sve_v1.c
|
||||
DTRMMUTCOPY_M = trmm_utcopy_sve_v1.c
|
||||
DTRMMLTCOPY_M = trmm_ltcopy_sve_v1.c
|
||||
|
||||
DSYMMUCOPY_M = symm_ucopy_sve.c
|
||||
DSYMMLCOPY_M = symm_lcopy_sve.c
|
||||
|
||||
CGEMMKERNEL = cgemm_kernel_$(CGEMM_UNROLL_M)x$(CGEMM_UNROLL_N).S
|
||||
CTRMMKERNEL = ctrmm_kernel_$(CGEMM_UNROLL_M)x$(CGEMM_UNROLL_N).S
|
||||
ifneq ($(CGEMM_UNROLL_M), $(CGEMM_UNROLL_N))
|
||||
CGEMMINCOPY = ../generic/zgemm_ncopy_$(CGEMM_UNROLL_M).c
|
||||
CGEMMITCOPY = ../generic/zgemm_tcopy_$(CGEMM_UNROLL_M).c
|
||||
CGEMMINCOPYOBJ = cgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
CGEMMITCOPYOBJ = cgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
endif
|
||||
CGEMMONCOPY = ../generic/zgemm_ncopy_$(CGEMM_UNROLL_N).c
|
||||
CGEMMOTCOPY = ../generic/zgemm_tcopy_$(CGEMM_UNROLL_N).c
|
||||
CGEMMONCOPYOBJ = cgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
CGEMMOTCOPYOBJ = cgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
ZGEMMKERNEL = zgemm_kernel_$(ZGEMM_UNROLL_M)x$(ZGEMM_UNROLL_N).S
|
||||
ZTRMMKERNEL = ztrmm_kernel_$(ZGEMM_UNROLL_M)x$(ZGEMM_UNROLL_N).S
|
||||
ifneq ($(ZGEMM_UNROLL_M), $(ZGEMM_UNROLL_N))
|
||||
ZGEMMINCOPY = ../generic/zgemm_ncopy_$(ZGEMM_UNROLL_M).c
|
||||
ZGEMMITCOPY = ../generic/zgemm_tcopy_$(ZGEMM_UNROLL_M).c
|
||||
ZGEMMINCOPYOBJ = zgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
ZGEMMITCOPYOBJ = zgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
endif
|
||||
ZGEMMONCOPY = ../generic/zgemm_ncopy_$(ZGEMM_UNROLL_N).c
|
||||
ZGEMMOTCOPY = ../generic/zgemm_tcopy_$(ZGEMM_UNROLL_N).c
|
||||
ZGEMMONCOPYOBJ = zgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
ZGEMMOTCOPYOBJ = zgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
@@ -0,0 +1,183 @@
|
||||
SAMINKERNEL = ../arm/amin.c
|
||||
DAMINKERNEL = ../arm/amin.c
|
||||
CAMINKERNEL = ../arm/zamin.c
|
||||
ZAMINKERNEL = ../arm/zamin.c
|
||||
|
||||
SMAXKERNEL = ../arm/max.c
|
||||
DMAXKERNEL = ../arm/max.c
|
||||
|
||||
SMINKERNEL = ../arm/min.c
|
||||
DMINKERNEL = ../arm/min.c
|
||||
|
||||
ISAMINKERNEL = ../arm/iamin.c
|
||||
IDAMINKERNEL = ../arm/iamin.c
|
||||
ICAMINKERNEL = ../arm/izamin.c
|
||||
IZAMINKERNEL = ../arm/izamin.c
|
||||
|
||||
ISMAXKERNEL = ../arm/imax.c
|
||||
IDMAXKERNEL = ../arm/imax.c
|
||||
|
||||
ISMINKERNEL = ../arm/imin.c
|
||||
IDMINKERNEL = ../arm/imin.c
|
||||
|
||||
STRSMKERNEL_LN = ../generic/trsm_kernel_LN.c
|
||||
STRSMKERNEL_LT = ../generic/trsm_kernel_LT.c
|
||||
STRSMKERNEL_RN = ../generic/trsm_kernel_RN.c
|
||||
STRSMKERNEL_RT = ../generic/trsm_kernel_RT.c
|
||||
|
||||
DTRSMKERNEL_LN = ../generic/trsm_kernel_LN.c
|
||||
DTRSMKERNEL_LT = ../generic/trsm_kernel_LT.c
|
||||
DTRSMKERNEL_RN = ../generic/trsm_kernel_RN.c
|
||||
DTRSMKERNEL_RT = ../generic/trsm_kernel_RT.c
|
||||
|
||||
CTRSMKERNEL_LN = ../generic/trsm_kernel_LN.c
|
||||
CTRSMKERNEL_LT = ../generic/trsm_kernel_LT.c
|
||||
CTRSMKERNEL_RN = ../generic/trsm_kernel_RN.c
|
||||
CTRSMKERNEL_RT = ../generic/trsm_kernel_RT.c
|
||||
|
||||
ZTRSMKERNEL_LN = ../generic/trsm_kernel_LN.c
|
||||
ZTRSMKERNEL_LT = ../generic/trsm_kernel_LT.c
|
||||
ZTRSMKERNEL_RN = ../generic/trsm_kernel_RN.c
|
||||
ZTRSMKERNEL_RT = ../generic/trsm_kernel_RT.c
|
||||
|
||||
SAMAXKERNEL = amax.S
|
||||
DAMAXKERNEL = amax.S
|
||||
CAMAXKERNEL = zamax.S
|
||||
ZAMAXKERNEL = zamax.S
|
||||
|
||||
SAXPYKERNEL = axpy.S
|
||||
DAXPYKERNEL = axpy.S
|
||||
CAXPYKERNEL = zaxpy.S
|
||||
ZAXPYKERNEL = zaxpy.S
|
||||
|
||||
SROTKERNEL = rot.S
|
||||
DROTKERNEL = rot.S
|
||||
CROTKERNEL = zrot.S
|
||||
ZROTKERNEL = zrot.S
|
||||
|
||||
SSCALKERNEL = scal.S
|
||||
DSCALKERNEL = scal.S
|
||||
CSCALKERNEL = zscal.S
|
||||
ZSCALKERNEL = zscal.S
|
||||
|
||||
SGEMVNKERNEL = gemv_n.S
|
||||
DGEMVNKERNEL = gemv_n.S
|
||||
CGEMVNKERNEL = zgemv_n.S
|
||||
ZGEMVNKERNEL = zgemv_n.S
|
||||
|
||||
SGEMVTKERNEL = gemv_t.S
|
||||
DGEMVTKERNEL = gemv_t.S
|
||||
CGEMVTKERNEL = zgemv_t.S
|
||||
ZGEMVTKERNEL = zgemv_t.S
|
||||
|
||||
|
||||
SASUMKERNEL = asum.S
|
||||
DASUMKERNEL = asum.S
|
||||
CASUMKERNEL = casum.S
|
||||
ZASUMKERNEL = zasum.S
|
||||
|
||||
SCOPYKERNEL = copy.S
|
||||
DCOPYKERNEL = copy.S
|
||||
CCOPYKERNEL = copy.S
|
||||
ZCOPYKERNEL = copy.S
|
||||
|
||||
SSWAPKERNEL = swap.S
|
||||
DSWAPKERNEL = swap.S
|
||||
CSWAPKERNEL = swap.S
|
||||
ZSWAPKERNEL = swap.S
|
||||
|
||||
ISAMAXKERNEL = iamax.S
|
||||
IDAMAXKERNEL = iamax.S
|
||||
ICAMAXKERNEL = izamax.S
|
||||
IZAMAXKERNEL = izamax.S
|
||||
|
||||
SNRM2KERNEL = nrm2.S
|
||||
DNRM2KERNEL = nrm2.S
|
||||
CNRM2KERNEL = znrm2.S
|
||||
ZNRM2KERNEL = znrm2.S
|
||||
|
||||
DDOTKERNEL = dot.S
|
||||
ifneq ($(C_COMPILER), PGI)
|
||||
SDOTKERNEL = ../generic/dot.c
|
||||
else
|
||||
SDOTKERNEL = dot.S
|
||||
endif
|
||||
ifneq ($(C_COMPILER), PGI)
|
||||
CDOTKERNEL = zdot.S
|
||||
ZDOTKERNEL = zdot.S
|
||||
else
|
||||
CDOTKERNEL = ../arm/zdot.c
|
||||
ZDOTKERNEL = ../arm/zdot.c
|
||||
endif
|
||||
DSDOTKERNEL = dot.S
|
||||
|
||||
DGEMM_BETA = dgemm_beta.S
|
||||
SGEMM_BETA = sgemm_beta.S
|
||||
|
||||
SGEMMKERNEL = sgemm_kernel_sve_v2x$(SGEMM_UNROLL_N).S
|
||||
STRMMKERNEL = strmm_kernel_sve_v1x$(SGEMM_UNROLL_N).S
|
||||
|
||||
SGEMMINCOPY = sgemm_ncopy_sve_v1.c
|
||||
SGEMMITCOPY = sgemm_tcopy_sve_v1.c
|
||||
SGEMMONCOPY = sgemm_ncopy_$(DGEMM_UNROLL_N).S
|
||||
SGEMMOTCOPY = sgemm_tcopy_$(DGEMM_UNROLL_N).S
|
||||
|
||||
SGEMMINCOPYOBJ = sgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
SGEMMITCOPYOBJ = sgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
SGEMMONCOPYOBJ = sgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
SGEMMOTCOPYOBJ = sgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
STRMMUNCOPY_M = trmm_uncopy_sve_v1.c
|
||||
STRMMLNCOPY_M = trmm_lncopy_sve_v1.c
|
||||
STRMMUTCOPY_M = trmm_utcopy_sve_v1.c
|
||||
STRMMLTCOPY_M = trmm_ltcopy_sve_v1.c
|
||||
|
||||
SSYMMUCOPY_M = symm_ucopy_sve.c
|
||||
SSYMMLCOPY_M = symm_lcopy_sve.c
|
||||
|
||||
DGEMMKERNEL = dgemm_kernel_sve_v2x$(DGEMM_UNROLL_N).S
|
||||
DTRMMKERNEL = dtrmm_kernel_sve_v1x$(DGEMM_UNROLL_N).S
|
||||
|
||||
DGEMMINCOPY = dgemm_ncopy_sve_v1.c
|
||||
DGEMMITCOPY = dgemm_tcopy_sve_v1.c
|
||||
DGEMMONCOPY = ../generic/gemm_ncopy_$(DGEMM_UNROLL_N).c
|
||||
DGEMMOTCOPY = ../generic/gemm_tcopy_$(DGEMM_UNROLL_N).c
|
||||
|
||||
DGEMMINCOPYOBJ = dgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
DGEMMITCOPYOBJ = dgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
DGEMMONCOPYOBJ = dgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
DGEMMOTCOPYOBJ = dgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
DTRMMUNCOPY_M = trmm_uncopy_sve_v1.c
|
||||
DTRMMLNCOPY_M = trmm_lncopy_sve_v1.c
|
||||
DTRMMUTCOPY_M = trmm_utcopy_sve_v1.c
|
||||
DTRMMLTCOPY_M = trmm_ltcopy_sve_v1.c
|
||||
|
||||
DSYMMUCOPY_M = symm_ucopy_sve.c
|
||||
DSYMMLCOPY_M = symm_lcopy_sve.c
|
||||
|
||||
CGEMMKERNEL = cgemm_kernel_$(CGEMM_UNROLL_M)x$(CGEMM_UNROLL_N).S
|
||||
CTRMMKERNEL = ctrmm_kernel_$(CGEMM_UNROLL_M)x$(CGEMM_UNROLL_N).S
|
||||
ifneq ($(CGEMM_UNROLL_M), $(CGEMM_UNROLL_N))
|
||||
CGEMMINCOPY = ../generic/zgemm_ncopy_$(CGEMM_UNROLL_M).c
|
||||
CGEMMITCOPY = ../generic/zgemm_tcopy_$(CGEMM_UNROLL_M).c
|
||||
CGEMMINCOPYOBJ = cgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
CGEMMITCOPYOBJ = cgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
endif
|
||||
CGEMMONCOPY = ../generic/zgemm_ncopy_$(CGEMM_UNROLL_N).c
|
||||
CGEMMOTCOPY = ../generic/zgemm_tcopy_$(CGEMM_UNROLL_N).c
|
||||
CGEMMONCOPYOBJ = cgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
CGEMMOTCOPYOBJ = cgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
ZGEMMKERNEL = zgemm_kernel_$(ZGEMM_UNROLL_M)x$(ZGEMM_UNROLL_N).S
|
||||
ZTRMMKERNEL = ztrmm_kernel_$(ZGEMM_UNROLL_M)x$(ZGEMM_UNROLL_N).S
|
||||
ifneq ($(ZGEMM_UNROLL_M), $(ZGEMM_UNROLL_N))
|
||||
ZGEMMINCOPY = ../generic/zgemm_ncopy_$(ZGEMM_UNROLL_M).c
|
||||
ZGEMMITCOPY = ../generic/zgemm_tcopy_$(ZGEMM_UNROLL_M).c
|
||||
ZGEMMINCOPYOBJ = zgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
ZGEMMITCOPYOBJ = zgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
endif
|
||||
ZGEMMONCOPY = ../generic/zgemm_ncopy_$(ZGEMM_UNROLL_N).c
|
||||
ZGEMMOTCOPY = ../generic/zgemm_tcopy_$(ZGEMM_UNROLL_N).c
|
||||
ZGEMMONCOPYOBJ = zgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
ZGEMMOTCOPYOBJ = zgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
@@ -141,7 +141,7 @@ SGEMMONCOPY = sgemm_ncopy_$(SGEMM_UNROLL_N).S
|
||||
SGEMMONCOPYOBJ = sgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
SGEMMOTCOPYOBJ = sgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
DGEMMKERNEL = dgemm_kernel_$(DGEMM_UNROLL_M)x$(DGEMM_UNROLL_N).S
|
||||
DGEMMKERNEL = dgemm_kernel_$(DGEMM_UNROLL_M)x$(DGEMM_UNROLL_N)_cortexa53.c
|
||||
DTRMMKERNEL = dtrmm_kernel_$(DGEMM_UNROLL_M)x$(DGEMM_UNROLL_N).S
|
||||
|
||||
ifneq ($(DGEMM_UNROLL_M), $(DGEMM_UNROLL_N))
|
||||
@@ -169,7 +169,7 @@ endif
|
||||
DGEMMONCOPYOBJ = dgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
DGEMMOTCOPYOBJ = dgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
CGEMMKERNEL = cgemm_kernel_$(CGEMM_UNROLL_M)x$(CGEMM_UNROLL_N).S
|
||||
CGEMMKERNEL = cgemm_kernel_$(CGEMM_UNROLL_M)x$(CGEMM_UNROLL_N)_cortexa53.c
|
||||
CTRMMKERNEL = ctrmm_kernel_$(CGEMM_UNROLL_M)x$(CGEMM_UNROLL_N).S
|
||||
ifneq ($(CGEMM_UNROLL_M), $(CGEMM_UNROLL_N))
|
||||
CGEMMINCOPY = ../generic/zgemm_ncopy_$(CGEMM_UNROLL_M).c
|
||||
@@ -182,7 +182,7 @@ CGEMMOTCOPY = ../generic/zgemm_tcopy_$(CGEMM_UNROLL_N).c
|
||||
CGEMMONCOPYOBJ = cgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
CGEMMOTCOPYOBJ = cgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
ZGEMMKERNEL = zgemm_kernel_$(ZGEMM_UNROLL_M)x$(ZGEMM_UNROLL_N).S
|
||||
ZGEMMKERNEL = zgemm_kernel_$(ZGEMM_UNROLL_M)x$(ZGEMM_UNROLL_N)_cortexa53.c
|
||||
ZTRMMKERNEL = ztrmm_kernel_$(ZGEMM_UNROLL_M)x$(ZGEMM_UNROLL_N).S
|
||||
ifneq ($(ZGEMM_UNROLL_M), $(ZGEMM_UNROLL_N))
|
||||
ZGEMMINCOPY = ../generic/zgemm_ncopy_$(ZGEMM_UNROLL_M).c
|
||||
|
||||
@@ -141,7 +141,7 @@ SGEMMONCOPY = sgemm_ncopy_$(SGEMM_UNROLL_N).S
|
||||
SGEMMONCOPYOBJ = sgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
SGEMMOTCOPYOBJ = sgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
DGEMMKERNEL = dgemm_kernel_$(DGEMM_UNROLL_M)x$(DGEMM_UNROLL_N).S
|
||||
DGEMMKERNEL = dgemm_kernel_$(DGEMM_UNROLL_M)x$(DGEMM_UNROLL_N)_cortexa53.c
|
||||
DTRMMKERNEL = dtrmm_kernel_$(DGEMM_UNROLL_M)x$(DGEMM_UNROLL_N).S
|
||||
|
||||
ifneq ($(DGEMM_UNROLL_M), $(DGEMM_UNROLL_N))
|
||||
@@ -169,7 +169,7 @@ endif
|
||||
DGEMMONCOPYOBJ = dgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
DGEMMOTCOPYOBJ = dgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
CGEMMKERNEL = cgemm_kernel_$(CGEMM_UNROLL_M)x$(CGEMM_UNROLL_N).S
|
||||
CGEMMKERNEL = cgemm_kernel_$(CGEMM_UNROLL_M)x$(CGEMM_UNROLL_N)_cortexa53.c
|
||||
CTRMMKERNEL = ctrmm_kernel_$(CGEMM_UNROLL_M)x$(CGEMM_UNROLL_N).S
|
||||
ifneq ($(CGEMM_UNROLL_M), $(CGEMM_UNROLL_N))
|
||||
CGEMMINCOPY = ../generic/zgemm_ncopy_$(CGEMM_UNROLL_M).c
|
||||
@@ -182,7 +182,7 @@ CGEMMOTCOPY = ../generic/zgemm_tcopy_$(CGEMM_UNROLL_N).c
|
||||
CGEMMONCOPYOBJ = cgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
CGEMMOTCOPYOBJ = cgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
|
||||
ZGEMMKERNEL = zgemm_kernel_$(ZGEMM_UNROLL_M)x$(ZGEMM_UNROLL_N).S
|
||||
ZGEMMKERNEL = zgemm_kernel_$(ZGEMM_UNROLL_M)x$(ZGEMM_UNROLL_N)_cortexa53.c
|
||||
ZTRMMKERNEL = ztrmm_kernel_$(ZGEMM_UNROLL_M)x$(ZGEMM_UNROLL_N).S
|
||||
ifneq ($(ZGEMM_UNROLL_M), $(ZGEMM_UNROLL_N))
|
||||
ZGEMMINCOPY = ../generic/zgemm_ncopy_$(ZGEMM_UNROLL_M).c
|
||||
|
||||
@@ -1 +1 @@
|
||||
include $(KERNELDIR)/KERNEL.ARMV8
|
||||
include $(KERNELDIR)/KERNEL.NEOVERSEN1
|
||||
|
||||
@@ -0,0 +1,898 @@
|
||||
/***************************************************************************
|
||||
Copyright (c) 2021, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written permission.
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
|
||||
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*****************************************************************************/
|
||||
|
||||
#include "common.h"
|
||||
#include <arm_neon.h>
|
||||
|
||||
#if defined(NN) || defined(NT) || defined(TN) || defined(TT)
|
||||
#define FMLA_RI "fmla "
|
||||
#define FMLA_IR "fmla "
|
||||
#define FMLA_II "fmls "
|
||||
#elif defined(NR) || defined(NC) || defined(TR) || defined(TC)
|
||||
#define FMLA_RI "fmls "
|
||||
#define FMLA_IR "fmla "
|
||||
#define FMLA_II "fmla "
|
||||
#elif defined(RN) || defined(RT) || defined(CN) || defined(CT)
|
||||
#define FMLA_RI "fmla "
|
||||
#define FMLA_IR "fmls "
|
||||
#define FMLA_II "fmla "
|
||||
#else
|
||||
#define FMLA_RI "fmls "
|
||||
#define FMLA_IR "fmls "
|
||||
#define FMLA_II "fmls "
|
||||
#endif
|
||||
#define FMLA_RR "fmla "
|
||||
|
||||
static inline void store_m8n1_contracted(float *C,
|
||||
float32x4_t c1r, float32x4_t c1i, float32x4_t c2r, float32x4_t c2i,
|
||||
float alphar, float alphai) {
|
||||
|
||||
float32x4x2_t ld1 = vld2q_f32(C), ld2 = vld2q_f32(C + 8);
|
||||
ld1.val[0] = vfmaq_n_f32(ld1.val[0], c1r, alphar);
|
||||
ld2.val[0] = vfmaq_n_f32(ld2.val[0], c2r, alphar);
|
||||
ld1.val[1] = vfmaq_n_f32(ld1.val[1], c1r, alphai);
|
||||
ld2.val[1] = vfmaq_n_f32(ld2.val[1], c2r, alphai);
|
||||
ld1.val[0] = vfmsq_n_f32(ld1.val[0], c1i, alphai);
|
||||
ld2.val[0] = vfmsq_n_f32(ld2.val[0], c2i, alphai);
|
||||
ld1.val[1] = vfmaq_n_f32(ld1.val[1], c1i, alphar);
|
||||
ld2.val[1] = vfmaq_n_f32(ld2.val[1], c2i, alphar);
|
||||
vst2q_f32(C, ld1);
|
||||
vst2q_f32(C + 8, ld2);
|
||||
}
|
||||
|
||||
static inline void kernel_8x4(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K, BLASLONG LDC) {
|
||||
|
||||
const float *c_pref = C;
|
||||
float32x4_t c1r, c1i, c2r, c2i, c3r, c3i, c4r, c4i;
|
||||
float32x4_t c5r, c5i, c6r, c6i, c7r, c7i, c8r, c8i;
|
||||
|
||||
/** x0 for filling A, x1-x6 for filling B (x5 and x6 for real, x2 and x4 for imag) */
|
||||
/** v0-v1 and v10-v11 for B, v2-v9 for A */
|
||||
__asm__ __volatile__(
|
||||
"cmp %[K],#0; mov %[c_pref],%[C]\n\t"
|
||||
"movi %[c1r].16b,#0; prfm pstl1keep,[%[c_pref]]\n\t"
|
||||
"movi %[c1i].16b,#0; prfm pstl1keep,[%[c_pref],#64]\n\t"
|
||||
"movi %[c2r].16b,#0; add %[c_pref],%[c_pref],%[LDC],LSL#3\n\t"
|
||||
"movi %[c2i].16b,#0; prfm pstl1keep,[%[c_pref]]\n\t"
|
||||
"movi %[c3r].16b,#0; prfm pstl1keep,[%[c_pref],#64]\n\t"
|
||||
"movi %[c3i].16b,#0; add %[c_pref],%[c_pref],%[LDC],LSL#3\n\t"
|
||||
"movi %[c4r].16b,#0; prfm pstl1keep,[%[c_pref]]\n\t"
|
||||
"movi %[c4i].16b,#0; prfm pstl1keep,[%[c_pref],#64]\n\t"
|
||||
"movi %[c5r].16b,#0; add %[c_pref],%[c_pref],%[LDC],LSL#3\n\t"
|
||||
"movi %[c5i].16b,#0; prfm pstl1keep,[%[c_pref]]\n\t"
|
||||
"movi %[c6r].16b,#0; prfm pstl1keep,[%[c_pref],#64]\n\t"
|
||||
"movi %[c6i].16b,#0\n\t"
|
||||
"movi %[c7r].16b,#0; movi %[c7i].16b,#0\n\t"
|
||||
"movi %[c8r].16b,#0; movi %[c8i].16b,#0\n\t"
|
||||
"beq 4f\n\t"
|
||||
"cmp %[K],#2\n\t"
|
||||
"ldp x1,x2,[%[sb]],#16; ldr q2,[%[sa]],#64\n\t"
|
||||
"ldp x3,x4,[%[sb]],#16; ldr d3,[%[sa],#-48]\n\t"
|
||||
"mov w5,w1; mov w6,w3; ldr x0,[%[sa],#-40]\n\t"
|
||||
"bfi x5,x2,#32,#32; bfi x6,x4,#32,#32; fmov d0,x5\n\t"
|
||||
"bfxil x2,x1,#32,#32; bfxil x4,x3,#32,#32; fmov v0.d[1],x6\n\t"
|
||||
|
||||
"blt 3f; beq 2f\n\t"
|
||||
"1:\n\t"
|
||||
"fmov v3.d[1],x0; ldr d4,[%[sa],#-32]\n\t"
|
||||
FMLA_RR "%[c1r].4s,v0.4s,v2.s[0]; ldr x0,[%[sa],#-24]\n\t"
|
||||
FMLA_IR "%[c1i].4s,v0.4s,v2.s[1]; ldr x1,[%[sb]],#64\n\t"
|
||||
FMLA_RR "%[c2r].4s,v0.4s,v2.s[2]\n\t"
|
||||
"fmov v4.d[1],x0; ldr d5,[%[sa],#-16]\n\t"
|
||||
FMLA_IR "%[c2i].4s,v0.4s,v2.s[3]; ldr x0,[%[sa],#-8]\n\t"
|
||||
FMLA_RR "%[c3r].4s,v0.4s,v3.s[0]; mov w5,w1\n\t"
|
||||
FMLA_IR "%[c3i].4s,v0.4s,v3.s[1]\n\t"
|
||||
"fmov v5.d[1],x0; fmov d1,x2\n\t"
|
||||
FMLA_RR "%[c4r].4s,v0.4s,v3.s[2]; ldr x2,[%[sb],#-56]\n\t"
|
||||
FMLA_IR "%[c4i].4s,v0.4s,v3.s[3]; ldr x3,[%[sb],#-48]\n\t"
|
||||
FMLA_RR "%[c5r].4s,v0.4s,v4.s[0]\n\t"
|
||||
"fmov v1.d[1],x4; ldr d6,[%[sa]]\n\t"
|
||||
FMLA_IR "%[c5i].4s,v0.4s,v4.s[1]; ldr x0,[%[sa],#8]\n\t"
|
||||
FMLA_RR "%[c6r].4s,v0.4s,v4.s[2]; ldr x4,[%[sb],#-40]\n\t"
|
||||
FMLA_IR "%[c6i].4s,v0.4s,v4.s[3]; bfi x5,x2,#32,#32\n\t"
|
||||
"fmov v6.d[1],x0; ldr d7,[%[sa],#16]\n\t"
|
||||
FMLA_RR "%[c7r].4s,v0.4s,v5.s[0]; ldr x0,[%[sa],#24]\n\t"
|
||||
FMLA_IR "%[c7i].4s,v0.4s,v5.s[1]; mov w6,w3\n\t"
|
||||
FMLA_RR "%[c8r].4s,v0.4s,v5.s[2]; bfxil x2,x1,#32,#32\n\t"
|
||||
"fmov v7.d[1],x0; fmov d10,x5\n\t"
|
||||
FMLA_IR "%[c8i].4s,v0.4s,v5.s[3]; bfi x6,x4,#32,#32\n\t"
|
||||
FMLA_II "%[c1r].4s,v1.4s,v2.s[1]; ldr x1,[%[sb],#-32]\n\t"
|
||||
FMLA_RI "%[c1i].4s,v1.4s,v2.s[0]; bfxil x4,x3,#32,#32\n\t"
|
||||
"fmov v10.d[1],x6; fmov d11,x2\n\t"
|
||||
FMLA_II "%[c2r].4s,v1.4s,v2.s[3]; ldr x2,[%[sb],#-24]\n\t"
|
||||
FMLA_RI "%[c2i].4s,v1.4s,v2.s[2]; ldr x3,[%[sb],#-16]\n\t"
|
||||
FMLA_II "%[c3r].4s,v1.4s,v3.s[1]; mov w5,w1\n\t"
|
||||
"fmov v11.d[1],x4; ldr d8,[%[sa],#32]\n\t"
|
||||
FMLA_RI "%[c3i].4s,v1.4s,v3.s[0]; ldr x0,[%[sa],#40]\n\t"
|
||||
FMLA_II "%[c4r].4s,v1.4s,v3.s[3]; ldr x4,[%[sb],#-8]\n\t"
|
||||
FMLA_RI "%[c4i].4s,v1.4s,v3.s[2]; bfi x5,x2,#32,#32\n\t"
|
||||
"fmov v8.d[1],x0; ldr d9,[%[sa],#48]\n\t"
|
||||
FMLA_II "%[c5r].4s,v1.4s,v4.s[1]; ldr x0,[%[sa],#56]\n\t"
|
||||
FMLA_RI "%[c5i].4s,v1.4s,v4.s[0]; mov w6,w3\n\t"
|
||||
FMLA_II "%[c6r].4s,v1.4s,v4.s[3]\n\t"
|
||||
"fmov v9.d[1],x0; fmov d0,x5\n\t"
|
||||
FMLA_RI "%[c6i].4s,v1.4s,v4.s[2]; bfi x6,x4,#32,#32\n\t"
|
||||
FMLA_II "%[c7r].4s,v1.4s,v5.s[1]\n\t"
|
||||
FMLA_RI "%[c7i].4s,v1.4s,v5.s[0]\n\t"
|
||||
"fmov v0.d[1],x6; ldr d2,[%[sa],#64]\n\t"
|
||||
FMLA_II "%[c8r].4s,v1.4s,v5.s[3]; ldr x0,[%[sa],#72]\n\t"
|
||||
FMLA_RI "%[c8i].4s,v1.4s,v5.s[2]\n\t"
|
||||
FMLA_RR "%[c1r].4s,v10.4s,v6.s[0]\n\t"
|
||||
"fmov v2.d[1],x0; ldr d3,[%[sa],#80]\n\t"
|
||||
FMLA_IR "%[c1i].4s,v10.4s,v6.s[1]\n\t"
|
||||
FMLA_RR "%[c2r].4s,v10.4s,v6.s[2]; ldr x0,[%[sa],#88]\n\t"
|
||||
FMLA_IR "%[c2i].4s,v10.4s,v6.s[3]; bfxil x2,x1,#32,#32\n\t"
|
||||
FMLA_RR "%[c3r].4s,v10.4s,v7.s[0]; bfxil x4,x3,#32,#32\n\t"
|
||||
FMLA_IR "%[c3i].4s,v10.4s,v7.s[1]; add %[sa],%[sa],#128\n\t"
|
||||
FMLA_RR "%[c4r].4s,v10.4s,v7.s[2]; prfm pldl1keep,[%[sb],#128]\n\t"
|
||||
FMLA_IR "%[c4i].4s,v10.4s,v7.s[3]; sub %[K],%[K],#2\n\t"
|
||||
FMLA_RR "%[c5r].4s,v10.4s,v8.s[0]; prfm pldl1keep,[%[sa],#128]\n\t"
|
||||
FMLA_IR "%[c5i].4s,v10.4s,v8.s[1]; prfm pldl1keep,[%[sa],#192]\n\t"
|
||||
FMLA_RR "%[c6r].4s,v10.4s,v8.s[2]; cmp %[K],#2\n\t"
|
||||
FMLA_IR "%[c6i].4s,v10.4s,v8.s[3]\n\t"
|
||||
FMLA_RR "%[c7r].4s,v10.4s,v9.s[0]\n\t" FMLA_IR "%[c7i].4s,v10.4s,v9.s[1]\n\t"
|
||||
FMLA_RR "%[c8r].4s,v10.4s,v9.s[2]\n\t" FMLA_IR "%[c8i].4s,v10.4s,v9.s[3]\n\t"
|
||||
FMLA_II "%[c1r].4s,v11.4s,v6.s[1]\n\t" FMLA_RI "%[c1i].4s,v11.4s,v6.s[0]\n\t"
|
||||
FMLA_II "%[c2r].4s,v11.4s,v6.s[3]\n\t" FMLA_RI "%[c2i].4s,v11.4s,v6.s[2]\n\t"
|
||||
FMLA_II "%[c3r].4s,v11.4s,v7.s[1]\n\t" FMLA_RI "%[c3i].4s,v11.4s,v7.s[0]\n\t"
|
||||
FMLA_II "%[c4r].4s,v11.4s,v7.s[3]\n\t" FMLA_RI "%[c4i].4s,v11.4s,v7.s[2]\n\t"
|
||||
FMLA_II "%[c5r].4s,v11.4s,v8.s[1]\n\t" FMLA_RI "%[c5i].4s,v11.4s,v8.s[0]\n\t"
|
||||
FMLA_II "%[c6r].4s,v11.4s,v8.s[3]\n\t" FMLA_RI "%[c6i].4s,v11.4s,v8.s[2]\n\t"
|
||||
FMLA_II "%[c7r].4s,v11.4s,v9.s[1]\n\t" FMLA_RI "%[c7i].4s,v11.4s,v9.s[0]\n\t"
|
||||
FMLA_II "%[c8r].4s,v11.4s,v9.s[3]\n\t" FMLA_RI "%[c8i].4s,v11.4s,v9.s[2]\n\t"
|
||||
"bgt 1b; blt 3f\n\t"
|
||||
"2:\n\t"
|
||||
"fmov v3.d[1],x0; ldr d4,[%[sa],#-32]\n\t"
|
||||
FMLA_RR "%[c1r].4s,v0.4s,v2.s[0]; ldr x0,[%[sa],#-24]\n\t"
|
||||
FMLA_IR "%[c1i].4s,v0.4s,v2.s[1]; ldr x1,[%[sb]],#32\n\t"
|
||||
FMLA_RR "%[c2r].4s,v0.4s,v2.s[2]\n\t"
|
||||
"fmov v4.d[1],x0; ldr d5,[%[sa],#-16]\n\t"
|
||||
FMLA_IR "%[c2i].4s,v0.4s,v2.s[3]; ldr x0,[%[sa],#-8]\n\t"
|
||||
FMLA_RR "%[c3r].4s,v0.4s,v3.s[0]; mov w5,w1\n\t"
|
||||
FMLA_IR "%[c3i].4s,v0.4s,v3.s[1]\n\t"
|
||||
"fmov v5.d[1],x0; fmov d1,x2\n\t"
|
||||
FMLA_RR "%[c4r].4s,v0.4s,v3.s[2]; ldr x2,[%[sb],#-24]\n\t"
|
||||
FMLA_IR "%[c4i].4s,v0.4s,v3.s[3]; ldr x3,[%[sb],#-16]\n\t"
|
||||
FMLA_RR "%[c5r].4s,v0.4s,v4.s[0]\n\t"
|
||||
"fmov v1.d[1],x4; ldr d6,[%[sa]]\n\t"
|
||||
FMLA_IR "%[c5i].4s,v0.4s,v4.s[1]; ldr x0,[%[sa],#8]\n\t"
|
||||
FMLA_RR "%[c6r].4s,v0.4s,v4.s[2]; ldr x4,[%[sb],#-8]\n\t"
|
||||
FMLA_IR "%[c6i].4s,v0.4s,v4.s[3]; bfi x5,x2,#32,#32\n\t"
|
||||
"fmov v6.d[1],x0; ldr d7,[%[sa],#16]\n\t"
|
||||
FMLA_RR "%[c7r].4s,v0.4s,v5.s[0]; ldr x0,[%[sa],#24]\n\t"
|
||||
FMLA_IR "%[c7i].4s,v0.4s,v5.s[1]; mov w6,w3\n\t"
|
||||
FMLA_RR "%[c8r].4s,v0.4s,v5.s[2]; bfxil x2,x1,#32,#32\n\t"
|
||||
"fmov v7.d[1],x0; fmov d10,x5\n\t"
|
||||
FMLA_IR "%[c8i].4s,v0.4s,v5.s[3]; bfi x6,x4,#32,#32\n\t"
|
||||
FMLA_II "%[c1r].4s,v1.4s,v2.s[1]\n\t"
|
||||
FMLA_RI "%[c1i].4s,v1.4s,v2.s[0]; bfxil x4,x3,#32,#32\n\t"
|
||||
"fmov v10.d[1],x6; fmov d11,x2\n\t"
|
||||
FMLA_II "%[c2r].4s,v1.4s,v2.s[3]\n\t"
|
||||
FMLA_RI "%[c2i].4s,v1.4s,v2.s[2]\n\t"
|
||||
FMLA_II "%[c3r].4s,v1.4s,v3.s[1]\n\t"
|
||||
"fmov v11.d[1],x4; ldr d8,[%[sa],#32]\n\t"
|
||||
FMLA_RI "%[c3i].4s,v1.4s,v3.s[0]; ldr x0,[%[sa],#40]\n\t"
|
||||
FMLA_II "%[c4r].4s,v1.4s,v3.s[3]; sub %[K],%[K],#2\n\t"
|
||||
FMLA_RI "%[c4i].4s,v1.4s,v3.s[2]\n\t"
|
||||
"fmov v8.d[1],x0; ldr d9,[%[sa],#48]\n\t"
|
||||
FMLA_II "%[c5r].4s,v1.4s,v4.s[1]; ldr x0,[%[sa],#56]\n\t"
|
||||
FMLA_RI "%[c5i].4s,v1.4s,v4.s[0]; add %[sa],%[sa],#64\n\t"
|
||||
FMLA_II "%[c6r].4s,v1.4s,v4.s[3]\n\t"
|
||||
"fmov v9.d[1],x0\n\t"
|
||||
FMLA_RI "%[c6i].4s,v1.4s,v4.s[2]\n\t"
|
||||
FMLA_II "%[c7r].4s,v1.4s,v5.s[1]\n\t" FMLA_RI "%[c7i].4s,v1.4s,v5.s[0]\n\t"
|
||||
FMLA_II "%[c8r].4s,v1.4s,v5.s[3]\n\t" FMLA_RI "%[c8i].4s,v1.4s,v5.s[2]\n\t"
|
||||
FMLA_RR "%[c1r].4s,v10.4s,v6.s[0]\n\t" FMLA_IR "%[c1i].4s,v10.4s,v6.s[1]\n\t"
|
||||
FMLA_RR "%[c2r].4s,v10.4s,v6.s[2]\n\t" FMLA_IR "%[c2i].4s,v10.4s,v6.s[3]\n\t"
|
||||
FMLA_RR "%[c3r].4s,v10.4s,v7.s[0]\n\t" FMLA_IR "%[c3i].4s,v10.4s,v7.s[1]\n\t"
|
||||
FMLA_RR "%[c4r].4s,v10.4s,v7.s[2]\n\t" FMLA_IR "%[c4i].4s,v10.4s,v7.s[3]\n\t"
|
||||
FMLA_RR "%[c5r].4s,v10.4s,v8.s[0]\n\t" FMLA_IR "%[c5i].4s,v10.4s,v8.s[1]\n\t"
|
||||
FMLA_RR "%[c6r].4s,v10.4s,v8.s[2]\n\t" FMLA_IR "%[c6i].4s,v10.4s,v8.s[3]\n\t"
|
||||
FMLA_RR "%[c7r].4s,v10.4s,v9.s[0]\n\t" FMLA_IR "%[c7i].4s,v10.4s,v9.s[1]\n\t"
|
||||
FMLA_RR "%[c8r].4s,v10.4s,v9.s[2]\n\t" FMLA_IR "%[c8i].4s,v10.4s,v9.s[3]\n\t"
|
||||
FMLA_II "%[c1r].4s,v11.4s,v6.s[1]\n\t" FMLA_RI "%[c1i].4s,v11.4s,v6.s[0]\n\t"
|
||||
FMLA_II "%[c2r].4s,v11.4s,v6.s[3]\n\t" FMLA_RI "%[c2i].4s,v11.4s,v6.s[2]\n\t"
|
||||
FMLA_II "%[c3r].4s,v11.4s,v7.s[1]\n\t" FMLA_RI "%[c3i].4s,v11.4s,v7.s[0]\n\t"
|
||||
FMLA_II "%[c4r].4s,v11.4s,v7.s[3]\n\t" FMLA_RI "%[c4i].4s,v11.4s,v7.s[2]\n\t"
|
||||
FMLA_II "%[c5r].4s,v11.4s,v8.s[1]\n\t" FMLA_RI "%[c5i].4s,v11.4s,v8.s[0]\n\t"
|
||||
FMLA_II "%[c6r].4s,v11.4s,v8.s[3]\n\t" FMLA_RI "%[c6i].4s,v11.4s,v8.s[2]\n\t"
|
||||
FMLA_II "%[c7r].4s,v11.4s,v9.s[1]\n\t" FMLA_RI "%[c7i].4s,v11.4s,v9.s[0]\n\t"
|
||||
FMLA_II "%[c8r].4s,v11.4s,v9.s[3]\n\t" FMLA_RI "%[c8i].4s,v11.4s,v9.s[2]\n\t"
|
||||
"b 4f\n\t"
|
||||
"3:\n\t"
|
||||
"fmov v3.d[1],x0; ldr d4,[%[sa],#-32]\n\t"
|
||||
FMLA_RR "%[c1r].4s,v0.4s,v2.s[0]; ldr x0,[%[sa],#-24]\n\t"
|
||||
FMLA_IR "%[c1i].4s,v0.4s,v2.s[1]\n\t"
|
||||
FMLA_RR "%[c2r].4s,v0.4s,v2.s[2]\n\t"
|
||||
"fmov v4.d[1],x0; ldr d5,[%[sa],#-16]\n\t"
|
||||
FMLA_IR "%[c2i].4s,v0.4s,v2.s[3]; ldr x0,[%[sa],#-8]\n\t"
|
||||
FMLA_RR "%[c3r].4s,v0.4s,v3.s[0]\n\t"
|
||||
FMLA_IR "%[c3i].4s,v0.4s,v3.s[1]\n\t"
|
||||
"fmov v5.d[1],x0; fmov d1,x2\n\t"
|
||||
FMLA_RR "%[c4r].4s,v0.4s,v3.s[2]\n\t"
|
||||
FMLA_IR "%[c4i].4s,v0.4s,v3.s[3]\n\t"
|
||||
FMLA_RR "%[c5r].4s,v0.4s,v4.s[0]\n\t"
|
||||
"fmov v1.d[1],x4\n\t"
|
||||
FMLA_IR "%[c5i].4s,v0.4s,v4.s[1]; sub %[K],%[K],#1\n\t"
|
||||
FMLA_RR "%[c6r].4s,v0.4s,v4.s[2]\n\t" FMLA_IR "%[c6i].4s,v0.4s,v4.s[3]\n\t"
|
||||
FMLA_RR "%[c7r].4s,v0.4s,v5.s[0]\n\t" FMLA_IR "%[c7i].4s,v0.4s,v5.s[1]\n\t"
|
||||
FMLA_RR "%[c8r].4s,v0.4s,v5.s[2]\n\t" FMLA_IR "%[c8i].4s,v0.4s,v5.s[3]\n\t"
|
||||
FMLA_II "%[c1r].4s,v1.4s,v2.s[1]\n\t" FMLA_RI "%[c1i].4s,v1.4s,v2.s[0]\n\t"
|
||||
FMLA_II "%[c2r].4s,v1.4s,v2.s[3]\n\t" FMLA_RI "%[c2i].4s,v1.4s,v2.s[2]\n\t"
|
||||
FMLA_II "%[c3r].4s,v1.4s,v3.s[1]\n\t" FMLA_RI "%[c3i].4s,v1.4s,v3.s[0]\n\t"
|
||||
FMLA_II "%[c4r].4s,v1.4s,v3.s[3]\n\t" FMLA_RI "%[c4i].4s,v1.4s,v3.s[2]\n\t"
|
||||
FMLA_II "%[c5r].4s,v1.4s,v4.s[1]\n\t" FMLA_RI "%[c5i].4s,v1.4s,v4.s[0]\n\t"
|
||||
FMLA_II "%[c6r].4s,v1.4s,v4.s[3]\n\t" FMLA_RI "%[c6i].4s,v1.4s,v4.s[2]\n\t"
|
||||
FMLA_II "%[c7r].4s,v1.4s,v5.s[1]\n\t" FMLA_RI "%[c7i].4s,v1.4s,v5.s[0]\n\t"
|
||||
FMLA_II "%[c8r].4s,v1.4s,v5.s[3]\n\t" FMLA_RI "%[c8i].4s,v1.4s,v5.s[2]\n\t"
|
||||
"4:\n\t"
|
||||
"mov %[c_pref],%[C]\n\t"
|
||||
"zip1 v0.4s,%[c1r].4s,%[c2r].4s; prfm pstl1keep,[%[c_pref]]\n\t"
|
||||
"zip1 v4.4s,%[c1i].4s,%[c2i].4s; prfm pstl1keep,[%[c_pref],#64]\n\t"
|
||||
"zip1 v1.4s,%[c3r].4s,%[c4r].4s; add %[c_pref],%[c_pref],%[LDC],LSL#3\n\t"
|
||||
"zip1 v5.4s,%[c3i].4s,%[c4i].4s; prfm pstl1keep,[%[c_pref]]\n\t"
|
||||
"zip2 v2.4s,%[c1r].4s,%[c2r].4s; prfm pstl1keep,[%[c_pref],#64]\n\t"
|
||||
"zip2 v6.4s,%[c1i].4s,%[c2i].4s; add %[c_pref],%[c_pref],%[LDC],LSL#3\n\t"
|
||||
"zip2 v3.4s,%[c3r].4s,%[c4r].4s; prfm pstl1keep,[%[c_pref]]\n\t"
|
||||
"zip2 v7.4s,%[c3i].4s,%[c4i].4s; prfm pstl1keep,[%[c_pref],#64]\n\t"
|
||||
"zip1 %[c1r].2d,v0.2d,v1.2d; add %[c_pref],%[c_pref],%[LDC],LSL#3\n\t"
|
||||
"zip1 %[c1i].2d,v4.2d,v5.2d; prfm pstl1keep,[%[c_pref]]\n\t"
|
||||
"zip2 %[c2r].2d,v0.2d,v1.2d; prfm pstl1keep,[%[c_pref],#64]\n\t"
|
||||
"zip2 %[c2i].2d,v4.2d,v5.2d\n\t"
|
||||
"zip1 %[c3r].2d,v2.2d,v3.2d; zip1 %[c3i].2d,v6.2d,v7.2d\n\t"
|
||||
"zip2 %[c4r].2d,v2.2d,v3.2d; zip2 %[c4i].2d,v6.2d,v7.2d\n\t"
|
||||
"zip1 v0.4s,%[c5r].4s,%[c6r].4s; zip1 v4.4s,%[c5i].4s,%[c6i].4s\n\t"
|
||||
"zip1 v1.4s,%[c7r].4s,%[c8r].4s; zip1 v5.4s,%[c7i].4s,%[c8i].4s\n\t"
|
||||
"zip2 v2.4s,%[c5r].4s,%[c6r].4s; zip2 v6.4s,%[c5i].4s,%[c6i].4s\n\t"
|
||||
"zip2 v3.4s,%[c7r].4s,%[c8r].4s; zip2 v7.4s,%[c7i].4s,%[c8i].4s\n\t"
|
||||
"zip1 %[c5r].2d,v0.2d,v1.2d; zip1 %[c5i].2d,v4.2d,v5.2d\n\t"
|
||||
"zip2 %[c6r].2d,v0.2d,v1.2d; zip2 %[c6i].2d,v4.2d,v5.2d\n\t"
|
||||
"zip1 %[c7r].2d,v2.2d,v3.2d; zip1 %[c7i].2d,v6.2d,v7.2d\n\t"
|
||||
"zip2 %[c8r].2d,v2.2d,v3.2d; zip2 %[c8i].2d,v6.2d,v7.2d\n\t"
|
||||
:[c1r]"=w"(c1r), [c1i]"=w"(c1i), [c2r]"=w"(c2r), [c2i]"=w"(c2i),
|
||||
[c3r]"=w"(c3r), [c3i]"=w"(c3i), [c4r]"=w"(c4r), [c4i]"=w"(c4i),
|
||||
[c5r]"=w"(c5r), [c5i]"=w"(c5i), [c6r]"=w"(c6r), [c6i]"=w"(c6i),
|
||||
[c7r]"=w"(c7r), [c7i]"=w"(c7i), [c8r]"=w"(c8r), [c8i]"=w"(c8i),
|
||||
[K]"+r"(K), [sa]"+r"(sa), [sb]"+r"(sb), [c_pref]"+r"(c_pref)
|
||||
:[C]"r"(C), [LDC]"r"(LDC)
|
||||
:"cc","memory","x0","x1","x2","x3","x4","x5","x6",
|
||||
"v0","v1","v2","v3","v4","v5","v6","v7","v8","v9","v10","v11");
|
||||
|
||||
store_m8n1_contracted(C, c1r, c1i, c5r, c5i, alphar, alphai); C += LDC * 2;
|
||||
store_m8n1_contracted(C, c2r, c2i, c6r, c6i, alphar, alphai); C += LDC * 2;
|
||||
store_m8n1_contracted(C, c3r, c3i, c7r, c7i, alphar, alphai); C += LDC * 2;
|
||||
store_m8n1_contracted(C, c4r, c4i, c8r, c8i, alphar, alphai);
|
||||
}
|
||||
|
||||
static inline float32x4x4_t acc_expanded_m2n2(float32x4x4_t acc,
|
||||
float32x4_t a, float32x4_t b) {
|
||||
|
||||
acc.val[0] = vfmaq_laneq_f32(acc.val[0], a, b, 0);
|
||||
acc.val[1] = vfmaq_laneq_f32(acc.val[1], a, b, 1);
|
||||
acc.val[2] = vfmaq_laneq_f32(acc.val[2], a, b, 2);
|
||||
acc.val[3] = vfmaq_laneq_f32(acc.val[3], a, b, 3);
|
||||
return acc;
|
||||
}
|
||||
|
||||
static inline float32x4x4_t expand_alpha(float alphar, float alphai) {
|
||||
float32x4x4_t ret;
|
||||
const float maskp[] = { -1, 1, -1, 1 };
|
||||
const float maskn[] = { 1, -1, 1, -1 };
|
||||
const float32x4_t vrevp = vld1q_f32(maskp);
|
||||
const float32x4_t vrevn = vld1q_f32(maskn);
|
||||
#if defined(NN) || defined(NT) || defined(TN) || defined(TT)
|
||||
ret.val[0] = vdupq_n_f32(alphar);
|
||||
ret.val[1] = vdupq_n_f32(-alphai);
|
||||
ret.val[2] = vmulq_f32(ret.val[1], vrevn);
|
||||
ret.val[3] = vmulq_f32(ret.val[0], vrevp);
|
||||
#elif defined(NR) || defined(NC) || defined(TR) || defined(TC)
|
||||
ret.val[0] = vdupq_n_f32(alphar);
|
||||
ret.val[1] = vdupq_n_f32(alphai);
|
||||
ret.val[2] = vmulq_f32(ret.val[1], vrevp);
|
||||
ret.val[3] = vmulq_f32(ret.val[0], vrevn);
|
||||
#elif defined(RN) || defined(RT) || defined(CN) || defined(CT)
|
||||
ret.val[2] = vdupq_n_f32(alphai);
|
||||
ret.val[3] = vdupq_n_f32(alphar);
|
||||
ret.val[0] = vmulq_f32(ret.val[3], vrevn);
|
||||
ret.val[1] = vmulq_f32(ret.val[2], vrevp);
|
||||
#else
|
||||
ret.val[2] = vdupq_n_f32(alphai);
|
||||
ret.val[3] = vdupq_n_f32(-alphar);
|
||||
ret.val[0] = vmulq_f32(ret.val[3], vrevp);
|
||||
ret.val[1] = vmulq_f32(ret.val[2], vrevn);
|
||||
#endif
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline void store_expanded_m2n2(float *C, BLASLONG LDC,
|
||||
float32x4x4_t acc, float32x4x4_t expanded_alpha) {
|
||||
|
||||
float32x4_t ld1 = vld1q_f32(C), ld2 = vld1q_f32(C + LDC * 2);
|
||||
ld1 = vfmaq_f32(ld1, acc.val[0], expanded_alpha.val[0]);
|
||||
ld2 = vfmaq_f32(ld2, acc.val[2], expanded_alpha.val[0]);
|
||||
acc.val[0] = vrev64q_f32(acc.val[0]);
|
||||
acc.val[2] = vrev64q_f32(acc.val[2]);
|
||||
ld1 = vfmaq_f32(ld1, acc.val[1], expanded_alpha.val[1]);
|
||||
ld2 = vfmaq_f32(ld2, acc.val[3], expanded_alpha.val[1]);
|
||||
acc.val[1] = vrev64q_f32(acc.val[1]);
|
||||
acc.val[3] = vrev64q_f32(acc.val[3]);
|
||||
ld1 = vfmaq_f32(ld1, acc.val[0], expanded_alpha.val[2]);
|
||||
ld2 = vfmaq_f32(ld2, acc.val[2], expanded_alpha.val[2]);
|
||||
ld1 = vfmaq_f32(ld1, acc.val[1], expanded_alpha.val[3]);
|
||||
ld2 = vfmaq_f32(ld2, acc.val[3], expanded_alpha.val[3]);
|
||||
vst1q_f32(C, ld1);
|
||||
vst1q_f32(C + LDC * 2, ld2);
|
||||
}
|
||||
|
||||
static inline float32x4x4_t init_expanded_m2n2() {
|
||||
float32x4x4_t ret = {{ vdupq_n_f32(0), vdupq_n_f32(0),
|
||||
vdupq_n_f32(0), vdupq_n_f32(0) }};
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline void kernel_4x4(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K, BLASLONG LDC) {
|
||||
|
||||
float32x4x4_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init_expanded_m2n2();
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4),
|
||||
a3 = vld1q_f32(sa + 8), a4 = vld1q_f32(sa + 12); sa += 16;
|
||||
float32x4_t b1 = vld1q_f32(sb), b2 = vld1q_f32(sb + 4),
|
||||
b3 = vld1q_f32(sb + 8), b4 = vld1q_f32(sb + 12); sb += 16;
|
||||
c1 = acc_expanded_m2n2(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n2(c2, a2, b1);
|
||||
c3 = acc_expanded_m2n2(c3, a1, b2);
|
||||
c4 = acc_expanded_m2n2(c4, a2, b2);
|
||||
c1 = acc_expanded_m2n2(c1, a3, b3);
|
||||
c2 = acc_expanded_m2n2(c2, a4, b3);
|
||||
c3 = acc_expanded_m2n2(c3, a3, b4);
|
||||
c4 = acc_expanded_m2n2(c4, a4, b4);
|
||||
}
|
||||
if (K) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4);
|
||||
float32x4_t b1 = vld1q_f32(sb), b2 = vld1q_f32(sb + 4);
|
||||
c1 = acc_expanded_m2n2(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n2(c2, a2, b1);
|
||||
c3 = acc_expanded_m2n2(c3, a1, b2);
|
||||
c4 = acc_expanded_m2n2(c4, a2, b2);
|
||||
}
|
||||
|
||||
float32x4x4_t e_alpha = expand_alpha(alphar, alphai);
|
||||
store_expanded_m2n2(C, LDC, c1, e_alpha);
|
||||
store_expanded_m2n2(C + 4, LDC, c2, e_alpha);
|
||||
C += LDC * 4;
|
||||
store_expanded_m2n2(C, LDC, c3, e_alpha);
|
||||
store_expanded_m2n2(C + 4, LDC, c4, e_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_8x2(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K, BLASLONG LDC) {
|
||||
|
||||
float32x4x4_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init_expanded_m2n2();
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4);
|
||||
float32x4_t a3 = vld1q_f32(sa + 8), a4 = vld1q_f32(sa + 12);
|
||||
float32x4_t a5 = vld1q_f32(sa + 16), a6 = vld1q_f32(sa + 20);
|
||||
float32x4_t a7 = vld1q_f32(sa + 24), a8 = vld1q_f32(sa + 28); sa += 32;
|
||||
float32x4_t b1 = vld1q_f32(sb), b2 = vld1q_f32(sb + 4); sb += 8;
|
||||
c1 = acc_expanded_m2n2(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n2(c2, a2, b1);
|
||||
c3 = acc_expanded_m2n2(c3, a3, b1);
|
||||
c4 = acc_expanded_m2n2(c4, a4, b1);
|
||||
c1 = acc_expanded_m2n2(c1, a5, b2);
|
||||
c2 = acc_expanded_m2n2(c2, a6, b2);
|
||||
c3 = acc_expanded_m2n2(c3, a7, b2);
|
||||
c4 = acc_expanded_m2n2(c4, a8, b2);
|
||||
}
|
||||
if (K) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4);
|
||||
float32x4_t a3 = vld1q_f32(sa + 8), a4 = vld1q_f32(sa + 12);
|
||||
float32x4_t b1 = vld1q_f32(sb);
|
||||
c1 = acc_expanded_m2n2(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n2(c2, a2, b1);
|
||||
c3 = acc_expanded_m2n2(c3, a3, b1);
|
||||
c4 = acc_expanded_m2n2(c4, a4, b1);
|
||||
}
|
||||
|
||||
float32x4x4_t e_alpha = expand_alpha(alphar, alphai);
|
||||
store_expanded_m2n2(C, LDC, c1, e_alpha);
|
||||
store_expanded_m2n2(C + 4, LDC, c2, e_alpha);
|
||||
store_expanded_m2n2(C + 8, LDC, c3, e_alpha);
|
||||
store_expanded_m2n2(C + 12, LDC, c4, e_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_4x2(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K, BLASLONG LDC) {
|
||||
|
||||
float32x4x4_t c1, c2;
|
||||
c1 = c2 = init_expanded_m2n2();
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4);
|
||||
float32x4_t a3 = vld1q_f32(sa + 8), a4 = vld1q_f32(sa + 12); sa += 16;
|
||||
float32x4_t b1 = vld1q_f32(sb), b2 = vld1q_f32(sb + 4); sb += 8;
|
||||
c1 = acc_expanded_m2n2(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n2(c2, a2, b1);
|
||||
c1 = acc_expanded_m2n2(c1, a3, b2);
|
||||
c2 = acc_expanded_m2n2(c2, a4, b2);
|
||||
}
|
||||
if (K) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4);
|
||||
float32x4_t b1 = vld1q_f32(sb);
|
||||
c1 = acc_expanded_m2n2(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n2(c2, a2, b1);
|
||||
}
|
||||
|
||||
float32x4x4_t e_alpha = expand_alpha(alphar, alphai);
|
||||
store_expanded_m2n2(C, LDC, c1, e_alpha);
|
||||
store_expanded_m2n2(C + 4, LDC, c2, e_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_2x4(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K, BLASLONG LDC) {
|
||||
|
||||
float32x4x4_t c1, c2;
|
||||
c1 = c2 = init_expanded_m2n2();
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4); sa += 8;
|
||||
float32x4_t b1 = vld1q_f32(sb), b2 = vld1q_f32(sb + 4);
|
||||
float32x4_t b3 = vld1q_f32(sb + 8), b4 = vld1q_f32(sb + 12); sb += 16;
|
||||
c1 = acc_expanded_m2n2(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n2(c2, a1, b2);
|
||||
c1 = acc_expanded_m2n2(c1, a2, b3);
|
||||
c2 = acc_expanded_m2n2(c2, a2, b4);
|
||||
}
|
||||
if (K) {
|
||||
float32x4_t a1 = vld1q_f32(sa);
|
||||
float32x4_t b1 = vld1q_f32(sb), b2 = vld1q_f32(sb + 4);
|
||||
c1 = acc_expanded_m2n2(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n2(c2, a1, b2);
|
||||
}
|
||||
|
||||
float32x4x4_t e_alpha = expand_alpha(alphar, alphai);
|
||||
store_expanded_m2n2(C, LDC, c1, e_alpha);
|
||||
store_expanded_m2n2(C + LDC * 4, LDC, c2, e_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_2x2(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K, BLASLONG LDC) {
|
||||
|
||||
float32x4x4_t c1, c2;
|
||||
c1 = c2 = init_expanded_m2n2();
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4); sa += 8;
|
||||
float32x4_t b1 = vld1q_f32(sb), b2 = vld1q_f32(sb + 4); sb += 8;
|
||||
c1 = acc_expanded_m2n2(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n2(c2, a2, b2);
|
||||
}
|
||||
c1.val[0] = vaddq_f32(c1.val[0], c2.val[0]);
|
||||
c1.val[1] = vaddq_f32(c1.val[1], c2.val[1]);
|
||||
c1.val[2] = vaddq_f32(c1.val[2], c2.val[2]);
|
||||
c1.val[3] = vaddq_f32(c1.val[3], c2.val[3]);
|
||||
if (K) {
|
||||
float32x4_t a1 = vld1q_f32(sa);
|
||||
float32x4_t b1 = vld1q_f32(sb);
|
||||
c1 = acc_expanded_m2n2(c1, a1, b1);
|
||||
}
|
||||
|
||||
store_expanded_m2n2(C, LDC, c1, expand_alpha(alphar, alphai));
|
||||
}
|
||||
|
||||
static inline float32x4x2_t acc_expanded_m2n1(float32x4x2_t acc,
|
||||
float32x4_t a, float32x2_t b) {
|
||||
|
||||
acc.val[0] = vfmaq_lane_f32(acc.val[0], a, b, 0);
|
||||
acc.val[1] = vfmaq_lane_f32(acc.val[1], a, b, 1);
|
||||
return acc;
|
||||
}
|
||||
|
||||
static inline void store_expanded_m2n1(float *C,
|
||||
float32x4x2_t acc, float32x4x4_t expanded_alpha) {
|
||||
|
||||
float32x4_t ld1 = vld1q_f32(C);
|
||||
ld1 = vfmaq_f32(ld1, acc.val[0], expanded_alpha.val[0]);
|
||||
acc.val[0] = vrev64q_f32(acc.val[0]);
|
||||
ld1 = vfmaq_f32(ld1, acc.val[1], expanded_alpha.val[1]);
|
||||
acc.val[1] = vrev64q_f32(acc.val[1]);
|
||||
ld1 = vfmaq_f32(ld1, acc.val[0], expanded_alpha.val[2]);
|
||||
ld1 = vfmaq_f32(ld1, acc.val[1], expanded_alpha.val[3]);
|
||||
vst1q_f32(C, ld1);
|
||||
}
|
||||
|
||||
static inline float32x4x2_t init_expanded_m2n1() {
|
||||
float32x4x2_t ret = {{ vdupq_n_f32(0), vdupq_n_f32(0) }};
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline void kernel_8x1(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K) {
|
||||
|
||||
float32x4x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init_expanded_m2n1();
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4),
|
||||
a3 = vld1q_f32(sa + 8), a4 = vld1q_f32(sa + 12),
|
||||
a5 = vld1q_f32(sa + 16), a6 = vld1q_f32(sa + 20),
|
||||
a7 = vld1q_f32(sa + 24), a8 = vld1q_f32(sa + 28); sa += 32;
|
||||
float32x2_t b1 = vld1_f32(sb), b2 = vld1_f32(sb + 2); sb += 4;
|
||||
c1 = acc_expanded_m2n1(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n1(c2, a2, b1);
|
||||
c3 = acc_expanded_m2n1(c3, a3, b1);
|
||||
c4 = acc_expanded_m2n1(c4, a4, b1);
|
||||
c1 = acc_expanded_m2n1(c1, a5, b2);
|
||||
c2 = acc_expanded_m2n1(c2, a6, b2);
|
||||
c3 = acc_expanded_m2n1(c3, a7, b2);
|
||||
c4 = acc_expanded_m2n1(c4, a8, b2);
|
||||
}
|
||||
if (K) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4),
|
||||
a3 = vld1q_f32(sa + 8), a4 = vld1q_f32(sa + 12);
|
||||
float32x2_t b1 = vld1_f32(sb);
|
||||
c1 = acc_expanded_m2n1(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n1(c2, a2, b1);
|
||||
c3 = acc_expanded_m2n1(c3, a3, b1);
|
||||
c4 = acc_expanded_m2n1(c4, a4, b1);
|
||||
}
|
||||
|
||||
float32x4x4_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
store_expanded_m2n1(C, c1, expanded_alpha);
|
||||
store_expanded_m2n1(C + 4, c2, expanded_alpha);
|
||||
store_expanded_m2n1(C + 8, c3, expanded_alpha);
|
||||
store_expanded_m2n1(C + 12, c4, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_4x1(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K) {
|
||||
|
||||
float32x4x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init_expanded_m2n1();
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4),
|
||||
a3 = vld1q_f32(sa + 8), a4 = vld1q_f32(sa + 12); sa += 16;
|
||||
float32x2_t b1 = vld1_f32(sb), b2 = vld1_f32(sb + 2); sb += 4;
|
||||
c1 = acc_expanded_m2n1(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n1(c2, a2, b1);
|
||||
c3 = acc_expanded_m2n1(c3, a3, b2);
|
||||
c4 = acc_expanded_m2n1(c4, a4, b2);
|
||||
}
|
||||
c1.val[0] = vaddq_f32(c1.val[0], c3.val[0]);
|
||||
c1.val[1] = vaddq_f32(c1.val[1], c3.val[1]);
|
||||
c2.val[0] = vaddq_f32(c2.val[0], c4.val[0]);
|
||||
c2.val[1] = vaddq_f32(c2.val[1], c4.val[1]);
|
||||
if (K) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4);
|
||||
float32x2_t b1 = vld1_f32(sb);
|
||||
c1 = acc_expanded_m2n1(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n1(c2, a2, b1);
|
||||
}
|
||||
|
||||
float32x4x4_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
store_expanded_m2n1(C, c1, expanded_alpha);
|
||||
store_expanded_m2n1(C + 4, c2, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_2x1(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K) {
|
||||
|
||||
float32x4x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init_expanded_m2n1();
|
||||
|
||||
for (; K > 3; K -= 4) {
|
||||
float32x4_t a1 = vld1q_f32(sa), a2 = vld1q_f32(sa + 4),
|
||||
a3 = vld1q_f32(sa + 8), a4 = vld1q_f32(sa + 12); sa += 16;
|
||||
float32x2_t b1 = vld1_f32(sb), b2 = vld1_f32(sb + 2),
|
||||
b3 = vld1_f32(sb + 4), b4 = vld1_f32(sb + 6); sb += 8;
|
||||
c1 = acc_expanded_m2n1(c1, a1, b1);
|
||||
c2 = acc_expanded_m2n1(c2, a2, b2);
|
||||
c3 = acc_expanded_m2n1(c3, a3, b3);
|
||||
c4 = acc_expanded_m2n1(c4, a4, b4);
|
||||
}
|
||||
c1.val[0] = vaddq_f32(c1.val[0], c3.val[0]);
|
||||
c1.val[1] = vaddq_f32(c1.val[1], c3.val[1]);
|
||||
c2.val[0] = vaddq_f32(c2.val[0], c4.val[0]);
|
||||
c2.val[1] = vaddq_f32(c2.val[1], c4.val[1]);
|
||||
c1.val[0] = vaddq_f32(c1.val[0], c2.val[0]);
|
||||
c1.val[1] = vaddq_f32(c1.val[1], c2.val[1]);
|
||||
for (; K; K--) {
|
||||
float32x4_t a1 = vld1q_f32(sa); sa += 4;
|
||||
float32x2_t b1 = vld1_f32(sb); sb += 2;
|
||||
c1 = acc_expanded_m2n1(c1, a1, b1);
|
||||
}
|
||||
|
||||
float32x4x4_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
store_expanded_m2n1(C, c1, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline float32x2x4_t expand_alpha_d(float alphar, float alphai) {
|
||||
float32x2x4_t ret;
|
||||
const float maskp[] = { -1, 1 };
|
||||
const float maskn[] = { 1, -1 };
|
||||
const float32x2_t vrevp = vld1_f32(maskp);
|
||||
const float32x2_t vrevn = vld1_f32(maskn);
|
||||
#if defined(NN) || defined(NT) || defined(TN) || defined(TT)
|
||||
ret.val[0] = vdup_n_f32(alphar);
|
||||
ret.val[1] = vdup_n_f32(-alphai);
|
||||
ret.val[2] = vmul_f32(ret.val[1], vrevn);
|
||||
ret.val[3] = vmul_f32(ret.val[0], vrevp);
|
||||
#elif defined(NR) || defined(NC) || defined(TR) || defined(TC)
|
||||
ret.val[0] = vdup_n_f32(alphar);
|
||||
ret.val[1] = vdup_n_f32(alphai);
|
||||
ret.val[2] = vmul_f32(ret.val[1], vrevp);
|
||||
ret.val[3] = vmul_f32(ret.val[0], vrevn);
|
||||
#elif defined(RN) || defined(RT) || defined(CN) || defined(CT)
|
||||
ret.val[2] = vdup_n_f32(alphai);
|
||||
ret.val[3] = vdup_n_f32(alphar);
|
||||
ret.val[0] = vmul_f32(ret.val[3], vrevn);
|
||||
ret.val[1] = vmul_f32(ret.val[2], vrevp);
|
||||
#else
|
||||
ret.val[2] = vdup_n_f32(alphai);
|
||||
ret.val[3] = vdup_n_f32(-alphar);
|
||||
ret.val[0] = vmul_f32(ret.val[3], vrevp);
|
||||
ret.val[1] = vmul_f32(ret.val[2], vrevn);
|
||||
#endif
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline float32x2x2_t acc_expanded_m1n1(float32x2x2_t acc,
|
||||
float32x2_t a, float32x2_t b) {
|
||||
|
||||
acc.val[0] = vfma_lane_f32(acc.val[0], a, b, 0);
|
||||
acc.val[1] = vfma_lane_f32(acc.val[1], a, b, 1);
|
||||
return acc;
|
||||
}
|
||||
|
||||
static inline void store_expanded_m1n1(float *C,
|
||||
float32x2x2_t acc, float32x2x4_t expanded_alpha) {
|
||||
|
||||
float32x2_t ld1 = vld1_f32(C);
|
||||
ld1 = vfma_f32(ld1, acc.val[0], expanded_alpha.val[0]);
|
||||
acc.val[0] = vrev64_f32(acc.val[0]);
|
||||
ld1 = vfma_f32(ld1, acc.val[1], expanded_alpha.val[1]);
|
||||
acc.val[1] = vrev64_f32(acc.val[1]);
|
||||
ld1 = vfma_f32(ld1, acc.val[0], expanded_alpha.val[2]);
|
||||
ld1 = vfma_f32(ld1, acc.val[1], expanded_alpha.val[3]);
|
||||
vst1_f32(C, ld1);
|
||||
}
|
||||
|
||||
static inline float32x2x2_t init_expanded_m1n1() {
|
||||
float32x2x2_t ret = {{ vdup_n_f32(0), vdup_n_f32(0) }};
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline void kernel_1x4(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K, BLASLONG LDC) {
|
||||
|
||||
float32x2x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init_expanded_m1n1();
|
||||
|
||||
for (; K; K--) {
|
||||
float32x2_t a1 = vld1_f32(sa); sa += 2;
|
||||
c1 = acc_expanded_m1n1(c1, a1, vld1_f32(sb));
|
||||
c2 = acc_expanded_m1n1(c2, a1, vld1_f32(sb + 2));
|
||||
c3 = acc_expanded_m1n1(c3, a1, vld1_f32(sb + 4));
|
||||
c4 = acc_expanded_m1n1(c4, a1, vld1_f32(sb + 6));
|
||||
sb += 8;
|
||||
}
|
||||
|
||||
float32x2x4_t expanded_alpha = expand_alpha_d(alphar, alphai);
|
||||
store_expanded_m1n1(C, c1, expanded_alpha); C += LDC * 2;
|
||||
store_expanded_m1n1(C, c2, expanded_alpha); C += LDC * 2;
|
||||
store_expanded_m1n1(C, c3, expanded_alpha); C += LDC * 2;
|
||||
store_expanded_m1n1(C, c4, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_1x2(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K, BLASLONG LDC) {
|
||||
|
||||
float32x2x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init_expanded_m1n1();
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float32x2_t a1 = vld1_f32(sa), a2 = vld1_f32(sa + 2); sa += 4;
|
||||
c1 = acc_expanded_m1n1(c1, a1, vld1_f32(sb));
|
||||
c2 = acc_expanded_m1n1(c2, a1, vld1_f32(sb + 2));
|
||||
c3 = acc_expanded_m1n1(c3, a2, vld1_f32(sb + 4));
|
||||
c4 = acc_expanded_m1n1(c4, a2, vld1_f32(sb + 6));
|
||||
sb += 8;
|
||||
}
|
||||
c1.val[0] = vadd_f32(c1.val[0], c3.val[0]);
|
||||
c1.val[1] = vadd_f32(c1.val[1], c3.val[1]);
|
||||
c2.val[0] = vadd_f32(c2.val[0], c4.val[0]);
|
||||
c2.val[1] = vadd_f32(c2.val[1], c4.val[1]);
|
||||
if (K) {
|
||||
float32x2_t a1 = vld1_f32(sa);
|
||||
c1 = acc_expanded_m1n1(c1, a1, vld1_f32(sb));
|
||||
c2 = acc_expanded_m1n1(c2, a1, vld1_f32(sb + 2));
|
||||
}
|
||||
|
||||
float32x2x4_t expanded_alpha = expand_alpha_d(alphar, alphai);
|
||||
store_expanded_m1n1(C, c1, expanded_alpha); C += LDC * 2;
|
||||
store_expanded_m1n1(C, c2, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_1x1(const float *sa, const float *sb, float *C,
|
||||
float alphar, float alphai, BLASLONG K) {
|
||||
|
||||
float32x2x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init_expanded_m1n1();
|
||||
|
||||
for (; K > 3; K -= 4) {
|
||||
c1 = acc_expanded_m1n1(c1, vld1_f32(sa), vld1_f32(sb));
|
||||
c2 = acc_expanded_m1n1(c2, vld1_f32(sa + 2), vld1_f32(sb + 2));
|
||||
c3 = acc_expanded_m1n1(c3, vld1_f32(sa + 4), vld1_f32(sb + 4));
|
||||
c4 = acc_expanded_m1n1(c4, vld1_f32(sa + 6), vld1_f32(sb + 6));
|
||||
sa += 8; sb += 8;
|
||||
}
|
||||
c1.val[0] = vadd_f32(c1.val[0], c3.val[0]);
|
||||
c1.val[1] = vadd_f32(c1.val[1], c3.val[1]);
|
||||
c2.val[0] = vadd_f32(c2.val[0], c4.val[0]);
|
||||
c2.val[1] = vadd_f32(c2.val[1], c4.val[1]);
|
||||
c1.val[0] = vadd_f32(c1.val[0], c2.val[0]);
|
||||
c1.val[1] = vadd_f32(c1.val[1], c2.val[1]);
|
||||
for (; K; K--) {
|
||||
c1 = acc_expanded_m1n1(c1, vld1_f32(sa), vld1_f32(sb));
|
||||
sa += 2; sb += 2;
|
||||
}
|
||||
|
||||
store_expanded_m1n1(C, c1, expand_alpha_d(alphar, alphai));
|
||||
}
|
||||
|
||||
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K, FLOAT alphar, FLOAT alphai,
|
||||
FLOAT *sa, FLOAT *sb, FLOAT *C, BLASLONG LDC) {
|
||||
|
||||
BLASLONG n_left = N;
|
||||
for (; n_left >= 8; n_left -= 8) {
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c1_ = C;
|
||||
FLOAT *c2_ = C + LDC * 8;
|
||||
const FLOAT *b1_ = sb;
|
||||
const FLOAT *b2_ = sb + K * 8;
|
||||
BLASLONG m_left = M;
|
||||
for (; m_left >= 8; m_left -= 8) {
|
||||
kernel_8x4(a_, b1_, c1_, alphar, alphai, K, LDC);
|
||||
kernel_8x4(a_, b2_, c2_, alphar, alphai, K, LDC);
|
||||
a_ += 16 * K;
|
||||
c1_ += 16;
|
||||
c2_ += 16;
|
||||
}
|
||||
if (m_left >= 4) {
|
||||
m_left -= 4;
|
||||
kernel_4x4(a_, b1_, c1_, alphar, alphai, K, LDC);
|
||||
kernel_4x4(a_, b2_, c2_, alphar, alphai, K, LDC);
|
||||
a_ += 8 * K;
|
||||
c1_ += 8;
|
||||
c2_ += 8;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
kernel_2x4(a_, b1_, c1_, alphar, alphai, K, LDC);
|
||||
kernel_2x4(a_, b2_, c2_, alphar, alphai, K, LDC);
|
||||
a_ += 4 * K;
|
||||
c1_ += 4;
|
||||
c2_ += 4;
|
||||
}
|
||||
if (m_left) {
|
||||
kernel_1x4(a_, b1_, c1_, alphar, alphai, K, LDC);
|
||||
kernel_1x4(a_, b2_, c2_, alphar, alphai, K, LDC);
|
||||
}
|
||||
C += 16 * LDC;
|
||||
sb += 16 * K;
|
||||
}
|
||||
|
||||
if (n_left >= 4) {
|
||||
n_left -= 4;
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c_ = C;
|
||||
BLASLONG m_left = M;
|
||||
for (; m_left >= 8; m_left -= 8) {
|
||||
kernel_8x4(a_, sb, c_, alphar, alphai, K, LDC);
|
||||
a_ += 16 * K;
|
||||
c_ += 16;
|
||||
}
|
||||
if (m_left >= 4) {
|
||||
m_left -= 4;
|
||||
kernel_4x4(a_, sb, c_, alphar, alphai, K, LDC);
|
||||
a_ += 8 * K;
|
||||
c_ += 8;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
kernel_2x4(a_, sb, c_, alphar, alphai, K, LDC);
|
||||
a_ += 4 * K;
|
||||
c_ += 4;
|
||||
}
|
||||
if (m_left) {
|
||||
kernel_1x4(a_, sb, c_, alphar, alphai, K, LDC);
|
||||
}
|
||||
C += 8 * LDC;
|
||||
sb += 8 * K;
|
||||
}
|
||||
|
||||
if (n_left >= 2) {
|
||||
n_left -= 2;
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c_ = C;
|
||||
BLASLONG m_left = M;
|
||||
for (; m_left >= 8; m_left -= 8) {
|
||||
kernel_8x2(a_, sb, c_, alphar, alphai, K, LDC);
|
||||
a_ += 16 * K;
|
||||
c_ += 16;
|
||||
}
|
||||
if (m_left >= 4) {
|
||||
m_left -= 4;
|
||||
kernel_4x2(a_, sb, c_, alphar, alphai, K, LDC);
|
||||
a_ += 8 * K;
|
||||
c_ += 8;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
kernel_2x2(a_, sb, c_, alphar, alphai, K, LDC);
|
||||
a_ += 4 * K;
|
||||
c_ += 4;
|
||||
}
|
||||
if (m_left) {
|
||||
kernel_1x2(a_, sb, c_, alphar, alphai, K, LDC);
|
||||
}
|
||||
C += 4 * LDC;
|
||||
sb += 4 * K;
|
||||
}
|
||||
|
||||
if (n_left) {
|
||||
BLASLONG m_left = M;
|
||||
for (; m_left >= 8; m_left -= 8) {
|
||||
kernel_8x1(sa, sb, C, alphar, alphai, K);
|
||||
sa += 16 * K;
|
||||
C += 16;
|
||||
}
|
||||
if (m_left >= 4) {
|
||||
m_left -= 4;
|
||||
kernel_4x1(sa, sb, C, alphar, alphai, K);
|
||||
sa += 8 * K;
|
||||
C += 8;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
kernel_2x1(sa, sb, C, alphar, alphai, K);
|
||||
sa += 4 * K;
|
||||
C += 4;
|
||||
}
|
||||
if (m_left) {
|
||||
kernel_1x1(sa, sb, C, alphar, alphai, K);
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,890 @@
|
||||
/***************************************************************************
|
||||
Copyright (c) 2021, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written permission.
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A00 PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
|
||||
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*****************************************************************************/
|
||||
|
||||
#include "common.h"
|
||||
#include <arm_neon.h>
|
||||
|
||||
/**********************************************************
|
||||
* Function: dgemm_kernel_arm_cortex_a53_4x4_m4n12
|
||||
* Operation: C[4][12] += alpha * sa[4][K] * sb[K][12]
|
||||
* Matrix orders:
|
||||
* sa: column-major (leading dimension == 4)
|
||||
* sb: 3 concatenated row-major 4-column submatrices
|
||||
* C: column-major (leading dimension == LDC)
|
||||
*********************************************************/
|
||||
static inline void dgemm_kernel_arm_cortex_a53_4x4_m4n12(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *C,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
/** prefetch 4x12 elements from matrix C for RW purpose */
|
||||
__asm__ __volatile__(
|
||||
"mov x0,%[C]\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]; add x0,x0,%[LDC],LSL #3\n\t"
|
||||
"prfm pstl1keep,[x0]; prfm pstl1keep,[x0,#24]\n\t"
|
||||
::[C]"r"(C), [LDC]"r"(LDC):"x0");
|
||||
|
||||
/** 3 pointers to 3 submatrices of sb respectively */
|
||||
const FLOAT *b1_ = sb;
|
||||
const FLOAT *b2_ = sb + K * 4;
|
||||
const FLOAT *b3_ = sb + K * 8;
|
||||
|
||||
/** register mapping of 4x12 elements of C, row-id ==> coordinate-M, column-id ==> coordinate-N */
|
||||
/** v8.d[0] v10.d[0] v12.d[0] v14.d[0] v16.d[0] v18.d[0] v20.d[0] v22.d[0] v24.d[0] v26.d[0] v28.d[0] v30.d[0] */
|
||||
/** v8.d[1] v10.d[1] v12.d[1] v14.d[1] v16.d[1] v18.d[1] v20.d[1] v22.d[1] v24.d[1] v26.d[1] v28.d[1] v30.d[1] */
|
||||
/** v9.d[0] v11.d[0] v13.d[0] v15.d[0] v17.d[0] v19.d[0] v21.d[0] v23.d[0] v25.d[0] v27.d[0] v29.d[0] v31.d[0] */
|
||||
/** v9.d[1] v11.d[1] v13.d[1] v15.d[1] v17.d[1] v19.d[1] v21.d[1] v23.d[1] v25.d[1] v27.d[1] v29.d[1] v31.d[1] */
|
||||
|
||||
__asm__ __volatile__(
|
||||
"cmp %[K],#0\n\t"
|
||||
/** fill registers holding elements of C with 0.0 */
|
||||
"movi v8.16b,#0; movi v9.16b,#0; movi v10.16b,#0; movi v11.16b,#0\n\t"
|
||||
"movi v12.16b,#0; movi v13.16b,#0; movi v14.16b,#0; movi v15.16b,#0\n\t"
|
||||
"movi v16.16b,#0; movi v17.16b,#0; movi v18.16b,#0; movi v19.16b,#0\n\t"
|
||||
"movi v20.16b,#0; movi v21.16b,#0; movi v22.16b,#0; movi v23.16b,#0\n\t"
|
||||
"movi v24.16b,#0; movi v25.16b,#0; movi v26.16b,#0; movi v27.16b,#0\n\t"
|
||||
"movi v28.16b,#0; movi v29.16b,#0; movi v30.16b,#0; movi v31.16b,#0\n\t"
|
||||
"beq 4f; cmp %[K],#2\n\t"
|
||||
/** register v0-v3 for loading A, v4-v7 for loading B, x0 for transporting data */
|
||||
"ldp q0,q1,[%[sa]]; ldp q4,q5,[%[b1_]]\n\t"
|
||||
"ldr d6,[%[b2_]]; ldr x0,[%[b2_],#8]\n\t"
|
||||
"blt 3f; beq 2f\n\t"
|
||||
"1:\n\t"
|
||||
/** main loop with unroll_k = 2, specially designed for cortex-A53 NEON pipeline */
|
||||
"ldr d7,[%[b2_],#16]; fmov v6.d[1],x0\n\t"
|
||||
"fmla v8.2d,v0.2d,v4.d[0]; ldr x0,[%[b2_],#24]\n\t"
|
||||
"fmla v9.2d,v1.2d,v4.d[0]; prfm pldl1keep,[%[sa],#128]\n\t"
|
||||
"fmla v10.2d,v0.2d,v4.d[1]\n\t"
|
||||
"ldr d2,[%[sa],#32]; fmov v7.d[1],x0\n\t"
|
||||
"fmla v11.2d,v1.2d,v4.d[1]; ldr x0,[%[sa],#40]\n\t"
|
||||
"fmla v12.2d,v0.2d,v5.d[0]\n\t"
|
||||
"fmla v13.2d,v1.2d,v5.d[0]\n\t"
|
||||
"ldr d4,[%[b3_]]; fmov v2.d[1],x0\n\t"
|
||||
"fmla v14.2d,v0.2d,v5.d[1]; ldr x0,[%[b3_],#8]\n\t"
|
||||
"fmla v15.2d,v1.2d,v5.d[1]\n\t"
|
||||
"fmla v16.2d,v0.2d,v6.d[0]\n\t"
|
||||
"ldr d5,[%[b3_],#16]; fmov v4.d[1],x0\n\t"
|
||||
"fmla v17.2d,v1.2d,v6.d[0]; ldr x0,[%[b3_],#24]\n\t"
|
||||
"fmla v18.2d,v0.2d,v6.d[1]\n\t"
|
||||
"fmla v19.2d,v1.2d,v6.d[1]\n\t"
|
||||
"ldr d3,[%[sa],#48]; fmov v5.d[1],x0\n\t"
|
||||
"fmla v20.2d,v0.2d,v7.d[0]; ldr x0,[%[sa],#56]\n\t"
|
||||
"fmla v21.2d,v1.2d,v7.d[0]; add %[sa],%[sa],#64\n\t"
|
||||
"fmla v22.2d,v0.2d,v7.d[1]\n\t"
|
||||
"ldr d6,[%[b1_],#32]; fmov v3.d[1],x0\n\t"
|
||||
"fmla v23.2d,v1.2d,v7.d[1]; ldr x0,[%[b1_],#40]\n\t"
|
||||
"fmla v24.2d,v0.2d,v4.d[0]; prfm pldl1keep,[%[b1_],#128]\n\t"
|
||||
"fmla v25.2d,v1.2d,v4.d[0]\n\t"
|
||||
"ldr d7,[%[b1_],#48]; fmov v6.d[1],x0\n\t"
|
||||
"fmla v26.2d,v0.2d,v4.d[1]; ldr x0,[%[b1_],#56]\n\t"
|
||||
"fmla v27.2d,v1.2d,v4.d[1]; add %[b1_],%[b1_],#64\n\t"
|
||||
"fmla v28.2d,v0.2d,v5.d[0]\n\t"
|
||||
"ldr d4,[%[b2_],#32]; fmov v7.d[1],x0\n\t"
|
||||
"fmla v29.2d,v1.2d,v5.d[0]; ldr x0,[%[b2_],#40]\n\t"
|
||||
"fmla v30.2d,v0.2d,v5.d[1]; prfm pldl1keep,[%[b2_],#128]\n\t"
|
||||
"fmla v31.2d,v1.2d,v5.d[1]\n\t"
|
||||
"ldr d0,[%[sa]]; fmov v4.d[1],x0\n\t"
|
||||
"fmla v8.2d,v2.2d,v6.d[0]; ldr x0,[%[sa],#8]\n\t"
|
||||
"fmla v9.2d,v3.2d,v6.d[0]\n\t"
|
||||
"fmla v10.2d,v2.2d,v6.d[1]\n\t"
|
||||
"ldr d5,[%[b2_],#48]; fmov v0.d[1],x0\n\t"
|
||||
"fmla v11.2d,v3.2d,v6.d[1]; ldr x0,[%[b2_],#56]\n\t"
|
||||
"fmla v12.2d,v2.2d,v7.d[0]; add %[b2_],%[b2_],#64\n\t"
|
||||
"fmla v13.2d,v3.2d,v7.d[0]\n\t"
|
||||
"ldr d6,[%[b3_],#32]; fmov v5.d[1],x0\n\t"
|
||||
"fmla v14.2d,v2.2d,v7.d[1]; ldr x0,[%[b3_],#40]\n\t"
|
||||
"fmla v15.2d,v3.2d,v7.d[1]; prfm pldl1keep,[%[b3_],#128]\n\t"
|
||||
"fmla v16.2d,v2.2d,v4.d[0]\n\t"
|
||||
"ldr d7,[%[b3_],#48]; fmov v6.d[1],x0\n\t"
|
||||
"fmla v17.2d,v3.2d,v4.d[0]; ldr x0,[%[b3_],#56]\n\t"
|
||||
"fmla v18.2d,v2.2d,v4.d[1]; add %[b3_],%[b3_],#64\n\t"
|
||||
"fmla v19.2d,v3.2d,v4.d[1]\n\t"
|
||||
"ldr d1,[%[sa],#16]; fmov v7.d[1],x0\n\t"
|
||||
"fmla v20.2d,v2.2d,v5.d[0]; ldr x0,[%[sa],#24]\n\t"
|
||||
"fmla v21.2d,v3.2d,v5.d[0]\n\t"
|
||||
"fmla v22.2d,v2.2d,v5.d[1]\n\t"
|
||||
"ldr d4,[%[b1_]]; fmov v1.d[1],x0\n\t"
|
||||
"fmla v23.2d,v3.2d,v5.d[1]; ldr x0,[%[b1_],#8]\n\t"
|
||||
"fmla v24.2d,v2.2d,v6.d[0]\n\t"
|
||||
"fmla v25.2d,v3.2d,v6.d[0]\n\t"
|
||||
"ldr d5,[%[b1_],#16]; fmov v4.d[1],x0\n\t"
|
||||
"fmla v26.2d,v2.2d,v6.d[1]; ldr x0,[%[b1_],#24]\n\t"
|
||||
"fmla v27.2d,v3.2d,v6.d[1]; sub %[K],%[K],#2\n\t"
|
||||
"fmla v28.2d,v2.2d,v7.d[0]\n\t"
|
||||
"ldr d6,[%[b2_]]; fmov v5.d[1],x0\n\t"
|
||||
"fmla v29.2d,v3.2d,v7.d[0]; ldr x0,[%[b2_],#8]\n\t"
|
||||
"fmla v30.2d,v2.2d,v7.d[1]; cmp %[K],#2\n\t"
|
||||
"fmla v31.2d,v3.2d,v7.d[1]\n\t"
|
||||
"bgt 1b; blt 3f\n\t"
|
||||
"2:\n\t"
|
||||
/** tail part with k = 2 */
|
||||
"ldr d7,[%[b2_],#16]; fmov v6.d[1],x0\n\t"
|
||||
"fmla v8.2d,v0.2d,v4.d[0]; ldr x0,[%[b2_],#24]\n\t"
|
||||
"fmla v9.2d,v1.2d,v4.d[0]; prfm pldl1keep,[%[sa],#128]\n\t"
|
||||
"fmla v10.2d,v0.2d,v4.d[1]\n\t"
|
||||
"ldr d2,[%[sa],#32]; fmov v7.d[1],x0\n\t"
|
||||
"fmla v11.2d,v1.2d,v4.d[1]; ldr x0,[%[sa],#40]\n\t"
|
||||
"fmla v12.2d,v0.2d,v5.d[0]\n\t"
|
||||
"fmla v13.2d,v1.2d,v5.d[0]\n\t"
|
||||
"ldr d4,[%[b3_]]; fmov v2.d[1],x0\n\t"
|
||||
"fmla v14.2d,v0.2d,v5.d[1]; ldr x0,[%[b3_],#8]\n\t"
|
||||
"fmla v15.2d,v1.2d,v5.d[1]\n\t"
|
||||
"fmla v16.2d,v0.2d,v6.d[0]\n\t"
|
||||
"ldr d5,[%[b3_],#16]; fmov v4.d[1],x0\n\t"
|
||||
"fmla v17.2d,v1.2d,v6.d[0]; ldr x0,[%[b3_],#24]\n\t"
|
||||
"fmla v18.2d,v0.2d,v6.d[1]\n\t"
|
||||
"fmla v19.2d,v1.2d,v6.d[1]\n\t"
|
||||
"ldr d3,[%[sa],#48]; fmov v5.d[1],x0\n\t"
|
||||
"fmla v20.2d,v0.2d,v7.d[0]; ldr x0,[%[sa],#56]\n\t"
|
||||
"fmla v21.2d,v1.2d,v7.d[0]; add %[sa],%[sa],#64\n\t"
|
||||
"fmla v22.2d,v0.2d,v7.d[1]\n\t"
|
||||
"ldr d6,[%[b1_],#32]; fmov v3.d[1],x0\n\t"
|
||||
"fmla v23.2d,v1.2d,v7.d[1]; ldr x0,[%[b1_],#40]\n\t"
|
||||
"fmla v24.2d,v0.2d,v4.d[0]\n\t"
|
||||
"fmla v25.2d,v1.2d,v4.d[0]\n\t"
|
||||
"ldr d7,[%[b1_],#48]; fmov v6.d[1],x0\n\t"
|
||||
"fmla v26.2d,v0.2d,v4.d[1]; ldr x0,[%[b1_],#56]\n\t"
|
||||
"fmla v27.2d,v1.2d,v4.d[1]; add %[b1_],%[b1_],#64\n\t"
|
||||
"fmla v28.2d,v0.2d,v5.d[0]\n\t"
|
||||
"ldr d4,[%[b2_],#32]; fmov v7.d[1],x0\n\t"
|
||||
"fmla v29.2d,v1.2d,v5.d[0]; ldr x0,[%[b2_],#40]\n\t"
|
||||
"fmla v30.2d,v0.2d,v5.d[1]\n\t"
|
||||
"fmla v31.2d,v1.2d,v5.d[1]\n\t"
|
||||
"fmov v4.d[1],x0\n\t"
|
||||
"fmla v8.2d,v2.2d,v6.d[0]\n\t"
|
||||
"fmla v9.2d,v3.2d,v6.d[0]\n\t"
|
||||
"fmla v10.2d,v2.2d,v6.d[1]\n\t"
|
||||
"ldr d5,[%[b2_],#48]\n\t"
|
||||
"fmla v11.2d,v3.2d,v6.d[1]; ldr x0,[%[b2_],#56]\n\t"
|
||||
"fmla v12.2d,v2.2d,v7.d[0]; add %[b2_],%[b2_],#64\n\t"
|
||||
"fmla v13.2d,v3.2d,v7.d[0]\n\t"
|
||||
"ldr d6,[%[b3_],#32]; fmov v5.d[1],x0\n\t"
|
||||
"fmla v14.2d,v2.2d,v7.d[1]; ldr x0,[%[b3_],#40]\n\t"
|
||||
"fmla v15.2d,v3.2d,v7.d[1]\n\t"
|
||||
"fmla v16.2d,v2.2d,v4.d[0]\n\t"
|
||||
"ldr d7,[%[b3_],#48]; fmov v6.d[1],x0\n\t"
|
||||
"fmla v17.2d,v3.2d,v4.d[0]; ldr x0,[%[b3_],#56]\n\t"
|
||||
"fmla v18.2d,v2.2d,v4.d[1]; add %[b3_],%[b3_],#64\n\t"
|
||||
"fmla v19.2d,v3.2d,v4.d[1]\n\t"
|
||||
"fmov v7.d[1],x0\n\t"
|
||||
"fmla v20.2d,v2.2d,v5.d[0]\n\t"
|
||||
"fmla v21.2d,v3.2d,v5.d[0]\n\t"
|
||||
"fmla v22.2d,v2.2d,v5.d[1]\n\t"
|
||||
"fmla v23.2d,v3.2d,v5.d[1]\n\t"
|
||||
"fmla v24.2d,v2.2d,v6.d[0]\n\t"
|
||||
"fmla v25.2d,v3.2d,v6.d[0]\n\t"
|
||||
"fmla v26.2d,v2.2d,v6.d[1]\n\t"
|
||||
"fmla v27.2d,v3.2d,v6.d[1]; sub %[K],%[K],#2\n\t"
|
||||
"fmla v28.2d,v2.2d,v7.d[0]\n\t"
|
||||
"fmla v29.2d,v3.2d,v7.d[0]\n\t"
|
||||
"fmla v30.2d,v2.2d,v7.d[1]\n\t"
|
||||
"fmla v31.2d,v3.2d,v7.d[1]\n\t"
|
||||
"b 4f\n\t"
|
||||
"3:\n\t"
|
||||
/** tail part with k = 1 */
|
||||
"ldr d7,[%[b2_],#16]; fmov v6.d[1],x0\n\t"
|
||||
"fmla v8.2d,v0.2d,v4.d[0]; ldr x0,[%[b2_],#24]\n\t"
|
||||
"fmla v9.2d,v1.2d,v4.d[0]; add %[b2_],%[b2_],#32\n\t"
|
||||
"fmla v10.2d,v0.2d,v4.d[1]\n\t"
|
||||
"fmov v7.d[1],x0\n\t"
|
||||
"fmla v11.2d,v1.2d,v4.d[1]; add %[sa],%[sa],#32\n\t"
|
||||
"fmla v12.2d,v0.2d,v5.d[0]; add %[b1_],%[b1_],#32\n\t"
|
||||
"fmla v13.2d,v1.2d,v5.d[0]; sub %[K],%[K],#1\n\t"
|
||||
"ldr d4,[%[b3_]]\n\t"
|
||||
"fmla v14.2d,v0.2d,v5.d[1]; ldr x0,[%[b3_],#8]\n\t"
|
||||
"fmla v15.2d,v1.2d,v5.d[1]\n\t"
|
||||
"fmla v16.2d,v0.2d,v6.d[0]\n\t"
|
||||
"ldr d5,[%[b3_],#16]; fmov v4.d[1],x0\n\t"
|
||||
"fmla v17.2d,v1.2d,v6.d[0]; ldr x0,[%[b3_],#24]\n\t"
|
||||
"fmla v18.2d,v0.2d,v6.d[1]; add %[b3_],%[b3_],#32\n\t"
|
||||
"fmla v19.2d,v1.2d,v6.d[1]\n\t"
|
||||
"fmov v5.d[1],x0\n\t"
|
||||
"fmla v20.2d,v0.2d,v7.d[0]\n\t"
|
||||
"fmla v21.2d,v1.2d,v7.d[0]\n\t"
|
||||
"fmla v22.2d,v0.2d,v7.d[1]\n\t"
|
||||
"fmla v23.2d,v1.2d,v7.d[1]\n\t"
|
||||
"fmla v24.2d,v0.2d,v4.d[0]\n\t"
|
||||
"fmla v25.2d,v1.2d,v4.d[0]\n\t"
|
||||
"fmla v26.2d,v0.2d,v4.d[1]\n\t"
|
||||
"fmla v27.2d,v1.2d,v4.d[1]\n\t"
|
||||
"fmla v28.2d,v0.2d,v5.d[0]\n\t"
|
||||
"fmla v29.2d,v1.2d,v5.d[0]\n\t"
|
||||
"fmla v30.2d,v0.2d,v5.d[1]\n\t"
|
||||
"fmla v31.2d,v1.2d,v5.d[1]\n\t"
|
||||
/** store 4x12 elements to C */
|
||||
"4:\n\t"
|
||||
"ldr d0,%[alpha]; add x0,%[C],%[LDC],LSL #3\n\t"
|
||||
"ldp q1,q2,[%[C]]; ldp q3,q4,[x0]\n\t"
|
||||
"fmla v1.2d,v8.2d,v0.d[0]; fmla v2.2d,v9.2d,v0.d[0]\n\t"
|
||||
"fmla v3.2d,v10.2d,v0.d[0]; fmla v4.2d,v11.2d,v0.d[0]\n\t"
|
||||
"stp q1,q2,[%[C]]; add %[C],%[C],%[LDC],LSL #4\n\t"
|
||||
"stp q3,q4,[x0]; add x0,x0,%[LDC],LSL #4\n\t"
|
||||
"ldp q1,q2,[%[C]]; ldp q3,q4,[x0]\n\t"
|
||||
"fmla v1.2d,v12.2d,v0.d[0]; fmla v2.2d,v13.2d,v0.d[0]\n\t"
|
||||
"fmla v3.2d,v14.2d,v0.d[0]; fmla v4.2d,v15.2d,v0.d[0]\n\t"
|
||||
"stp q1,q2,[%[C]]; add %[C],%[C],%[LDC],LSL #4\n\t"
|
||||
"stp q3,q4,[x0]; add x0,x0,%[LDC],LSL #4\n\t"
|
||||
"ldp q1,q2,[%[C]]; ldp q3,q4,[x0]\n\t"
|
||||
"fmla v1.2d,v16.2d,v0.d[0]; fmla v2.2d,v17.2d,v0.d[0]\n\t"
|
||||
"fmla v3.2d,v18.2d,v0.d[0]; fmla v4.2d,v19.2d,v0.d[0]\n\t"
|
||||
"stp q1,q2,[%[C]]; add %[C],%[C],%[LDC],LSL #4\n\t"
|
||||
"stp q3,q4,[x0]; add x0,x0,%[LDC],LSL #4\n\t"
|
||||
"ldp q1,q2,[%[C]]; ldp q3,q4,[x0]\n\t"
|
||||
"fmla v1.2d,v20.2d,v0.d[0]; fmla v2.2d,v21.2d,v0.d[0]\n\t"
|
||||
"fmla v3.2d,v22.2d,v0.d[0]; fmla v4.2d,v23.2d,v0.d[0]\n\t"
|
||||
"stp q1,q2,[%[C]]; add %[C],%[C],%[LDC],LSL #4\n\t"
|
||||
"stp q3,q4,[x0]; add x0,x0,%[LDC],LSL #4\n\t"
|
||||
"ldp q1,q2,[%[C]]; ldp q3,q4,[x0]\n\t"
|
||||
"fmla v1.2d,v24.2d,v0.d[0]; fmla v2.2d,v25.2d,v0.d[0]\n\t"
|
||||
"fmla v3.2d,v26.2d,v0.d[0]; fmla v4.2d,v27.2d,v0.d[0]\n\t"
|
||||
"stp q1,q2,[%[C]]; add %[C],%[C],%[LDC],LSL #4\n\t"
|
||||
"stp q3,q4,[x0]; add x0,x0,%[LDC],LSL #4\n\t"
|
||||
"ldp q1,q2,[%[C]]; ldp q3,q4,[x0]\n\t"
|
||||
"fmla v1.2d,v28.2d,v0.d[0]; fmla v2.2d,v29.2d,v0.d[0]\n\t"
|
||||
"fmla v3.2d,v30.2d,v0.d[0]; fmla v4.2d,v31.2d,v0.d[0]\n\t"
|
||||
"stp q1,q2,[%[C]]; stp q3,q4,[x0]\n\t"
|
||||
:[sa]"+r"(sa), [b1_]"+r"(b1_), [b2_]"+r"(b2_), [b3_]"+r"(b3_), [C]"+r"(C), [K]"+r"(K)
|
||||
:[LDC]"r"(LDC), [alpha]"m"(alpha)
|
||||
:"cc", "memory", "x0", "v0", "v1", "v2", "v3", "v4", "v5", "v6", "v7",
|
||||
"v8", "v9", "v10", "v11", "v12", "v13", "v14", "v15", "v16", "v17", "v18", "v19",
|
||||
"v20", "v21", "v22", "v23", "v24", "v25", "v26", "v27", "v28", "v29", "v30", "v31");
|
||||
}
|
||||
|
||||
/**********************************************************
|
||||
* Operation:
|
||||
C[0] += alpha * up[0]; C[1] += alpha * up[1];
|
||||
C[2] += alpha * down[0]; C[3] += alpha * down[1];
|
||||
*********************************************************/
|
||||
static inline void dgemm_store_m4n1(FLOAT *C, float64x2_t up, float64x2_t down, FLOAT alpha) {
|
||||
float64x2_t t1 = vld1q_f64(C), t2 = vld1q_f64(C + 2);
|
||||
t1 = vfmaq_n_f64(t1, up, alpha);
|
||||
t2 = vfmaq_n_f64(t2, down, alpha);
|
||||
vst1q_f64(C, t1);
|
||||
vst1q_f64(C + 2, t2);
|
||||
}
|
||||
|
||||
/**********************************************************
|
||||
* Function: dgemm_kernel_arm64_4x4_m4n8
|
||||
* Operation: C[4][8] += alpha * sa[4][K] * sb[K][8]
|
||||
* Matrix orders:
|
||||
* sa: column-major (leading dimension == 4)
|
||||
* sb: 2 concatenated row-major 4-column submatrices
|
||||
* C: column-major (leading dimension == LDC)
|
||||
*********************************************************/
|
||||
static inline void dgemm_kernel_arm64_4x4_m4n8(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *C,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
const FLOAT *b1_ = sb;
|
||||
const FLOAT *b2_ = sb + K * 4;
|
||||
|
||||
/** register naming: c + m_id + n_id, m_id=1~2, n_id=1~8 */
|
||||
float64x2_t c11, c12, c13, c14, c15, c16, c17, c18;
|
||||
float64x2_t c21, c22, c23, c24, c25, c26, c27, c28;
|
||||
c11 = c12 = c13 = c14 = c15 = c16 = c17 = c18 = vdupq_n_f64(0);
|
||||
c21 = c22 = c23 = c24 = c25 = c26 = c27 = c28 = vdupq_n_f64(0);
|
||||
|
||||
for (; K; K--) {
|
||||
float64x2_t a1 = vld1q_f64(sa);
|
||||
float64x2_t a2 = vld1q_f64(sa + 2); sa += 4;
|
||||
|
||||
float64x2_t b1 = vld1q_f64(b1_);
|
||||
c11 = vfmaq_laneq_f64(c11, a1, b1, 0);
|
||||
c21 = vfmaq_laneq_f64(c21, a2, b1, 0);
|
||||
c12 = vfmaq_laneq_f64(c12, a1, b1, 1);
|
||||
c22 = vfmaq_laneq_f64(c22, a2, b1, 1);
|
||||
|
||||
float64x2_t b2 = vld1q_f64(b1_ + 2); b1_ += 4;
|
||||
c13 = vfmaq_laneq_f64(c13, a1, b2, 0);
|
||||
c23 = vfmaq_laneq_f64(c23, a2, b2, 0);
|
||||
c14 = vfmaq_laneq_f64(c14, a1, b2, 1);
|
||||
c24 = vfmaq_laneq_f64(c24, a2, b2, 1);
|
||||
|
||||
float64x2_t b3 = vld1q_f64(b2_);
|
||||
c15 = vfmaq_laneq_f64(c15, a1, b3, 0);
|
||||
c25 = vfmaq_laneq_f64(c25, a2, b3, 0);
|
||||
c16 = vfmaq_laneq_f64(c16, a1, b3, 1);
|
||||
c26 = vfmaq_laneq_f64(c26, a2, b3, 1);
|
||||
|
||||
float64x2_t b4 = vld1q_f64(b2_ + 2); b2_ += 4;
|
||||
c17 = vfmaq_laneq_f64(c17, a1, b4, 0);
|
||||
c27 = vfmaq_laneq_f64(c27, a2, b4, 0);
|
||||
c18 = vfmaq_laneq_f64(c18, a1, b4, 1);
|
||||
c28 = vfmaq_laneq_f64(c28, a2, b4, 1);
|
||||
}
|
||||
|
||||
dgemm_store_m4n1(C, c11, c21, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c12, c22, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c13, c23, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c14, c24, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c15, c25, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c16, c26, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c17, c27, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c18, c28, alpha);
|
||||
}
|
||||
|
||||
/**********************************************************
|
||||
* Function: dgemm_kernel_arm64_4x4_m4n4
|
||||
* Operation: C[4][4] += alpha * sa[4][K] * sb[K][4]
|
||||
* Matrix orders:
|
||||
* sa: column-major (leading dimension == 4)
|
||||
* sb: row-major (leading dimension == 4)
|
||||
* C: column-major (leading dimension == LDC)
|
||||
*********************************************************/
|
||||
static inline void dgemm_kernel_arm64_4x4_m4n4(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *C,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c11, c21, c12, c22, c13, c23, c14, c24;
|
||||
c11 = c21 = c12 = c22 = c13 = c23 = c14 = c24 = vdupq_n_f64(0);
|
||||
|
||||
for (; K; K--) {
|
||||
float64x2_t a1 = vld1q_f64(sa);
|
||||
float64x2_t a2 = vld1q_f64(sa + 2); sa += 4;
|
||||
float64x2_t b1 = vld1q_f64(sb);
|
||||
float64x2_t b2 = vld1q_f64(sb + 2); sb += 4;
|
||||
c11 = vfmaq_laneq_f64(c11, a1, b1, 0);
|
||||
c21 = vfmaq_laneq_f64(c21, a2, b1, 0);
|
||||
c12 = vfmaq_laneq_f64(c12, a1, b1, 1);
|
||||
c22 = vfmaq_laneq_f64(c22, a2, b1, 1);
|
||||
c13 = vfmaq_laneq_f64(c13, a1, b2, 0);
|
||||
c23 = vfmaq_laneq_f64(c23, a2, b2, 0);
|
||||
c14 = vfmaq_laneq_f64(c14, a1, b2, 1);
|
||||
c24 = vfmaq_laneq_f64(c24, a2, b2, 1);
|
||||
}
|
||||
|
||||
dgemm_store_m4n1(C, c11, c21, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c12, c22, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c13, c23, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c14, c24, alpha);
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m4n2(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *C,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c11_1, c11_2, c21_1, c21_2, c12_1, c12_2, c22_1, c22_2;
|
||||
c11_1 = c11_2 = c21_1 = c21_2 = c12_1 = c12_2 = c22_1 = c22_2 = vdupq_n_f64(0);
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float64x2_t b1 = vld1q_f64(sb), b2 = vld1q_f64(sb + 2); sb += 4;
|
||||
float64x2_t a1_1 = vld1q_f64(sa), a2_1 = vld1q_f64(sa + 2),
|
||||
a1_2 = vld1q_f64(sa + 4), a2_2 = vld1q_f64(sa + 6); sa += 8;
|
||||
c11_1 = vfmaq_laneq_f64(c11_1, a1_1, b1, 0);
|
||||
c21_1 = vfmaq_laneq_f64(c21_1, a2_1, b1, 0);
|
||||
c12_1 = vfmaq_laneq_f64(c12_1, a1_1, b1, 1);
|
||||
c22_1 = vfmaq_laneq_f64(c22_1, a2_1, b1, 1);
|
||||
c11_2 = vfmaq_laneq_f64(c11_2, a1_2, b2, 0);
|
||||
c21_2 = vfmaq_laneq_f64(c21_2, a2_2, b2, 0);
|
||||
c12_2 = vfmaq_laneq_f64(c12_2, a1_2, b2, 1);
|
||||
c22_2 = vfmaq_laneq_f64(c22_2, a2_2, b2, 1);
|
||||
}
|
||||
c11_1 = vaddq_f64(c11_1, c11_2);
|
||||
c21_1 = vaddq_f64(c21_1, c21_2);
|
||||
c12_1 = vaddq_f64(c12_1, c12_2);
|
||||
c22_1 = vaddq_f64(c22_1, c22_2);
|
||||
if (K) {
|
||||
float64x2_t b1 = vld1q_f64(sb); sb += 2;
|
||||
float64x2_t a1 = vld1q_f64(sa), a2 = vld1q_f64(sa + 2); sa += 4;
|
||||
c11_1 = vfmaq_laneq_f64(c11_1, a1, b1, 0);
|
||||
c21_1 = vfmaq_laneq_f64(c21_1, a2, b1, 0);
|
||||
c12_1 = vfmaq_laneq_f64(c12_1, a1, b1, 1);
|
||||
c22_1 = vfmaq_laneq_f64(c22_1, a2, b1, 1);
|
||||
}
|
||||
|
||||
dgemm_store_m4n1(C, c11_1, c21_1, alpha); C += LDC;
|
||||
dgemm_store_m4n1(C, c12_1, c22_1, alpha);
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m4n1(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *C,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c11_1, c11_2, c21_1, c21_2;
|
||||
c11_1 = c11_2 = c21_1 = c21_2 = vdupq_n_f64(0);
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float64x2_t b1 = vld1q_f64(sb); sb += 2;
|
||||
c11_1 = vfmaq_laneq_f64(c11_1, vld1q_f64(sa), b1, 0);
|
||||
c21_1 = vfmaq_laneq_f64(c21_1, vld1q_f64(sa + 2), b1, 0);
|
||||
c11_2 = vfmaq_laneq_f64(c11_2, vld1q_f64(sa + 4), b1, 1);
|
||||
c21_2 = vfmaq_laneq_f64(c21_2, vld1q_f64(sa + 6), b1, 1);
|
||||
sa += 8;
|
||||
}
|
||||
c11_1 = vaddq_f64(c11_1, c11_2);
|
||||
c21_1 = vaddq_f64(c21_1, c21_2);
|
||||
if (K) {
|
||||
double b1 = *sb++;
|
||||
c11_1 = vfmaq_n_f64(c11_1, vld1q_f64(sa), b1);
|
||||
c21_1 = vfmaq_n_f64(c21_1, vld1q_f64(sa + 2), b1);
|
||||
sa += 4;
|
||||
}
|
||||
|
||||
dgemm_store_m4n1(C, c11_1, c21_1, alpha);
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m2n12(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *c,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c01, c02, c03, c04, c11, c12, c13, c14, c21, c22, c23, c24;
|
||||
c01 = c02 = c03 = c04 = c11 = c12 = c13 = c14 =
|
||||
c21 = c22 = c23 = c24 = vdupq_n_f64(0);
|
||||
|
||||
const FLOAT *b1_ = sb;
|
||||
const FLOAT *b2_ = sb + 4 * K;
|
||||
const FLOAT *b3_ = b2_ + 4 * K;
|
||||
|
||||
for (; K; K--) {
|
||||
const float64x2_t a1 = vld1q_f64(sa); sa += 2;
|
||||
|
||||
float64x2_t b1 = vld1q_f64(b1_), b2 = vld1q_f64(b1_ + 2); b1_ += 4;
|
||||
c01 = vfmaq_laneq_f64(c01, a1, b1, 0);
|
||||
c02 = vfmaq_laneq_f64(c02, a1, b1, 1);
|
||||
c03 = vfmaq_laneq_f64(c03, a1, b2, 0);
|
||||
c04 = vfmaq_laneq_f64(c04, a1, b2, 1);
|
||||
|
||||
b1 = vld1q_f64(b2_); b2 = vld1q_f64(b2_ + 2); b2_ += 4;
|
||||
c11 = vfmaq_laneq_f64(c11, a1, b1, 0);
|
||||
c12 = vfmaq_laneq_f64(c12, a1, b1, 1);
|
||||
c13 = vfmaq_laneq_f64(c13, a1, b2, 0);
|
||||
c14 = vfmaq_laneq_f64(c14, a1, b2, 1);
|
||||
|
||||
b1 = vld1q_f64(b3_); b2 = vld1q_f64(b3_ + 2); b3_ += 4;
|
||||
c21 = vfmaq_laneq_f64(c21, a1, b1, 0);
|
||||
c22 = vfmaq_laneq_f64(c22, a1, b1, 1);
|
||||
c23 = vfmaq_laneq_f64(c23, a1, b2, 0);
|
||||
c24 = vfmaq_laneq_f64(c24, a1, b2, 1);
|
||||
}
|
||||
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c01, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c02, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c03, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c04, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c11, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c12, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c13, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c14, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c21, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c22, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c23, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c24, alpha));
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m2n8(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *c,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c01, c02, c03, c04, c11, c12, c13, c14;
|
||||
c01 = c02 = c03 = c04 = c11 = c12 = c13 = c14 = vdupq_n_f64(0);
|
||||
|
||||
const FLOAT *b1_ = sb;
|
||||
const FLOAT *b2_ = sb + 4 * K;
|
||||
|
||||
for (; K; K--) {
|
||||
const float64x2_t a1 = vld1q_f64(sa); sa += 2;
|
||||
|
||||
float64x2_t b1 = vld1q_f64(b1_), b2 = vld1q_f64(b1_ + 2); b1_ += 4;
|
||||
c01 = vfmaq_laneq_f64(c01, a1, b1, 0);
|
||||
c02 = vfmaq_laneq_f64(c02, a1, b1, 1);
|
||||
c03 = vfmaq_laneq_f64(c03, a1, b2, 0);
|
||||
c04 = vfmaq_laneq_f64(c04, a1, b2, 1);
|
||||
|
||||
b1 = vld1q_f64(b2_); b2 = vld1q_f64(b2_ + 2); b2_ += 4;
|
||||
c11 = vfmaq_laneq_f64(c11, a1, b1, 0);
|
||||
c12 = vfmaq_laneq_f64(c12, a1, b1, 1);
|
||||
c13 = vfmaq_laneq_f64(c13, a1, b2, 0);
|
||||
c14 = vfmaq_laneq_f64(c14, a1, b2, 1);
|
||||
}
|
||||
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c01, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c02, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c03, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c04, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c11, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c12, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c13, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c14, alpha));
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m2n4(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *c,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c1_1, c1_2, c2_1, c2_2, c3_1, c3_2, c4_1, c4_2;
|
||||
c1_1 = c1_2 = c2_1 = c2_2 = c3_1 = c3_2 = c4_1 = c4_2 = vdupq_n_f64(0);
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float64x2_t a1 = vld1q_f64(sa), a2 = vld1q_f64(sa + 2); sa += 4;
|
||||
float64x2_t b1_1 = vld1q_f64(sb), b2_1 = vld1q_f64(sb + 2);
|
||||
float64x2_t b1_2 = vld1q_f64(sb + 4), b2_2 = vld1q_f64(sb + 6); sb += 8;
|
||||
|
||||
c1_1 = vfmaq_laneq_f64(c1_1, a1, b1_1, 0);
|
||||
c2_1 = vfmaq_laneq_f64(c2_1, a1, b1_1, 1);
|
||||
c3_1 = vfmaq_laneq_f64(c3_1, a1, b2_1, 0);
|
||||
c4_1 = vfmaq_laneq_f64(c4_1, a1, b2_1, 1);
|
||||
|
||||
c1_2 = vfmaq_laneq_f64(c1_2, a2, b1_2, 0);
|
||||
c2_2 = vfmaq_laneq_f64(c2_2, a2, b1_2, 1);
|
||||
c3_2 = vfmaq_laneq_f64(c3_2, a2, b2_2, 0);
|
||||
c4_2 = vfmaq_laneq_f64(c4_2, a2, b2_2, 1);
|
||||
}
|
||||
c1_1 = vaddq_f64(c1_1, c1_2);
|
||||
c2_1 = vaddq_f64(c2_1, c2_2);
|
||||
c3_1 = vaddq_f64(c3_1, c3_2);
|
||||
c4_1 = vaddq_f64(c4_1, c4_2);
|
||||
if (K) {
|
||||
float64x2_t a1 = vld1q_f64(sa); sa += 2;
|
||||
float64x2_t b1 = vld1q_f64(sb), b2 = vld1q_f64(sb + 2); sb += 4;
|
||||
c1_1 = vfmaq_laneq_f64(c1_1, a1, b1, 0);
|
||||
c2_1 = vfmaq_laneq_f64(c2_1, a1, b1, 1);
|
||||
c3_1 = vfmaq_laneq_f64(c3_1, a1, b2, 0);
|
||||
c4_1 = vfmaq_laneq_f64(c4_1, a1, b2, 1);
|
||||
}
|
||||
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c1_1, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c2_1, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c3_1, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c4_1, alpha));
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m2n2(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *c,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c1_1, c1_2, c2_1, c2_2;
|
||||
c1_1 = c1_2 = c2_1 = c2_2 = vdupq_n_f64(0);
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float64x2_t a1 = vld1q_f64(sa), a2 = vld1q_f64(sa + 2); sa += 4;
|
||||
float64x2_t b1 = vld1q_f64(sb), b2 = vld1q_f64(sb + 2); sb += 4;
|
||||
|
||||
c1_1 = vfmaq_laneq_f64(c1_1, a1, b1, 0);
|
||||
c2_1 = vfmaq_laneq_f64(c2_1, a1, b1, 1);
|
||||
c1_2 = vfmaq_laneq_f64(c1_2, a2, b2, 0);
|
||||
c2_2 = vfmaq_laneq_f64(c2_2, a2, b2, 1);
|
||||
}
|
||||
c1_1 = vaddq_f64(c1_1, c1_2);
|
||||
c2_1 = vaddq_f64(c2_1, c2_2);
|
||||
if (K) {
|
||||
float64x2_t a1 = vld1q_f64(sa); sa += 2;
|
||||
float64x2_t b1 = vld1q_f64(sb); sb += 2;
|
||||
c1_1 = vfmaq_laneq_f64(c1_1, a1, b1, 0);
|
||||
c2_1 = vfmaq_laneq_f64(c2_1, a1, b1, 1);
|
||||
}
|
||||
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c1_1, alpha)); c += LDC;
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c2_1, alpha));
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m2n1(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *c,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = vdupq_n_f64(0);
|
||||
|
||||
for (; K > 3; K -= 4) {
|
||||
float64x2_t b12 = vld1q_f64(sb), b34 = vld1q_f64(sb + 2); sb += 4;
|
||||
c1 = vfmaq_laneq_f64(c1, vld1q_f64(sa), b12, 0);
|
||||
c2 = vfmaq_laneq_f64(c2, vld1q_f64(sa + 2), b12, 1);
|
||||
c3 = vfmaq_laneq_f64(c3, vld1q_f64(sa + 4), b34, 0);
|
||||
c4 = vfmaq_laneq_f64(c4, vld1q_f64(sa + 6), b34, 1);
|
||||
sa += 8;
|
||||
}
|
||||
c1 = vaddq_f64(c1, c2);
|
||||
c3 = vaddq_f64(c3, c4);
|
||||
c1 = vaddq_f64(c1, c3);
|
||||
for (; K; K--) {
|
||||
c1 = vfmaq_n_f64(c1, vld1q_f64(sa), *sb++);
|
||||
sa += 2;
|
||||
}
|
||||
|
||||
vst1q_f64(c, vfmaq_n_f64(vld1q_f64(c), c1, alpha));
|
||||
}
|
||||
|
||||
static inline void dgemm_store_m1n2(double *C, float64x2_t vc,
|
||||
double alpha, BLASLONG LDC) {
|
||||
double c0 = vgetq_lane_f64(vc, 0);
|
||||
double c1 = vgetq_lane_f64(vc, 1);
|
||||
C[0] += c0 * alpha;
|
||||
C[LDC] += c1 * alpha;
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m1n12(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *C,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c1, c2, c3, c4, c5, c6;
|
||||
c1 = c2 = c3 = c4 = c5 = c6 = vdupq_n_f64(0);
|
||||
|
||||
const double *b1_ = sb;
|
||||
const double *b2_ = sb + 4 * K;
|
||||
const double *b3_ = b2_ + 4 * K;
|
||||
|
||||
for (; K; K--) {
|
||||
const double a1 = *sa++;
|
||||
c1 = vfmaq_n_f64(c1, vld1q_f64(b1_), a1);
|
||||
c2 = vfmaq_n_f64(c2, vld1q_f64(b1_ + 2), a1); b1_ += 4;
|
||||
c3 = vfmaq_n_f64(c3, vld1q_f64(b2_), a1);
|
||||
c4 = vfmaq_n_f64(c4, vld1q_f64(b2_ + 2), a1); b2_ += 4;
|
||||
c5 = vfmaq_n_f64(c5, vld1q_f64(b3_), a1);
|
||||
c6 = vfmaq_n_f64(c6, vld1q_f64(b3_ + 2), a1); b3_ += 4;
|
||||
}
|
||||
|
||||
dgemm_store_m1n2(C, c1, alpha, LDC); C += LDC * 2;
|
||||
dgemm_store_m1n2(C, c2, alpha, LDC); C += LDC * 2;
|
||||
dgemm_store_m1n2(C, c3, alpha, LDC); C += LDC * 2;
|
||||
dgemm_store_m1n2(C, c4, alpha, LDC); C += LDC * 2;
|
||||
dgemm_store_m1n2(C, c5, alpha, LDC); C += LDC * 2;
|
||||
dgemm_store_m1n2(C, c6, alpha, LDC);
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m1n8(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *C,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = vdupq_n_f64(0);
|
||||
|
||||
const double *b1_ = sb;
|
||||
const double *b2_ = sb + 4 * K;
|
||||
|
||||
for (; K; K--) {
|
||||
const double a1 = *sa++;
|
||||
c1 = vfmaq_n_f64(c1, vld1q_f64(b1_), a1);
|
||||
c2 = vfmaq_n_f64(c2, vld1q_f64(b1_ + 2), a1); b1_ += 4;
|
||||
c3 = vfmaq_n_f64(c3, vld1q_f64(b2_), a1);
|
||||
c4 = vfmaq_n_f64(c4, vld1q_f64(b2_ + 2), a1); b2_ += 4;
|
||||
}
|
||||
|
||||
dgemm_store_m1n2(C, c1, alpha, LDC); C += LDC * 2;
|
||||
dgemm_store_m1n2(C, c2, alpha, LDC); C += LDC * 2;
|
||||
dgemm_store_m1n2(C, c3, alpha, LDC); C += LDC * 2;
|
||||
dgemm_store_m1n2(C, c4, alpha, LDC);
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m1n4(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *C,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c1_1, c1_2, c2_1, c2_2;
|
||||
c1_1 = c1_2 = c2_1 = c2_2 = vdupq_n_f64(0);
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float64x2_t a1 = vld1q_f64(sa); sa += 2;
|
||||
c1_1 = vfmaq_laneq_f64(c1_1, vld1q_f64(sb), a1, 0);
|
||||
c2_1 = vfmaq_laneq_f64(c2_1, vld1q_f64(sb + 2), a1, 0);
|
||||
c1_2 = vfmaq_laneq_f64(c1_2, vld1q_f64(sb + 4), a1, 1);
|
||||
c2_2 = vfmaq_laneq_f64(c2_2, vld1q_f64(sb + 6), a1, 1); sb += 8;
|
||||
}
|
||||
c1_1 = vaddq_f64(c1_1, c1_2);
|
||||
c2_1 = vaddq_f64(c2_1, c2_2);
|
||||
if (K) {
|
||||
double a1 = *sa++;
|
||||
c1_1 = vfmaq_n_f64(c1_1, vld1q_f64(sb), a1);
|
||||
c2_1 = vfmaq_n_f64(c2_1, vld1q_f64(sb + 2), a1);
|
||||
sb += 4;
|
||||
}
|
||||
|
||||
dgemm_store_m1n2(C, c1_1, alpha, LDC); C += LDC * 2;
|
||||
dgemm_store_m1n2(C, c2_1, alpha, LDC);
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m1n2(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *C,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = vdupq_n_f64(0);
|
||||
|
||||
for (; K > 3; K -= 4) {
|
||||
float64x2_t a12 = vld1q_f64(sa), a34 = vld1q_f64(sa + 2); sa += 4;
|
||||
c1 = vfmaq_laneq_f64(c1, vld1q_f64(sb), a12, 0);
|
||||
c2 = vfmaq_laneq_f64(c2, vld1q_f64(sb + 2), a12, 1);
|
||||
c3 = vfmaq_laneq_f64(c3, vld1q_f64(sb + 4), a34, 0);
|
||||
c4 = vfmaq_laneq_f64(c4, vld1q_f64(sb + 6), a34, 1); sb += 8;
|
||||
}
|
||||
c1 = vaddq_f64(c1, c2);
|
||||
c3 = vaddq_f64(c3, c4);
|
||||
c1 = vaddq_f64(c1, c3);
|
||||
for (; K; K--) {
|
||||
c1 = vfmaq_n_f64(c1, vld1q_f64(sb), *sa++);
|
||||
sb += 2;
|
||||
}
|
||||
|
||||
dgemm_store_m1n2(C, c1, alpha, LDC);
|
||||
}
|
||||
|
||||
static inline void dgemm_kernel_arm64_4x4_m1n1(
|
||||
const FLOAT *sa, const FLOAT *sb, FLOAT *C,
|
||||
BLASLONG K, BLASLONG LDC, FLOAT alpha) {
|
||||
|
||||
float64x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = vdupq_n_f64(0);
|
||||
|
||||
for (; K > 7; K -= 8) {
|
||||
c1 = vfmaq_f64(c1, vld1q_f64(sb), vld1q_f64(sa));
|
||||
c2 = vfmaq_f64(c2, vld1q_f64(sb + 2), vld1q_f64(sa + 2));
|
||||
c3 = vfmaq_f64(c3, vld1q_f64(sb + 4), vld1q_f64(sa + 4));
|
||||
c4 = vfmaq_f64(c4, vld1q_f64(sb + 6), vld1q_f64(sa + 6));
|
||||
sa += 8; sb += 8;
|
||||
}
|
||||
c1 = vaddq_f64(c1, c2);
|
||||
c3 = vaddq_f64(c3, c4);
|
||||
c1 = vaddq_f64(c1, c3);
|
||||
double cs1 = vpaddd_f64(c1);
|
||||
for (; K; K--) {
|
||||
cs1 += (*sa++) * (*sb++);
|
||||
}
|
||||
|
||||
C[0] += cs1 * alpha;
|
||||
}
|
||||
|
||||
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K, FLOAT alpha,
|
||||
FLOAT *sa, FLOAT *sb, FLOAT *C, BLASLONG LDC) {
|
||||
|
||||
for (; N >= 12; N -= 12) {
|
||||
BLASLONG m_left = M;
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c_ = C;
|
||||
for (; m_left >= 4; m_left -= 4) {
|
||||
dgemm_kernel_arm_cortex_a53_4x4_m4n12(a_, sb, c_, K, LDC, alpha);
|
||||
c_ += 4;
|
||||
a_ += 4 * K;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
dgemm_kernel_arm64_4x4_m2n12(a_, sb, c_, K, LDC, alpha);
|
||||
c_ += 2;
|
||||
a_ += 2 * K;
|
||||
}
|
||||
if (m_left) {
|
||||
dgemm_kernel_arm64_4x4_m1n12(a_, sb, c_, K, LDC, alpha);
|
||||
}
|
||||
sb += 12 * K;
|
||||
C += 12 * LDC;
|
||||
}
|
||||
|
||||
if (N >= 8) {
|
||||
N -= 8;
|
||||
BLASLONG m_left = M;
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c_ = C;
|
||||
for (; m_left >= 4; m_left -= 4) {
|
||||
dgemm_kernel_arm64_4x4_m4n8(a_, sb, c_, K, LDC, alpha);
|
||||
c_ += 4;
|
||||
a_ += 4 * K;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
dgemm_kernel_arm64_4x4_m2n8(a_, sb, c_, K, LDC, alpha);
|
||||
c_ += 2;
|
||||
a_ += 2 * K;
|
||||
}
|
||||
if (m_left) {
|
||||
dgemm_kernel_arm64_4x4_m1n8(a_, sb, c_, K, LDC, alpha);
|
||||
}
|
||||
sb += 8 * K;
|
||||
C += 8 * LDC;
|
||||
} else if (N >= 4) {
|
||||
N -= 4;
|
||||
BLASLONG m_left = M;
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c_ = C;
|
||||
for (; m_left >= 4; m_left -= 4) {
|
||||
dgemm_kernel_arm64_4x4_m4n4(a_, sb, c_, K, LDC, alpha);
|
||||
c_ += 4;
|
||||
a_ += 4 * K;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
dgemm_kernel_arm64_4x4_m2n4(a_, sb, c_, K, LDC, alpha);
|
||||
c_ += 2;
|
||||
a_ += 2 * K;
|
||||
}
|
||||
if (m_left) {
|
||||
dgemm_kernel_arm64_4x4_m1n4(a_, sb, c_, K, LDC, alpha);
|
||||
}
|
||||
sb += 4 * K;
|
||||
C += 4 * LDC;
|
||||
}
|
||||
|
||||
if (N >= 2) {
|
||||
N -= 2;
|
||||
BLASLONG m_left = M;
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c_ = C;
|
||||
for (; m_left >= 4; m_left -= 4) {
|
||||
dgemm_kernel_arm64_4x4_m4n2(a_, sb, c_, K, LDC, alpha);
|
||||
c_ += 4;
|
||||
a_ += 4 * K;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
dgemm_kernel_arm64_4x4_m2n2(a_, sb, c_, K, LDC, alpha);
|
||||
c_ += 2;
|
||||
a_ += 2 * K;
|
||||
}
|
||||
if (m_left) {
|
||||
dgemm_kernel_arm64_4x4_m1n2(a_, sb, c_, K, LDC, alpha);
|
||||
}
|
||||
sb += 2 * K;
|
||||
C += 2 * LDC;
|
||||
}
|
||||
|
||||
if (N) {
|
||||
BLASLONG m_left = M;
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c_ = C;
|
||||
for (; m_left >= 4; m_left -= 4) {
|
||||
dgemm_kernel_arm64_4x4_m4n1(a_, sb, c_, K, LDC, alpha);
|
||||
c_ += 4;
|
||||
a_ += 4 * K;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
dgemm_kernel_arm64_4x4_m2n1(a_, sb, c_, K, LDC, alpha);
|
||||
c_ += 2;
|
||||
a_ += 2 * K;
|
||||
}
|
||||
if (m_left) {
|
||||
dgemm_kernel_arm64_4x4_m1n1(a_, sb, c_, K, LDC, alpha);
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,874 @@
|
||||
/*******************************************************************************
|
||||
Copyright (c) 2015, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written permission.
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
|
||||
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*******************************************************************************/
|
||||
|
||||
#define ASSEMBLER
|
||||
#include "common.h"
|
||||
|
||||
/* X0 X1 X2 s0 X3 x4 x5 x6 */
|
||||
/*int CNAME(BLASLONG bm,BLASLONG bn,BLASLONG bk,FLOAT alpha0,FLOAT* ba,FLOAT* bb,FLOAT* C,BLASLONG ldc )*/
|
||||
|
||||
#define origM x0
|
||||
#define origN x1
|
||||
#define origK x2
|
||||
#define origPA x3
|
||||
#define origPB x4
|
||||
#define pC x5
|
||||
#define LDC x6
|
||||
#define temp x7
|
||||
#define counterL x8
|
||||
#define counterI x9
|
||||
#define counterJ x10
|
||||
#define pB x11
|
||||
#define pCRow0 x12
|
||||
#define pCRow1 x13
|
||||
#define pCRow2 x14
|
||||
|
||||
#define lanes x15
|
||||
#define pA x16
|
||||
#define alpha x17
|
||||
|
||||
#define alpha0 d10
|
||||
#define alphaZ z2.d
|
||||
|
||||
#define A_PRE_SIZE 1536
|
||||
#define B_PRE_SIZE 512
|
||||
#define C_PRE_SIZE 128
|
||||
|
||||
// 00 origM
|
||||
// 01 origN
|
||||
// 02 origK
|
||||
// 03 origPA
|
||||
// 04 origPB
|
||||
// 05 pC
|
||||
// 06 origLDC -> LDC
|
||||
// 07 temp
|
||||
// 08 counterL
|
||||
// 09 counterI
|
||||
// 10 counterJ
|
||||
// 11 pB
|
||||
// 12 pCRow0
|
||||
// 13 pCRow1
|
||||
// 14 pCRow2
|
||||
// 15 lanes
|
||||
// 16 pA
|
||||
// 17
|
||||
// 18 must save
|
||||
// 19 must save
|
||||
// 20 must save
|
||||
// 21 must save
|
||||
// 22 must save
|
||||
// 23 must save
|
||||
// 24 must save
|
||||
// 25 must save
|
||||
// 26 must save
|
||||
// 27 must save
|
||||
// 28 must save
|
||||
// 29 frame
|
||||
// 30 link
|
||||
// 31 sp
|
||||
|
||||
//v00 ALPHA -> pA0_0
|
||||
//v01 pA0_1
|
||||
//v02 ALPHA0
|
||||
//v03
|
||||
//v04
|
||||
//v05
|
||||
//v06
|
||||
//v07
|
||||
//v08 must save pB0_0
|
||||
//v09 must save pB0_1
|
||||
//v10 must save pB0_2
|
||||
//v11 must save pB0_3
|
||||
//v12 must save pB0_4
|
||||
//v13 must save pB0_5
|
||||
//v14 must save pB0_6
|
||||
//v15 must save pB0_7
|
||||
//v16 must save C0
|
||||
//v17 must save C1
|
||||
//v18 must save C2
|
||||
//v19 must save C3
|
||||
//v20 must save C4
|
||||
//v21 must save C5
|
||||
//v22 must save C6
|
||||
//v23 must save C7
|
||||
|
||||
/*******************************************************************************
|
||||
* Macro definitions
|
||||
*******************************************************************************/
|
||||
|
||||
.macro INITv1x8
|
||||
dup z16.d, #0
|
||||
dup z17.d, #0
|
||||
dup z18.d, #0
|
||||
dup z19.d, #0
|
||||
dup z20.d, #0
|
||||
dup z21.d, #0
|
||||
dup z22.d, #0
|
||||
dup z23.d, #0
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x8_I
|
||||
ld1d z0.d, p1/z, [pA]
|
||||
ld1d z1.d, p1/z, [pA, lanes, lsl #3] // next one
|
||||
add pA, pA, lanes, lsl #4 // pA = pA + lanes * 2 * 8
|
||||
|
||||
ld1rd z8.d, p0/z, [pB]
|
||||
ld1rd z9.d, p0/z, [pB, 8]
|
||||
ld1rd z10.d, p0/z, [pB, 16]
|
||||
ld1rd z11.d, p0/z, [pB, 24]
|
||||
ld1rd z12.d, p0/z, [pB, 32]
|
||||
ld1rd z13.d, p0/z, [pB, 40]
|
||||
ld1rd z14.d, p0/z, [pB, 48]
|
||||
ld1rd z15.d, p0/z, [pB, 56]
|
||||
|
||||
add pB, pB, 64
|
||||
|
||||
fmla z16.d, p1/m, z0.d, z8.d
|
||||
ld1rd z8.d, p0/z, [pB]
|
||||
fmla z17.d, p1/m, z0.d, z9.d
|
||||
ld1rd z9.d, p0/z, [pB, 8]
|
||||
fmla z18.d, p1/m, z0.d, z10.d
|
||||
ld1rd z10.d, p0/z, [pB, 16]
|
||||
fmla z19.d, p1/m, z0.d, z11.d
|
||||
ld1rd z11.d, p0/z, [pB, 24]
|
||||
fmla z20.d, p1/m, z0.d, z12.d
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
ld1rd z12.d, p0/z, [pB, 32]
|
||||
fmla z21.d, p1/m, z0.d, z13.d
|
||||
ld1rd z13.d, p0/z, [pB, 40]
|
||||
fmla z22.d, p1/m, z0.d, z14.d
|
||||
ld1rd z14.d, p0/z, [pB, 48]
|
||||
fmla z23.d, p1/m, z0.d, z15.d
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE+64]
|
||||
ld1rd z15.d, p0/z, [pB, 56]
|
||||
|
||||
add pB, pB, 64
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x8_M1
|
||||
ld1d z1.d, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #3 // pA = pA + lanes * 8
|
||||
|
||||
fmla z16.d, p1/m, z0.d, z8.d
|
||||
ld1rd z8.d, p0/z, [pB]
|
||||
fmla z17.d, p1/m, z0.d, z9.d
|
||||
ld1rd z9.d, p0/z, [pB, 8]
|
||||
fmla z18.d, p1/m, z0.d, z10.d
|
||||
ld1rd z10.d, p0/z, [pB, 16]
|
||||
fmla z19.d, p1/m, z0.d, z11.d
|
||||
ld1rd z11.d, p0/z, [pB, 24]
|
||||
fmla z20.d, p1/m, z0.d, z12.d
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
ld1rd z12.d, p0/z, [pB, 32]
|
||||
fmla z21.d, p1/m, z0.d, z13.d
|
||||
ld1rd z13.d, p0/z, [pB, 40]
|
||||
fmla z22.d, p1/m, z0.d, z14.d
|
||||
ld1rd z14.d, p0/z, [pB, 48]
|
||||
fmla z23.d, p1/m, z0.d, z15.d
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE+64]
|
||||
ld1rd z15.d, p0/z, [pB, 56]
|
||||
|
||||
add pB, pB, 64
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x8_M2
|
||||
ld1d z0.d, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #3 // pA = pA + lanes * 8
|
||||
|
||||
fmla z16.d, p1/m, z1.d, z8.d
|
||||
ld1rd z8.d, p0/z, [pB]
|
||||
fmla z17.d, p1/m, z1.d, z9.d
|
||||
ld1rd z9.d, p0/z, [pB, 8]
|
||||
fmla z18.d, p1/m, z1.d, z10.d
|
||||
ld1rd z10.d, p0/z, [pB, 16]
|
||||
fmla z19.d, p1/m, z1.d, z11.d
|
||||
ld1rd z11.d, p0/z, [pB, 24]
|
||||
fmla z20.d, p1/m, z1.d, z12.d
|
||||
ld1rd z12.d, p0/z, [pB, 32]
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
fmla z21.d, p1/m, z1.d, z13.d
|
||||
ld1rd z13.d, p0/z, [pB, 40]
|
||||
fmla z22.d, p1/m, z1.d, z14.d
|
||||
ld1rd z14.d, p0/z, [pB, 48]
|
||||
fmla z23.d, p1/m, z1.d, z15.d
|
||||
ld1rd z15.d, p0/z, [pB, 56]
|
||||
|
||||
add pB, pB, 64
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x8_E
|
||||
fmla z16.d, p1/m, z1.d, z8.d
|
||||
fmla z17.d, p1/m, z1.d, z9.d
|
||||
fmla z18.d, p1/m, z1.d, z10.d
|
||||
fmla z19.d, p1/m, z1.d, z11.d
|
||||
fmla z20.d, p1/m, z1.d, z12.d
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
fmla z21.d, p1/m, z1.d, z13.d
|
||||
fmla z22.d, p1/m, z1.d, z14.d
|
||||
fmla z23.d, p1/m, z1.d, z15.d
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x8_SUB
|
||||
ld1d z0.d, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #3 // pA = pA + lanes * 8
|
||||
|
||||
ld1rd z8.d, p0/z, [pB]
|
||||
ld1rd z9.d, p0/z, [pB, 8]
|
||||
ld1rd z10.d, p0/z, [pB, 16]
|
||||
ld1rd z11.d, p0/z, [pB, 24]
|
||||
ld1rd z12.d, p0/z, [pB, 32]
|
||||
ld1rd z13.d, p0/z, [pB, 40]
|
||||
ld1rd z14.d, p0/z, [pB, 48]
|
||||
ld1rd z15.d, p0/z, [pB, 56]
|
||||
|
||||
add pB, pB, 64
|
||||
|
||||
fmla z16.d, p1/m, z0.d, z8.d
|
||||
fmla z17.d, p1/m, z0.d, z9.d
|
||||
fmla z18.d, p1/m, z0.d, z10.d
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
fmla z19.d, p1/m, z0.d, z11.d
|
||||
fmla z20.d, p1/m, z0.d, z12.d
|
||||
fmla z21.d, p1/m, z0.d, z13.d
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
fmla z22.d, p1/m, z0.d, z14.d
|
||||
fmla z23.d, p1/m, z0.d, z15.d
|
||||
|
||||
.endm
|
||||
|
||||
.macro SAVEv1x8
|
||||
|
||||
prfm PLDL2KEEP, [pCRow0, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow0, LDC
|
||||
ld1d z24.d, p1/z, [pCRow0]
|
||||
fmla z24.d, p1/m, z16.d, alphaZ
|
||||
st1d z24.d, p1, [pCRow0]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
add pCRow2, pCRow1, LDC
|
||||
ld1d z25.d, p1/z, [pCRow1]
|
||||
fmla z25.d, p1/m, z17.d, alphaZ
|
||||
st1d z25.d, p1, [pCRow1]
|
||||
prfm PLDL2KEEP, [pCRow2, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow2, LDC
|
||||
ld1d z26.d, p1/z, [pCRow2]
|
||||
fmla z26.d, p1/m, z18.d, alphaZ
|
||||
st1d z26.d, p1, [pCRow2]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
add pCRow2, pCRow1, LDC
|
||||
ld1d z27.d, p1/z, [pCRow1]
|
||||
fmla z27.d, p1/m, z19.d, alphaZ
|
||||
st1d z27.d, p1, [pCRow1]
|
||||
prfm PLDL2KEEP, [pCRow2, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow2, LDC
|
||||
ld1d z28.d, p1/z, [pCRow2]
|
||||
fmla z28.d, p1/m, z20.d, alphaZ
|
||||
st1d z28.d, p1, [pCRow2]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
add pCRow2, pCRow1, LDC
|
||||
ld1d z29.d, p1/z, [pCRow1]
|
||||
fmla z29.d, p1/m, z21.d, alphaZ
|
||||
st1d z29.d, p1, [pCRow1]
|
||||
prfm PLDL2KEEP, [pCRow2, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow2, LDC
|
||||
ld1d z30.d, p1/z, [pCRow2]
|
||||
fmla z30.d, p1/m, z22.d, alphaZ
|
||||
st1d z30.d, p1, [pCRow2]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
ld1d z31.d, p1/z, [pCRow1]
|
||||
fmla z31.d, p1/m, z23.d, alphaZ
|
||||
st1d z31.d, p1, [pCRow1]
|
||||
|
||||
add pCRow0, pCRow0, lanes, lsl #3 // pC = pC + lanes * 8
|
||||
|
||||
.endm
|
||||
|
||||
/******************************************************************************/
|
||||
|
||||
.macro INITv1x4
|
||||
dup z16.d, #0
|
||||
dup z17.d, #0
|
||||
dup z18.d, #0
|
||||
dup z19.d, #0
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x4_SUB
|
||||
ld1d z0.d, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #3 // pA = pA + lanes * 8
|
||||
|
||||
ld1rd z8.d, p0/z, [pB]
|
||||
ld1rd z9.d, p0/z, [pB, 8]
|
||||
ld1rd z10.d, p0/z, [pB, 16]
|
||||
ld1rd z11.d, p0/z, [pB, 24]
|
||||
|
||||
add pB, pB, 32
|
||||
|
||||
fmla z16.d, p1/m, z0.d, z8.d
|
||||
fmla z17.d, p1/m, z0.d, z9.d
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
fmla z18.d, p1/m, z0.d, z10.d
|
||||
fmla z19.d, p1/m, z0.d, z11.d
|
||||
|
||||
.endm
|
||||
|
||||
.macro SAVEv1x4
|
||||
|
||||
prfm PLDL2KEEP, [pCRow0, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow0, LDC
|
||||
ld1d z24.d, p1/z, [pCRow0]
|
||||
fmla z24.d, p1/m, z16.d, alphaZ
|
||||
st1d z24.d, p1, [pCRow0]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
add pCRow2, pCRow1, LDC
|
||||
ld1d z25.d, p1/z, [pCRow1]
|
||||
fmla z25.d, p1/m, z17.d, alphaZ
|
||||
st1d z25.d, p1, [pCRow1]
|
||||
prfm PLDL2KEEP, [pCRow2, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow2, LDC
|
||||
ld1d z26.d, p1/z, [pCRow2]
|
||||
fmla z26.d, p1/m, z18.d, alphaZ
|
||||
st1d z26.d, p1, [pCRow2]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
ld1d z27.d, p1/z, [pCRow1]
|
||||
fmla z27.d, p1/m, z19.d, alphaZ
|
||||
st1d z27.d, p1, [pCRow1]
|
||||
|
||||
add pCRow0, pCRow0, lanes, lsl #3 // pC = pC + lanes * 8
|
||||
|
||||
.endm
|
||||
|
||||
/******************************************************************************/
|
||||
|
||||
.macro INITv1x2
|
||||
dup z16.d, #0
|
||||
dup z17.d, #0
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x2_SUB
|
||||
ld1d z0.d, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #3 // pA = pA + lanes * 8
|
||||
|
||||
ld1rd z8.d, p0/z, [pB]
|
||||
ld1rd z9.d, p0/z, [pB, 8]
|
||||
|
||||
add pB, pB, 16
|
||||
|
||||
fmla z16.d, p1/m, z0.d, z8.d
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
fmla z17.d, p1/m, z0.d, z9.d
|
||||
|
||||
.endm
|
||||
|
||||
.macro SAVEv1x2
|
||||
|
||||
prfm PLDL2KEEP, [pCRow0, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow0, LDC
|
||||
ld1d z24.d, p1/z, [pCRow0]
|
||||
fmla z24.d, p1/m, z16.d, alphaZ
|
||||
st1d z24.d, p1, [pCRow0]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
ld1d z25.d, p1/z, [pCRow1]
|
||||
fmla z25.d, p1/m, z17.d, alphaZ
|
||||
st1d z25.d, p1, [pCRow1]
|
||||
|
||||
add pCRow0, pCRow0, lanes, lsl #3 // pC = pC + lanes * 8
|
||||
|
||||
.endm
|
||||
|
||||
/******************************************************************************/
|
||||
|
||||
.macro INITv1x1
|
||||
dup z16.d, #0
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x1_SUB
|
||||
ld1d z0.d, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #3 // pA = pA + lanes * 8
|
||||
|
||||
ld1rd z8.d, p0/z, [pB]
|
||||
|
||||
add pB, pB, 8
|
||||
|
||||
fmla z16.d, p1/m, z0.d, z8.d
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
|
||||
.endm
|
||||
|
||||
.macro SAVEv1x1
|
||||
|
||||
prfm PLDL2KEEP, [pCRow0, #C_PRE_SIZE]
|
||||
|
||||
ld1d z24.d, p1/z, [pCRow0]
|
||||
fmla z24.d, p1/m, z16.d, alphaZ
|
||||
st1d z24.d, p1, [pCRow0]
|
||||
|
||||
|
||||
add pCRow0, pCRow0, lanes, lsl #3 // pC = pC + lanes * 8
|
||||
|
||||
.endm
|
||||
|
||||
|
||||
/*******************************************************************************
|
||||
* End of macro definitions
|
||||
*******************************************************************************/
|
||||
|
||||
PROLOGUE
|
||||
|
||||
.align 5
|
||||
add sp, sp, #-(11 * 16)
|
||||
stp d8, d9, [sp, #(0 * 16)]
|
||||
stp d10, d11, [sp, #(1 * 16)]
|
||||
stp d12, d13, [sp, #(2 * 16)]
|
||||
stp d14, d15, [sp, #(3 * 16)]
|
||||
stp d16, d17, [sp, #(4 * 16)]
|
||||
stp x18, x19, [sp, #(5 * 16)]
|
||||
stp x20, x21, [sp, #(6 * 16)]
|
||||
stp x22, x23, [sp, #(7 * 16)]
|
||||
stp x24, x25, [sp, #(8 * 16)]
|
||||
stp x26, x27, [sp, #(9 * 16)]
|
||||
str x28, [sp, #(10 * 16)]
|
||||
|
||||
prfm PLDL1KEEP, [origPB]
|
||||
prfm PLDL1KEEP, [origPA]
|
||||
|
||||
fmov alpha, d0
|
||||
dup alphaZ, alpha
|
||||
|
||||
lsl LDC, LDC, #3 // ldc = ldc * 8
|
||||
ptrue p0.d // create true predicate
|
||||
|
||||
mov pB, origPB
|
||||
// Loop over N
|
||||
mov counterJ, origN
|
||||
asr counterJ, counterJ, #3 // J = J / 8
|
||||
cmp counterJ, #0
|
||||
ble .Ldgemm_kernel_L4_BEGIN
|
||||
|
||||
/******************************************************************************/
|
||||
/* Repeat this as long as there are 8 left in N */
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_BEGIN:
|
||||
mov pCRow0, pC
|
||||
|
||||
add pC, pC, LDC, lsl #3 // add 8 x LDC
|
||||
|
||||
mov pA, origPA // pA = start of A array
|
||||
|
||||
.Ldgemm_kernel_L8_Mv1_BEGIN:
|
||||
|
||||
/* Loop over M is done in an SVE fashion. This has the benefit of the last M%SVE_LEN iterations being done in a single sweep */
|
||||
mov counterI, #0
|
||||
whilelt p1.d, counterI, origM
|
||||
cntp lanes, p0, p1.d // lanes contain number of active SVE lanes in M dimension
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_Mv1_20:
|
||||
|
||||
mov pB, origPB
|
||||
INITv1x8 // fill with zeros
|
||||
|
||||
asr counterL , origK, #3 // L = K / 8
|
||||
cmp counterL , #2 // is there at least 4 to do?
|
||||
blt .Ldgemm_kernel_L8_Mv1_32
|
||||
|
||||
KERNELv1x8_I
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
|
||||
subs counterL, counterL, #2 // subtract 2
|
||||
ble .Ldgemm_kernel_L8_Mv1_22a
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_Mv1_22:
|
||||
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bgt .Ldgemm_kernel_L8_Mv1_22
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_Mv1_22a:
|
||||
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_E
|
||||
|
||||
b .Ldgemm_kernel_L8_Mv1_44
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_Mv1_32:
|
||||
|
||||
tst counterL, #1
|
||||
ble .Ldgemm_kernel_L8_Mv1_40
|
||||
|
||||
KERNELv1x8_I
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_E
|
||||
|
||||
|
||||
b .Ldgemm_kernel_L8_Mv1_44
|
||||
|
||||
.Ldgemm_kernel_L8_Mv1_40:
|
||||
|
||||
INITv1x8
|
||||
|
||||
.Ldgemm_kernel_L8_Mv1_44:
|
||||
|
||||
ands counterL , origK, #7
|
||||
ble .Ldgemm_kernel_L8_Mv1_100
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_Mv1_46:
|
||||
|
||||
KERNELv1x8_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bne .Ldgemm_kernel_L8_Mv1_46
|
||||
|
||||
.Ldgemm_kernel_L8_Mv1_100:
|
||||
prfm PLDL1KEEP, [pA]
|
||||
prfm PLDL1KEEP, [pA, #64]
|
||||
prfm PLDL1KEEP, [origPB]
|
||||
|
||||
SAVEv1x8
|
||||
|
||||
.Ldgemm_kernel_L8_Mv1_END:
|
||||
|
||||
incd counterI
|
||||
whilelt p1.d, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.d // lanes contain number of active SVE lanes in M dimension
|
||||
b.any .Ldgemm_kernel_L8_Mv1_20
|
||||
|
||||
.Ldgemm_kernel_L8_END:
|
||||
|
||||
lsl temp, origK, #6
|
||||
add origPB, origPB, temp // B = B + K * 8 * 8
|
||||
|
||||
subs counterJ, counterJ , #1 // j--
|
||||
bgt .Ldgemm_kernel_L8_BEGIN
|
||||
|
||||
/******************************************************************************/
|
||||
/* Repeat the same thing if 4 left in N */
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L4_BEGIN:
|
||||
|
||||
mov counterJ , origN
|
||||
tst counterJ , #4
|
||||
ble .Ldgemm_kernel_L2_BEGIN
|
||||
|
||||
|
||||
mov pCRow0, pC
|
||||
|
||||
add pC, pC, LDC, lsl #2 // add 4 x LDC
|
||||
|
||||
mov pA, origPA // pA = start of A array
|
||||
|
||||
.Ldgemm_kernel_L4_Mv1_BEGIN:
|
||||
|
||||
mov counterI, #0
|
||||
whilelt p1.d, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.d
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L4_Mv1_20:
|
||||
|
||||
mov pB, origPB
|
||||
INITv1x4 // fill with zeros
|
||||
|
||||
asr counterL , origK, #3 // L = K / 8
|
||||
cmp counterL , #0 // is there at least 4 to do?
|
||||
ble .Ldgemm_kernel_L4_Mv1_44
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L4_Mv1_22:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x4_SUB
|
||||
KERNELv1x4_SUB
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x4_SUB
|
||||
KERNELv1x4_SUB
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x4_SUB
|
||||
KERNELv1x4_SUB
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x4_SUB
|
||||
KERNELv1x4_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bgt .Ldgemm_kernel_L4_Mv1_22
|
||||
|
||||
.Ldgemm_kernel_L4_Mv1_44:
|
||||
|
||||
ands counterL , origK, #7
|
||||
ble .Ldgemm_kernel_L4_Mv1_100
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L4_Mv1_46:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x4_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bne .Ldgemm_kernel_L4_Mv1_46
|
||||
|
||||
.Ldgemm_kernel_L4_Mv1_100:
|
||||
prfm PLDL1KEEP, [pA]
|
||||
prfm PLDL1KEEP, [pA, #64]
|
||||
prfm PLDL1KEEP, [origPB]
|
||||
|
||||
SAVEv1x4
|
||||
|
||||
.Ldgemm_kernel_L4_Mv1_END:
|
||||
|
||||
incd counterI
|
||||
whilelt p1.d, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.d
|
||||
b.any .Ldgemm_kernel_L4_Mv1_20
|
||||
|
||||
|
||||
.Ldgemm_kernel_L4_END:
|
||||
lsl temp, origK, #5
|
||||
add origPB, origPB, temp // B = B + K * 4 * 8
|
||||
|
||||
/******************************************************************************/
|
||||
/* Repeat the same thing if 2 left in N */
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L2_BEGIN:
|
||||
|
||||
mov counterJ , origN
|
||||
tst counterJ , #2
|
||||
ble .Ldgemm_kernel_L1_BEGIN
|
||||
|
||||
mov pCRow0, pC
|
||||
|
||||
add pC, pC, LDC, lsl #1 // add 2 x LDC
|
||||
|
||||
mov pA, origPA // pA = start of A array
|
||||
|
||||
.Ldgemm_kernel_L2_Mv1_BEGIN:
|
||||
|
||||
mov counterI, #0
|
||||
whilelt p1.d, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.d
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L2_Mv1_20:
|
||||
|
||||
mov pB, origPB
|
||||
INITv1x2 // fill with zeros
|
||||
|
||||
asr counterL , origK, #3 // L = K / 8
|
||||
cmp counterL , #0 // is there at least 4 to do?
|
||||
ble .Ldgemm_kernel_L2_Mv1_44
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L2_Mv1_22:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bgt .Ldgemm_kernel_L2_Mv1_22
|
||||
|
||||
.Ldgemm_kernel_L2_Mv1_44:
|
||||
|
||||
ands counterL , origK, #7
|
||||
ble .Ldgemm_kernel_L2_Mv1_100
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L2_Mv1_46:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x2_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bne .Ldgemm_kernel_L2_Mv1_46
|
||||
|
||||
.Ldgemm_kernel_L2_Mv1_100:
|
||||
prfm PLDL1KEEP, [pA]
|
||||
prfm PLDL1KEEP, [pA, #64]
|
||||
prfm PLDL1KEEP, [origPB]
|
||||
|
||||
SAVEv1x2
|
||||
|
||||
.Ldgemm_kernel_L2_Mv1_END:
|
||||
|
||||
incd counterI
|
||||
whilelt p1.d, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.d
|
||||
b.any .Ldgemm_kernel_L2_Mv1_20
|
||||
|
||||
|
||||
.Ldgemm_kernel_L2_END:
|
||||
add origPB, origPB, origK, lsl #4 // B = B + K * 2 * 8
|
||||
|
||||
/******************************************************************************/
|
||||
/* Repeat the same thing if 1 left in N */
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L1_BEGIN:
|
||||
|
||||
mov counterJ , origN
|
||||
tst counterJ , #1
|
||||
ble .Ldgemm_kernel_L999 // done
|
||||
|
||||
mov pCRow0, pC
|
||||
|
||||
add pC, pC, LDC // add 1 x LDC
|
||||
|
||||
mov pA, origPA // pA = start of A array
|
||||
|
||||
.Ldgemm_kernel_L1_Mv1_BEGIN:
|
||||
|
||||
mov counterI, #0
|
||||
whilelt p1.d, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.d
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L1_Mv1_20:
|
||||
|
||||
mov pB, origPB
|
||||
INITv1x1 // fill with zeros
|
||||
|
||||
asr counterL , origK, #3 // L = K / 8
|
||||
cmp counterL , #0 // is there at least 8 to do?
|
||||
ble .Ldgemm_kernel_L1_Mv1_44
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L1_Mv1_22:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bgt .Ldgemm_kernel_L1_Mv1_22
|
||||
|
||||
.Ldgemm_kernel_L1_Mv1_44:
|
||||
|
||||
ands counterL , origK, #7
|
||||
ble .Ldgemm_kernel_L1_Mv1_100
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L1_Mv1_46:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x1_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bgt .Ldgemm_kernel_L1_Mv1_46
|
||||
|
||||
.Ldgemm_kernel_L1_Mv1_100:
|
||||
prfm PLDL1KEEP, [pA]
|
||||
prfm PLDL1KEEP, [pA, #64]
|
||||
prfm PLDL1KEEP, [origPB]
|
||||
|
||||
SAVEv1x1
|
||||
|
||||
.Ldgemm_kernel_L1_Mv1_END:
|
||||
|
||||
incd counterI
|
||||
whilelt p1.d, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.d
|
||||
b.any .Ldgemm_kernel_L1_Mv1_20
|
||||
|
||||
|
||||
.Ldgemm_kernel_L1_END:
|
||||
|
||||
/******************************************************************************/
|
||||
|
||||
.Ldgemm_kernel_L999:
|
||||
mov x0, #0 // set return value
|
||||
ldp d8, d9, [sp, #(0 * 16)]
|
||||
ldp d10, d11, [sp, #(1 * 16)]
|
||||
ldp d12, d13, [sp, #(2 * 16)]
|
||||
ldp d14, d15, [sp, #(3 * 16)]
|
||||
ldp d16, d17, [sp, #(4 * 16)]
|
||||
ldp x18, x19, [sp, #(5 * 16)]
|
||||
ldp x20, x21, [sp, #(6 * 16)]
|
||||
ldp x22, x23, [sp, #(7 * 16)]
|
||||
ldp x24, x25, [sp, #(8 * 16)]
|
||||
ldp x26, x27, [sp, #(9 * 16)]
|
||||
ldr x28, [sp, #(10 * 16)]
|
||||
add sp, sp, #(11*16)
|
||||
ret
|
||||
|
||||
EPILOGUE
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,79 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "common.h"
|
||||
#include <arm_sve.h>
|
||||
|
||||
// TODO: write in assembly with proper unrolling of inner loop
|
||||
int CNAME(BLASLONG m, BLASLONG n, IFLOAT *a, BLASLONG lda, IFLOAT *b){
|
||||
|
||||
BLASLONG j;
|
||||
IFLOAT *aoffset, *aoffset1, *boffset;
|
||||
|
||||
svint64_t lda_vec = svindex_s64(0LL, lda);
|
||||
uint64_t sve_size = svcntd();
|
||||
|
||||
aoffset = a;
|
||||
boffset = b;
|
||||
|
||||
j = 0;
|
||||
svbool_t pg = svwhilelt_b64(j, n);
|
||||
uint64_t active = svcntp_b64(svptrue_b64(), pg);
|
||||
do {
|
||||
|
||||
aoffset1 = aoffset;
|
||||
|
||||
uint64_t i_cnt = m;
|
||||
while (i_cnt--) {
|
||||
svfloat64_t a_vec = svld1_gather_index(pg, (double *) aoffset1, lda_vec);
|
||||
svst1_f64(pg, (double *) boffset, a_vec);
|
||||
aoffset1++;
|
||||
boffset += active;
|
||||
}
|
||||
aoffset += sve_size * lda;
|
||||
|
||||
j += svcntd();
|
||||
pg = svwhilelt_b64(j, n);
|
||||
active = svcntp_b64(svptrue_b64(), pg);
|
||||
|
||||
|
||||
} while (svptest_any(svptrue_b64(), pg));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -50,11 +50,10 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define B03 x16
|
||||
#define B04 x17
|
||||
|
||||
#define I x18
|
||||
#define J x19
|
||||
#define I x19
|
||||
#define J x20
|
||||
|
||||
#define TEMP1 x20
|
||||
#define TEMP2 x21
|
||||
#define TEMP1 x21
|
||||
|
||||
#define A_PREFETCH 2560
|
||||
#define B_PREFETCH 256
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "common.h"
|
||||
#include <arm_sve.h>
|
||||
|
||||
// TODO: write in assembly with proper unrolling of inner loop
|
||||
int CNAME(BLASLONG m, BLASLONG n, IFLOAT *a, BLASLONG lda, IFLOAT *b){
|
||||
|
||||
BLASLONG j;
|
||||
IFLOAT *aoffset, *aoffset1, *boffset;
|
||||
|
||||
uint64_t sve_size = svcntd();
|
||||
|
||||
aoffset = a;
|
||||
boffset = b;
|
||||
|
||||
j = 0;
|
||||
svbool_t pg = svwhilelt_b64(j, n);
|
||||
uint64_t active = svcntp_b64(svptrue_b64(), pg);
|
||||
do {
|
||||
|
||||
aoffset1 = aoffset;
|
||||
|
||||
uint64_t i_cnt = m;
|
||||
while (i_cnt--) {
|
||||
svfloat64_t a_vec = svld1(pg, (double *)aoffset1);
|
||||
svst1_f64(pg, (double *) boffset, a_vec);
|
||||
aoffset1 += lda;
|
||||
boffset += active;
|
||||
}
|
||||
aoffset += sve_size;
|
||||
|
||||
j += svcntd();
|
||||
pg = svwhilelt_b64(j, n);
|
||||
active = svcntp_b64(svptrue_b64(), pg);
|
||||
|
||||
} while (svptest_any(svptrue_b64(), pg));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -49,9 +49,10 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define pCRow3 x15
|
||||
#define pA x16
|
||||
#define alpha x17
|
||||
#define temp x18
|
||||
//#define temp x18
|
||||
#define tempOffset x19
|
||||
#define tempK x20
|
||||
#define temp x21
|
||||
|
||||
#define alpha0 d10
|
||||
#define alphaV0 v10.d[0]
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,874 @@
|
||||
/*******************************************************************************
|
||||
Copyright (c) 2015, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written permission.
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
|
||||
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*******************************************************************************/
|
||||
|
||||
#define ASSEMBLER
|
||||
#include "common.h"
|
||||
|
||||
/* X0 X1 X2 s0 X3 x4 x5 x6 */
|
||||
/*int CNAME(BLASLONG bm,BLASLONG bn,BLASLONG bk,FLOAT alpha0,FLOAT* ba,FLOAT* bb,FLOAT* C,BLASLONG ldc )*/
|
||||
|
||||
#define origM x0
|
||||
#define origN x1
|
||||
#define origK x2
|
||||
#define origPA x3
|
||||
#define origPB x4
|
||||
#define pC x5
|
||||
#define LDC x6
|
||||
#define temp x7
|
||||
#define counterL x8
|
||||
#define counterI x9
|
||||
#define counterJ x10
|
||||
#define pB x11
|
||||
#define pCRow0 x12
|
||||
#define pCRow1 x13
|
||||
#define pCRow2 x14
|
||||
|
||||
#define lanes x15
|
||||
#define pA x16
|
||||
#define alpha w17
|
||||
|
||||
#define alpha0 s10
|
||||
#define alphaZ z2.s
|
||||
|
||||
#define A_PRE_SIZE 1536
|
||||
#define B_PRE_SIZE 512
|
||||
#define C_PRE_SIZE 128
|
||||
|
||||
// 00 origM
|
||||
// 01 origN
|
||||
// 02 origK
|
||||
// 03 origPA
|
||||
// 04 origPB
|
||||
// 05 pC
|
||||
// 06 origLDC -> LDC
|
||||
// 07 temp
|
||||
// 08 counterL
|
||||
// 09 counterI
|
||||
// 10 counterJ
|
||||
// 11 pB
|
||||
// 12 pCRow0
|
||||
// 13 pCRow1
|
||||
// 14 pCRow2
|
||||
// 15 lanes
|
||||
// 16 pA
|
||||
// 17
|
||||
// 18 must save
|
||||
// 19 must save
|
||||
// 20 must save
|
||||
// 21 must save
|
||||
// 22 must save
|
||||
// 23 must save
|
||||
// 24 must save
|
||||
// 25 must save
|
||||
// 26 must save
|
||||
// 27 must save
|
||||
// 28 must save
|
||||
// 29 frame
|
||||
// 30 link
|
||||
// 31 sp
|
||||
|
||||
//v00 ALPHA -> pA0_0
|
||||
//v01 pA0_1
|
||||
//v02 ALPHA0
|
||||
//v03
|
||||
//v04
|
||||
//v05
|
||||
//v06
|
||||
//v07
|
||||
//v08 must save pB0_0
|
||||
//v09 must save pB0_1
|
||||
//v10 must save pB0_2
|
||||
//v11 must save pB0_3
|
||||
//v12 must save pB0_4
|
||||
//v13 must save pB0_5
|
||||
//v14 must save pB0_6
|
||||
//v15 must save pB0_7
|
||||
//v16 must save C0
|
||||
//v17 must save C1
|
||||
//v18 must save C2
|
||||
//v19 must save C3
|
||||
//v20 must save C4
|
||||
//v21 must save C5
|
||||
//v22 must save C6
|
||||
//v23 must save C7
|
||||
|
||||
/*******************************************************************************
|
||||
* Macro definitions
|
||||
*******************************************************************************/
|
||||
|
||||
.macro INITv1x8
|
||||
dup z16.s, #0
|
||||
dup z17.s, #0
|
||||
dup z18.s, #0
|
||||
dup z19.s, #0
|
||||
dup z20.s, #0
|
||||
dup z21.s, #0
|
||||
dup z22.s, #0
|
||||
dup z23.s, #0
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x8_I
|
||||
ld1w z0.s, p1/z, [pA]
|
||||
ld1w z1.s, p1/z, [pA, lanes, lsl #2] // next one
|
||||
add pA, pA, lanes, lsl #3 // pA = pA + lanes * 2 * 4
|
||||
|
||||
ld1rw z8.s, p0/z, [pB]
|
||||
ld1rw z9.s, p0/z, [pB, 4]
|
||||
ld1rw z10.s, p0/z, [pB, 8]
|
||||
ld1rw z11.s, p0/z, [pB, 12]
|
||||
ld1rw z12.s, p0/z, [pB, 16]
|
||||
ld1rw z13.s, p0/z, [pB, 20]
|
||||
ld1rw z14.s, p0/z, [pB, 24]
|
||||
ld1rw z15.s, p0/z, [pB, 28]
|
||||
|
||||
add pB, pB, 32
|
||||
|
||||
fmla z16.s, p1/m, z0.s, z8.s
|
||||
ld1rw z8.s, p0/z, [pB]
|
||||
fmla z17.s, p1/m, z0.s, z9.s
|
||||
ld1rw z9.s, p0/z, [pB, 4]
|
||||
fmla z18.s, p1/m, z0.s, z10.s
|
||||
ld1rw z10.s, p0/z, [pB, 8]
|
||||
fmla z19.s, p1/m, z0.s, z11.s
|
||||
ld1rw z11.s, p0/z, [pB, 12]
|
||||
fmla z20.s, p1/m, z0.s, z12.s
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
ld1rw z12.s, p0/z, [pB, 16]
|
||||
fmla z21.s, p1/m, z0.s, z13.s
|
||||
ld1rw z13.s, p0/z, [pB, 20]
|
||||
fmla z22.s, p1/m, z0.s, z14.s
|
||||
ld1rw z14.s, p0/z, [pB, 24]
|
||||
fmla z23.s, p1/m, z0.s, z15.s
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE+64]
|
||||
ld1rw z15.s, p0/z, [pB, 28]
|
||||
|
||||
add pB, pB, 32
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x8_M1
|
||||
ld1w z1.s, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #2 // pA = pA + lanes * 4
|
||||
|
||||
fmla z16.s, p1/m, z0.s, z8.s
|
||||
ld1rw z8.s, p0/z, [pB]
|
||||
fmla z17.s, p1/m, z0.s, z9.s
|
||||
ld1rw z9.s, p0/z, [pB, 4]
|
||||
fmla z18.s, p1/m, z0.s, z10.s
|
||||
ld1rw z10.s, p0/z, [pB, 8]
|
||||
fmla z19.s, p1/m, z0.s, z11.s
|
||||
ld1rw z11.s, p0/z, [pB, 12]
|
||||
fmla z20.s, p1/m, z0.s, z12.s
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
ld1rw z12.s, p0/z, [pB, 16]
|
||||
fmla z21.s, p1/m, z0.s, z13.s
|
||||
ld1rw z13.s, p0/z, [pB, 20]
|
||||
fmla z22.s, p1/m, z0.s, z14.s
|
||||
ld1rw z14.s, p0/z, [pB, 24]
|
||||
fmla z23.s, p1/m, z0.s, z15.s
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE+64]
|
||||
ld1rw z15.s, p0/z, [pB, 28]
|
||||
|
||||
add pB, pB, 32
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x8_M2
|
||||
ld1w z0.s, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #2 // pA = pA + lanes * 4
|
||||
|
||||
fmla z16.s, p1/m, z1.s, z8.s
|
||||
ld1rw z8.s, p0/z, [pB]
|
||||
fmla z17.s, p1/m, z1.s, z9.s
|
||||
ld1rw z9.s, p0/z, [pB, 4]
|
||||
fmla z18.s, p1/m, z1.s, z10.s
|
||||
ld1rw z10.s, p0/z, [pB, 8]
|
||||
fmla z19.s, p1/m, z1.s, z11.s
|
||||
ld1rw z11.s, p0/z, [pB, 12]
|
||||
fmla z20.s, p1/m, z1.s, z12.s
|
||||
ld1rw z12.s, p0/z, [pB, 16]
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
fmla z21.s, p1/m, z1.s, z13.s
|
||||
ld1rw z13.s, p0/z, [pB, 20]
|
||||
fmla z22.s, p1/m, z1.s, z14.s
|
||||
ld1rw z14.s, p0/z, [pB, 24]
|
||||
fmla z23.s, p1/m, z1.s, z15.s
|
||||
ld1rw z15.s, p0/z, [pB, 28]
|
||||
|
||||
add pB, pB, 32
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x8_E
|
||||
fmla z16.s, p1/m, z1.s, z8.s
|
||||
fmla z17.s, p1/m, z1.s, z9.s
|
||||
fmla z18.s, p1/m, z1.s, z10.s
|
||||
fmla z19.s, p1/m, z1.s, z11.s
|
||||
fmla z20.s, p1/m, z1.s, z12.s
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
fmla z21.s, p1/m, z1.s, z13.s
|
||||
fmla z22.s, p1/m, z1.s, z14.s
|
||||
fmla z23.s, p1/m, z1.s, z15.s
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x8_SUB
|
||||
ld1w z0.s, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #2 // pA = pA + lanes * 4
|
||||
|
||||
ld1rw z8.s, p0/z, [pB]
|
||||
ld1rw z9.s, p0/z, [pB, 4]
|
||||
ld1rw z10.s, p0/z, [pB, 8]
|
||||
ld1rw z11.s, p0/z, [pB, 12]
|
||||
ld1rw z12.s, p0/z, [pB, 16]
|
||||
ld1rw z13.s, p0/z, [pB, 20]
|
||||
ld1rw z14.s, p0/z, [pB, 24]
|
||||
ld1rw z15.s, p0/z, [pB, 28]
|
||||
|
||||
add pB, pB, 32
|
||||
|
||||
fmla z16.s, p1/m, z0.s, z8.s
|
||||
fmla z17.s, p1/m, z0.s, z9.s
|
||||
fmla z18.s, p1/m, z0.s, z10.s
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
fmla z19.s, p1/m, z0.s, z11.s
|
||||
fmla z20.s, p1/m, z0.s, z12.s
|
||||
fmla z21.s, p1/m, z0.s, z13.s
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
fmla z22.s, p1/m, z0.s, z14.s
|
||||
fmla z23.s, p1/m, z0.s, z15.s
|
||||
|
||||
.endm
|
||||
|
||||
.macro SAVEv1x8
|
||||
|
||||
prfm PLDL2KEEP, [pCRow0, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow0, LDC
|
||||
ld1w z24.s, p1/z, [pCRow0]
|
||||
fmla z24.s, p1/m, z16.s, alphaZ
|
||||
st1w z24.s, p1, [pCRow0]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
add pCRow2, pCRow1, LDC
|
||||
ld1w z25.s, p1/z, [pCRow1]
|
||||
fmla z25.s, p1/m, z17.s, alphaZ
|
||||
st1w z25.s, p1, [pCRow1]
|
||||
prfm PLDL2KEEP, [pCRow2, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow2, LDC
|
||||
ld1w z26.s, p1/z, [pCRow2]
|
||||
fmla z26.s, p1/m, z18.s, alphaZ
|
||||
st1w z26.s, p1, [pCRow2]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
add pCRow2, pCRow1, LDC
|
||||
ld1w z27.s, p1/z, [pCRow1]
|
||||
fmla z27.s, p1/m, z19.s, alphaZ
|
||||
st1w z27.s, p1, [pCRow1]
|
||||
prfm PLDL2KEEP, [pCRow2, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow2, LDC
|
||||
ld1w z28.s, p1/z, [pCRow2]
|
||||
fmla z28.s, p1/m, z20.s, alphaZ
|
||||
st1w z28.s, p1, [pCRow2]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
add pCRow2, pCRow1, LDC
|
||||
ld1w z29.s, p1/z, [pCRow1]
|
||||
fmla z29.s, p1/m, z21.s, alphaZ
|
||||
st1w z29.s, p1, [pCRow1]
|
||||
prfm PLDL2KEEP, [pCRow2, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow2, LDC
|
||||
ld1w z30.s, p1/z, [pCRow2]
|
||||
fmla z30.s, p1/m, z22.s, alphaZ
|
||||
st1w z30.s, p1, [pCRow2]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
ld1w z31.s, p1/z, [pCRow1]
|
||||
fmla z31.s, p1/m, z23.s, alphaZ
|
||||
st1w z31.s, p1, [pCRow1]
|
||||
|
||||
add pCRow0, pCRow0, lanes, lsl #2 // pC = pC + lanes * 4
|
||||
|
||||
.endm
|
||||
|
||||
/******************************************************************************/
|
||||
|
||||
.macro INITv1x4
|
||||
dup z16.s, #0
|
||||
dup z17.s, #0
|
||||
dup z18.s, #0
|
||||
dup z19.s, #0
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x4_SUB
|
||||
ld1w z0.s, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #2 // pA = pA + lanes * 4
|
||||
|
||||
ld1rw z8.s, p0/z, [pB]
|
||||
ld1rw z9.s, p0/z, [pB, 4]
|
||||
ld1rw z10.s, p0/z, [pB, 8]
|
||||
ld1rw z11.s, p0/z, [pB, 12]
|
||||
|
||||
add pB, pB, 16
|
||||
|
||||
fmla z16.s, p1/m, z0.s, z8.s
|
||||
fmla z17.s, p1/m, z0.s, z9.s
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
fmla z18.s, p1/m, z0.s, z10.s
|
||||
fmla z19.s, p1/m, z0.s, z11.s
|
||||
|
||||
.endm
|
||||
|
||||
.macro SAVEv1x4
|
||||
|
||||
prfm PLDL2KEEP, [pCRow0, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow0, LDC
|
||||
ld1w z24.s, p1/z, [pCRow0]
|
||||
fmla z24.s, p1/m, z16.s, alphaZ
|
||||
st1w z24.s, p1, [pCRow0]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
add pCRow2, pCRow1, LDC
|
||||
ld1w z25.s, p1/z, [pCRow1]
|
||||
fmla z25.s, p1/m, z17.s, alphaZ
|
||||
st1w z25.s, p1, [pCRow1]
|
||||
prfm PLDL2KEEP, [pCRow2, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow2, LDC
|
||||
ld1w z26.s, p1/z, [pCRow2]
|
||||
fmla z26.s, p1/m, z18.s, alphaZ
|
||||
st1w z26.s, p1, [pCRow2]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
ld1w z27.s, p1/z, [pCRow1]
|
||||
fmla z27.s, p1/m, z19.s, alphaZ
|
||||
st1w z27.s, p1, [pCRow1]
|
||||
|
||||
add pCRow0, pCRow0, lanes, lsl #2 // pC = pC + lanes * 4
|
||||
|
||||
.endm
|
||||
|
||||
/******************************************************************************/
|
||||
|
||||
.macro INITv1x2
|
||||
dup z16.s, #0
|
||||
dup z17.s, #0
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x2_SUB
|
||||
ld1w z0.s, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #2 // pA = pA + lanes * 4
|
||||
|
||||
ld1rw z8.s, p0/z, [pB]
|
||||
ld1rw z9.s, p0/z, [pB, 4]
|
||||
|
||||
add pB, pB, 8
|
||||
|
||||
fmla z16.s, p1/m, z0.s, z8.s
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
fmla z17.s, p1/m, z0.s, z9.s
|
||||
|
||||
.endm
|
||||
|
||||
.macro SAVEv1x2
|
||||
|
||||
prfm PLDL2KEEP, [pCRow0, #C_PRE_SIZE]
|
||||
|
||||
add pCRow1, pCRow0, LDC
|
||||
ld1w z24.s, p1/z, [pCRow0]
|
||||
fmla z24.s, p1/m, z16.s, alphaZ
|
||||
st1w z24.s, p1, [pCRow0]
|
||||
prfm PLDL2KEEP, [pCRow1, #C_PRE_SIZE]
|
||||
|
||||
ld1w z25.s, p1/z, [pCRow1]
|
||||
fmla z25.s, p1/m, z17.s, alphaZ
|
||||
st1w z25.s, p1, [pCRow1]
|
||||
|
||||
add pCRow0, pCRow0, lanes, lsl #2 // pC = pC + lanes * 4
|
||||
|
||||
.endm
|
||||
|
||||
/******************************************************************************/
|
||||
|
||||
.macro INITv1x1
|
||||
dup z16.s, #0
|
||||
.endm
|
||||
|
||||
.macro KERNELv1x1_SUB
|
||||
ld1w z0.s, p1/z, [pA]
|
||||
add pA, pA, lanes, lsl #2 // pA = pA + lanes * 8
|
||||
|
||||
ld1rw z8.s, p0/z, [pB]
|
||||
|
||||
add pB, pB, 4
|
||||
|
||||
fmla z16.s, p1/m, z0.s, z8.s
|
||||
prfm PLDL1KEEP, [pA, #A_PRE_SIZE]
|
||||
|
||||
.endm
|
||||
|
||||
.macro SAVEv1x1
|
||||
|
||||
prfm PLDL2KEEP, [pCRow0, #C_PRE_SIZE]
|
||||
|
||||
ld1w z24.s, p1/z, [pCRow0]
|
||||
fmla z24.s, p1/m, z16.s, alphaZ
|
||||
st1w z24.s, p1, [pCRow0]
|
||||
|
||||
|
||||
add pCRow0, pCRow0, lanes, lsl #2 // pC = pC + lanes * 4
|
||||
|
||||
.endm
|
||||
|
||||
|
||||
/*******************************************************************************
|
||||
* End of macro definitions
|
||||
*******************************************************************************/
|
||||
|
||||
PROLOGUE
|
||||
|
||||
.align 5
|
||||
add sp, sp, #-(11 * 16)
|
||||
stp d8, d9, [sp, #(0 * 16)]
|
||||
stp d10, d11, [sp, #(1 * 16)]
|
||||
stp d12, d13, [sp, #(2 * 16)]
|
||||
stp d14, d15, [sp, #(3 * 16)]
|
||||
stp d16, d17, [sp, #(4 * 16)]
|
||||
stp x18, x19, [sp, #(5 * 16)]
|
||||
stp x20, x21, [sp, #(6 * 16)]
|
||||
stp x22, x23, [sp, #(7 * 16)]
|
||||
stp x24, x25, [sp, #(8 * 16)]
|
||||
stp x26, x27, [sp, #(9 * 16)]
|
||||
str x28, [sp, #(10 * 16)]
|
||||
|
||||
prfm PLDL1KEEP, [origPB]
|
||||
prfm PLDL1KEEP, [origPA]
|
||||
|
||||
fmov alpha, s0
|
||||
dup alphaZ, alpha
|
||||
|
||||
lsl LDC, LDC, #2 // ldc = ldc * 4
|
||||
ptrue p0.s // create true predicate
|
||||
|
||||
mov pB, origPB
|
||||
// Loop over N
|
||||
mov counterJ, origN
|
||||
asr counterJ, counterJ, #3 // J = J / 8
|
||||
cmp counterJ, #0
|
||||
ble .Ldgemm_kernel_L4_BEGIN
|
||||
|
||||
/******************************************************************************/
|
||||
/* Repeat this as long as there are 8 left in N */
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_BEGIN:
|
||||
mov pCRow0, pC
|
||||
|
||||
add pC, pC, LDC, lsl #3 // add 8 x LDC
|
||||
|
||||
mov pA, origPA // pA = start of A array
|
||||
|
||||
.Ldgemm_kernel_L8_Mv1_BEGIN:
|
||||
|
||||
/* Loop over M is done in an SVE fashion. This has the benefit of the last M%SVE_LEN iterations being done in a single sweep */
|
||||
mov counterI, #0
|
||||
whilelt p1.s, counterI, origM
|
||||
cntp lanes, p0, p1.s // lanes contain number of active SVE lanes in M dimension
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_Mv1_20:
|
||||
|
||||
mov pB, origPB
|
||||
INITv1x8 // fill with zeros
|
||||
|
||||
asr counterL , origK, #3 // L = K / 8
|
||||
cmp counterL , #2 // is there at least 4 to do?
|
||||
blt .Ldgemm_kernel_L8_Mv1_32
|
||||
|
||||
KERNELv1x8_I
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
|
||||
subs counterL, counterL, #2 // subtract 2
|
||||
ble .Ldgemm_kernel_L8_Mv1_22a
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_Mv1_22:
|
||||
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bgt .Ldgemm_kernel_L8_Mv1_22
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_Mv1_22a:
|
||||
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_E
|
||||
|
||||
b .Ldgemm_kernel_L8_Mv1_44
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_Mv1_32:
|
||||
|
||||
tst counterL, #1
|
||||
ble .Ldgemm_kernel_L8_Mv1_40
|
||||
|
||||
KERNELv1x8_I
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_M2
|
||||
KERNELv1x8_M1
|
||||
KERNELv1x8_E
|
||||
|
||||
|
||||
b .Ldgemm_kernel_L8_Mv1_44
|
||||
|
||||
.Ldgemm_kernel_L8_Mv1_40:
|
||||
|
||||
INITv1x8
|
||||
|
||||
.Ldgemm_kernel_L8_Mv1_44:
|
||||
|
||||
ands counterL , origK, #7
|
||||
ble .Ldgemm_kernel_L8_Mv1_100
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L8_Mv1_46:
|
||||
|
||||
KERNELv1x8_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bne .Ldgemm_kernel_L8_Mv1_46
|
||||
|
||||
.Ldgemm_kernel_L8_Mv1_100:
|
||||
prfm PLDL1KEEP, [pA]
|
||||
prfm PLDL1KEEP, [pA, #64]
|
||||
prfm PLDL1KEEP, [origPB]
|
||||
|
||||
SAVEv1x8
|
||||
|
||||
.Ldgemm_kernel_L8_Mv1_END:
|
||||
|
||||
incw counterI
|
||||
whilelt p1.s, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.s // lanes contain number of active SVE lanes in M dimension
|
||||
b.any .Ldgemm_kernel_L8_Mv1_20
|
||||
|
||||
.Ldgemm_kernel_L8_END:
|
||||
|
||||
lsl temp, origK, #5
|
||||
add origPB, origPB, temp // B = B + K * 8 * 4
|
||||
|
||||
subs counterJ, counterJ , #1 // j--
|
||||
bgt .Ldgemm_kernel_L8_BEGIN
|
||||
|
||||
/******************************************************************************/
|
||||
/* Repeat the same thing if 4 left in N */
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L4_BEGIN:
|
||||
|
||||
mov counterJ , origN
|
||||
tst counterJ , #4
|
||||
ble .Ldgemm_kernel_L2_BEGIN
|
||||
|
||||
|
||||
mov pCRow0, pC
|
||||
|
||||
add pC, pC, LDC, lsl #2 // add 4 x LDC
|
||||
|
||||
mov pA, origPA // pA = start of A array
|
||||
|
||||
.Ldgemm_kernel_L4_Mv1_BEGIN:
|
||||
|
||||
mov counterI, #0
|
||||
whilelt p1.s, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.s
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L4_Mv1_20:
|
||||
|
||||
mov pB, origPB
|
||||
INITv1x4 // fill with zeros
|
||||
|
||||
asr counterL , origK, #3 // L = K / 8
|
||||
cmp counterL , #0 // is there at least 4 to do?
|
||||
ble .Ldgemm_kernel_L4_Mv1_44
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L4_Mv1_22:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x4_SUB
|
||||
KERNELv1x4_SUB
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x4_SUB
|
||||
KERNELv1x4_SUB
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x4_SUB
|
||||
KERNELv1x4_SUB
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x4_SUB
|
||||
KERNELv1x4_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bgt .Ldgemm_kernel_L4_Mv1_22
|
||||
|
||||
.Ldgemm_kernel_L4_Mv1_44:
|
||||
|
||||
ands counterL , origK, #7
|
||||
ble .Ldgemm_kernel_L4_Mv1_100
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L4_Mv1_46:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x4_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bne .Ldgemm_kernel_L4_Mv1_46
|
||||
|
||||
.Ldgemm_kernel_L4_Mv1_100:
|
||||
prfm PLDL1KEEP, [pA]
|
||||
prfm PLDL1KEEP, [pA, #64]
|
||||
prfm PLDL1KEEP, [origPB]
|
||||
|
||||
SAVEv1x4
|
||||
|
||||
.Ldgemm_kernel_L4_Mv1_END:
|
||||
|
||||
incw counterI
|
||||
whilelt p1.s, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.s
|
||||
b.any .Ldgemm_kernel_L4_Mv1_20
|
||||
|
||||
|
||||
.Ldgemm_kernel_L4_END:
|
||||
lsl temp, origK, #4
|
||||
add origPB, origPB, temp // B = B + K * 4 * 4
|
||||
|
||||
/******************************************************************************/
|
||||
/* Repeat the same thing if 2 left in N */
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L2_BEGIN:
|
||||
|
||||
mov counterJ , origN
|
||||
tst counterJ , #2
|
||||
ble .Ldgemm_kernel_L1_BEGIN
|
||||
|
||||
mov pCRow0, pC
|
||||
|
||||
add pC, pC, LDC, lsl #1 // add 2 x LDC
|
||||
|
||||
mov pA, origPA // pA = start of A array
|
||||
|
||||
.Ldgemm_kernel_L2_Mv1_BEGIN:
|
||||
|
||||
mov counterI, #0
|
||||
whilelt p1.s, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.s
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L2_Mv1_20:
|
||||
|
||||
mov pB, origPB
|
||||
INITv1x2 // fill with zeros
|
||||
|
||||
asr counterL , origK, #3 // L = K / 8
|
||||
cmp counterL , #0 // is there at least 4 to do?
|
||||
ble .Ldgemm_kernel_L2_Mv1_44
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L2_Mv1_22:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
KERNELv1x2_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bgt .Ldgemm_kernel_L2_Mv1_22
|
||||
|
||||
.Ldgemm_kernel_L2_Mv1_44:
|
||||
|
||||
ands counterL , origK, #7
|
||||
ble .Ldgemm_kernel_L2_Mv1_100
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L2_Mv1_46:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x2_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bne .Ldgemm_kernel_L2_Mv1_46
|
||||
|
||||
.Ldgemm_kernel_L2_Mv1_100:
|
||||
prfm PLDL1KEEP, [pA]
|
||||
prfm PLDL1KEEP, [pA, #64]
|
||||
prfm PLDL1KEEP, [origPB]
|
||||
|
||||
SAVEv1x2
|
||||
|
||||
.Ldgemm_kernel_L2_Mv1_END:
|
||||
|
||||
incw counterI
|
||||
whilelt p1.s, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.s
|
||||
b.any .Ldgemm_kernel_L2_Mv1_20
|
||||
|
||||
|
||||
.Ldgemm_kernel_L2_END:
|
||||
add origPB, origPB, origK, lsl #3 // B = B + K * 2 * 4
|
||||
|
||||
/******************************************************************************/
|
||||
/* Repeat the same thing if 1 left in N */
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L1_BEGIN:
|
||||
|
||||
mov counterJ , origN
|
||||
tst counterJ , #1
|
||||
ble .Ldgemm_kernel_L999 // done
|
||||
|
||||
mov pCRow0, pC
|
||||
|
||||
add pC, pC, LDC // add 1 x LDC
|
||||
|
||||
mov pA, origPA // pA = start of A array
|
||||
|
||||
.Ldgemm_kernel_L1_Mv1_BEGIN:
|
||||
|
||||
mov counterI, #0
|
||||
whilelt p1.s, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.s
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L1_Mv1_20:
|
||||
|
||||
mov pB, origPB
|
||||
INITv1x1 // fill with zeros
|
||||
|
||||
asr counterL , origK, #3 // L = K / 8
|
||||
cmp counterL , #0 // is there at least 8 to do?
|
||||
ble .Ldgemm_kernel_L1_Mv1_44
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L1_Mv1_22:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
KERNELv1x1_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bgt .Ldgemm_kernel_L1_Mv1_22
|
||||
|
||||
.Ldgemm_kernel_L1_Mv1_44:
|
||||
|
||||
ands counterL , origK, #7
|
||||
ble .Ldgemm_kernel_L1_Mv1_100
|
||||
|
||||
.align 5
|
||||
.Ldgemm_kernel_L1_Mv1_46:
|
||||
|
||||
prfm PLDL1KEEP, [pB, #B_PRE_SIZE]
|
||||
KERNELv1x1_SUB
|
||||
|
||||
subs counterL, counterL, #1
|
||||
bgt .Ldgemm_kernel_L1_Mv1_46
|
||||
|
||||
.Ldgemm_kernel_L1_Mv1_100:
|
||||
prfm PLDL1KEEP, [pA]
|
||||
prfm PLDL1KEEP, [pA, #64]
|
||||
prfm PLDL1KEEP, [origPB]
|
||||
|
||||
SAVEv1x1
|
||||
|
||||
.Ldgemm_kernel_L1_Mv1_END:
|
||||
|
||||
incw counterI
|
||||
whilelt p1.s, counterI, origM //SVE instruction
|
||||
cntp lanes, p0, p1.s
|
||||
b.any .Ldgemm_kernel_L1_Mv1_20
|
||||
|
||||
|
||||
.Ldgemm_kernel_L1_END:
|
||||
|
||||
/******************************************************************************/
|
||||
|
||||
.Ldgemm_kernel_L999:
|
||||
mov x0, #0 // set return value
|
||||
ldp d8, d9, [sp, #(0 * 16)]
|
||||
ldp d10, d11, [sp, #(1 * 16)]
|
||||
ldp d12, d13, [sp, #(2 * 16)]
|
||||
ldp d14, d15, [sp, #(3 * 16)]
|
||||
ldp d16, d17, [sp, #(4 * 16)]
|
||||
ldp x18, x19, [sp, #(5 * 16)]
|
||||
ldp x20, x21, [sp, #(6 * 16)]
|
||||
ldp x22, x23, [sp, #(7 * 16)]
|
||||
ldp x24, x25, [sp, #(8 * 16)]
|
||||
ldp x26, x27, [sp, #(9 * 16)]
|
||||
ldr x28, [sp, #(10 * 16)]
|
||||
add sp, sp, #(11*16)
|
||||
ret
|
||||
|
||||
EPILOGUE
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,78 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "common.h"
|
||||
#include <arm_sve.h>
|
||||
|
||||
// TODO: write in assembly with proper unrolling of inner loop
|
||||
int CNAME(BLASLONG m, BLASLONG n, IFLOAT *a, BLASLONG lda, IFLOAT *b){
|
||||
|
||||
BLASLONG j;
|
||||
IFLOAT *aoffset, *aoffset1, *boffset;
|
||||
|
||||
svint32_t lda_vec = svindex_s32(0LL, lda);
|
||||
uint32_t sve_size = svcntw();
|
||||
|
||||
aoffset = a;
|
||||
boffset = b;
|
||||
|
||||
j = 0;
|
||||
svbool_t pg = svwhilelt_b32(j, n);
|
||||
uint32_t active = svcntp_b32(svptrue_b32(), pg);
|
||||
do {
|
||||
|
||||
aoffset1 = aoffset;
|
||||
|
||||
uint32_t i_cnt = m;
|
||||
while (i_cnt--) {
|
||||
svfloat32_t a_vec = svld1_gather_index(pg, (float *) aoffset1, lda_vec);
|
||||
svst1_f32(pg, (float *) boffset, a_vec);
|
||||
aoffset1++;
|
||||
boffset += active;
|
||||
}
|
||||
aoffset += sve_size * lda;
|
||||
|
||||
j += svcntw();
|
||||
pg = svwhilelt_b32(j, n);
|
||||
active = svcntp_b32(svptrue_b32(), pg);
|
||||
|
||||
} while (svptest_any(svptrue_b32(), pg));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -30,7 +30,7 @@ All rights reserved.
|
||||
#define B00 x22
|
||||
|
||||
|
||||
#define I x18
|
||||
#define I x21
|
||||
#define J x19
|
||||
|
||||
#define TEMP1 x20
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "common.h"
|
||||
#include <arm_sve.h>
|
||||
|
||||
// TODO: write in assembly with proper unrolling of inner loop
|
||||
int CNAME(BLASLONG m, BLASLONG n, IFLOAT *a, BLASLONG lda, IFLOAT *b){
|
||||
|
||||
BLASLONG j;
|
||||
IFLOAT *aoffset, *aoffset1, *boffset;
|
||||
|
||||
uint32_t sve_size = svcntw();
|
||||
|
||||
aoffset = a;
|
||||
boffset = b;
|
||||
|
||||
j = 0;
|
||||
svbool_t pg = svwhilelt_b32(j, n);
|
||||
uint32_t active = svcntp_b32(svptrue_b32(), pg);
|
||||
do {
|
||||
|
||||
aoffset1 = aoffset;
|
||||
|
||||
uint32_t i_cnt = m;
|
||||
while (i_cnt--) {
|
||||
svfloat32_t a_vec = svld1(pg, (float *) aoffset1);
|
||||
svst1_f32(pg, (float *) boffset, a_vec);
|
||||
aoffset1 += lda;
|
||||
boffset += active;
|
||||
}
|
||||
aoffset += sve_size;
|
||||
|
||||
j += svcntw();
|
||||
pg = svwhilelt_b32(j, n);
|
||||
active = svcntp_b32(svptrue_b32(), pg);
|
||||
|
||||
} while (svptest_any(svptrue_b32(), pg));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -49,9 +49,10 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define pCRow3 x15
|
||||
#define pA x16
|
||||
#define alpha w17
|
||||
#define temp x18
|
||||
//#define temp x18
|
||||
#define tempOffset x19
|
||||
#define tempK x20
|
||||
#define temp x21
|
||||
|
||||
#define alpha0 s10
|
||||
#define alphaV0 v10.s[0]
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,143 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "common.h"
|
||||
#include <arm_sve.h>
|
||||
|
||||
int CNAME(BLASLONG m, BLASLONG n, FLOAT *a, BLASLONG lda, BLASLONG posX, BLASLONG posY, FLOAT *b){
|
||||
|
||||
BLASLONG i, offset;
|
||||
|
||||
#if defined(DOUBLE)
|
||||
uint64_t sve_size = svcntd();
|
||||
svint64_t posY_vec = svdup_s64(posY);
|
||||
svint64_t posX_vec = svdup_s64(posX);
|
||||
svint64_t lda_vec = svdup_s64(lda);
|
||||
svint64_t one_vec = svdup_s64(1LL);
|
||||
|
||||
int64_t j = 0;
|
||||
svbool_t pg = svwhilelt_b64(j, n);
|
||||
int64_t active = svcntp_b64(svptrue_b64(), pg);
|
||||
svint64_t index_neg = svindex_s64(0LL, -1LL);
|
||||
svint64_t index = svindex_s64(0LL, 1LL);
|
||||
do {
|
||||
offset = posX - posY;
|
||||
svint64_t vec_off = svdup_s64(offset);
|
||||
svbool_t cmp = svcmpgt(pg, vec_off, index_neg);
|
||||
|
||||
svint64_t temp = svadd_z(pg, posX_vec, index);
|
||||
svint64_t temp1 = svmla_z(pg, temp, posY_vec, lda_vec);
|
||||
svint64_t temp2 = svmla_z(pg, posY_vec, temp, lda);
|
||||
svint64_t gat_ind = svsel(cmp, temp1, temp2);
|
||||
|
||||
i = m;
|
||||
while (i>0) {
|
||||
svfloat64_t data_vec = svld1_gather_index(pg, a, gat_ind);
|
||||
|
||||
gat_ind = svadd_m(cmp, gat_ind, lda_vec);
|
||||
gat_ind = svadd_m(svnot_z(pg, cmp) , gat_ind, one_vec);
|
||||
|
||||
svst1(pg, b, data_vec);
|
||||
|
||||
b += active;
|
||||
offset --;
|
||||
vec_off = svsub_z(pg, vec_off, one_vec);
|
||||
cmp = svcmpgt(pg, vec_off, index_neg);
|
||||
|
||||
i--;
|
||||
}
|
||||
|
||||
posX += sve_size;
|
||||
posX_vec = svdup_s64(posX);
|
||||
j += sve_size;
|
||||
pg = svwhilelt_b64(j, n);
|
||||
active = svcntp_b64(svptrue_b64(), pg);
|
||||
} while (svptest_any(svptrue_b64(), pg));
|
||||
|
||||
#else
|
||||
uint32_t sve_size = svcntw();
|
||||
svint32_t posY_vec = svdup_s32(posY);
|
||||
svint32_t posX_vec = svdup_s32(posX);
|
||||
svint32_t lda_vec = svdup_s32(lda);
|
||||
svint32_t one_vec = svdup_s32(1);
|
||||
|
||||
int32_t N = n;
|
||||
int32_t j = 0;
|
||||
svbool_t pg = svwhilelt_b32(j, N);
|
||||
int32_t active = svcntp_b32(svptrue_b32(), pg);
|
||||
svint32_t index_neg = svindex_s32(0, -1);
|
||||
svint32_t index = svindex_s32(0, 1);
|
||||
do {
|
||||
offset = posX - posY;
|
||||
svint32_t vec_off = svdup_s32(offset);
|
||||
svbool_t cmp = svcmpgt(pg, vec_off, index_neg);
|
||||
|
||||
svint32_t temp = svadd_z(pg, posX_vec, index);
|
||||
svint32_t temp1 = svmla_z(pg, temp, posY_vec, lda_vec);
|
||||
svint32_t temp2 = svmla_z(pg, posY_vec, temp, lda);
|
||||
svint32_t gat_ind = svsel(cmp, temp1, temp2);
|
||||
|
||||
i = m;
|
||||
while (i>0) {
|
||||
svfloat32_t data_vec = svld1_gather_index(pg, a, gat_ind);
|
||||
|
||||
gat_ind = svadd_m(cmp, gat_ind, lda_vec);
|
||||
gat_ind = svadd_m(svnot_z(pg, cmp) , gat_ind, one_vec);
|
||||
|
||||
svst1(pg, b, data_vec);
|
||||
|
||||
b += active;
|
||||
offset --;
|
||||
vec_off = svsub_z(pg, vec_off, one_vec);
|
||||
cmp = svcmpgt(pg, vec_off, index_neg);
|
||||
|
||||
i--;
|
||||
}
|
||||
|
||||
posX += sve_size;
|
||||
posX_vec = svdup_s32(posX);
|
||||
j += sve_size;
|
||||
pg = svwhilelt_b32(j, N);
|
||||
active = svcntp_b32(svptrue_b32(), pg);
|
||||
} while (svptest_any(svptrue_b32(), pg));
|
||||
|
||||
#endif
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,143 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "common.h"
|
||||
#include <arm_sve.h>
|
||||
|
||||
int CNAME(BLASLONG m, BLASLONG n, FLOAT *a, BLASLONG lda, BLASLONG posX, BLASLONG posY, FLOAT *b){
|
||||
|
||||
BLASLONG i, offset;
|
||||
|
||||
#if defined(DOUBLE)
|
||||
uint64_t sve_size = svcntd();
|
||||
svint64_t posY_vec = svdup_s64(posY);
|
||||
svint64_t posX_vec = svdup_s64(posX);
|
||||
svint64_t lda_vec = svdup_s64(lda);
|
||||
svint64_t one_vec = svdup_s64(1LL);
|
||||
|
||||
int64_t j = 0;
|
||||
svbool_t pg = svwhilelt_b64(j, n);
|
||||
int64_t active = svcntp_b64(svptrue_b64(), pg);
|
||||
svint64_t index_neg = svindex_s64(0LL, -1LL);
|
||||
svint64_t index = svindex_s64(0LL, 1LL);
|
||||
do {
|
||||
offset = posX - posY;
|
||||
svint64_t vec_off = svdup_s64(offset);
|
||||
svbool_t cmp = svcmpgt(pg, vec_off, index_neg);
|
||||
|
||||
svint64_t temp = svadd_z(pg, posX_vec, index);
|
||||
svint64_t temp1 = svmla_z(pg, temp, posY_vec, lda_vec);
|
||||
svint64_t temp2 = svmla_z(pg, posY_vec, temp, lda);
|
||||
svint64_t gat_ind = svsel(cmp, temp2, temp1);
|
||||
|
||||
i = m;
|
||||
while (i>0) {
|
||||
svfloat64_t data_vec = svld1_gather_index(pg, a, gat_ind);
|
||||
|
||||
gat_ind = svadd_m(cmp, gat_ind, one_vec);
|
||||
gat_ind = svadd_m(svnot_z(pg, cmp) , gat_ind, lda_vec);
|
||||
|
||||
svst1(pg, b, data_vec);
|
||||
|
||||
b += active;
|
||||
offset --;
|
||||
vec_off = svsub_z(pg, vec_off, one_vec);
|
||||
cmp = svcmpgt(pg, vec_off, index_neg);
|
||||
|
||||
i--;
|
||||
}
|
||||
|
||||
posX += sve_size;
|
||||
posX_vec = svdup_s64(posX);
|
||||
j += sve_size;
|
||||
pg = svwhilelt_b64(j, n);
|
||||
active = svcntp_b64(svptrue_b64(), pg);
|
||||
} while (svptest_any(svptrue_b64(), pg));
|
||||
|
||||
#else
|
||||
uint32_t sve_size = svcntw();
|
||||
svint32_t posY_vec = svdup_s32(posY);
|
||||
svint32_t posX_vec = svdup_s32(posX);
|
||||
svint32_t lda_vec = svdup_s32(lda);
|
||||
svint32_t one_vec = svdup_s32(1);
|
||||
|
||||
int32_t N = n;
|
||||
int32_t j = 0;
|
||||
svbool_t pg = svwhilelt_b32(j, N);
|
||||
int32_t active = svcntp_b32(svptrue_b32(), pg);
|
||||
svint32_t index_neg = svindex_s32(0, -1);
|
||||
svint32_t index = svindex_s32(0, 1);
|
||||
do {
|
||||
offset = posX - posY;
|
||||
svint32_t vec_off = svdup_s32(offset);
|
||||
svbool_t cmp = svcmpgt(pg, vec_off, index_neg);
|
||||
|
||||
svint32_t temp = svadd_z(pg, posX_vec, index);
|
||||
svint32_t temp1 = svmla_z(pg, temp, posY_vec, lda_vec);
|
||||
svint32_t temp2 = svmla_z(pg, posY_vec, temp, lda);
|
||||
svint32_t gat_ind = svsel(cmp, temp2, temp1);
|
||||
|
||||
i = m;
|
||||
while (i>0) {
|
||||
svfloat32_t data_vec = svld1_gather_index(pg, a, gat_ind);
|
||||
|
||||
gat_ind = svadd_m(cmp, gat_ind, one_vec);
|
||||
gat_ind = svadd_m(svnot_z(pg, cmp) , gat_ind, lda_vec);
|
||||
|
||||
svst1(pg, b, data_vec);
|
||||
|
||||
b += active;
|
||||
offset --;
|
||||
vec_off = svsub_z(pg, vec_off, one_vec);
|
||||
cmp = svcmpgt(pg, vec_off, index_neg);
|
||||
|
||||
i--;
|
||||
}
|
||||
|
||||
posX += sve_size;
|
||||
posX_vec = svdup_s32(posX);
|
||||
j += sve_size;
|
||||
pg = svwhilelt_b32(j, N);
|
||||
active = svcntp_b32(svptrue_b32(), pg);
|
||||
} while (svptest_any(svptrue_b32(), pg));
|
||||
|
||||
#endif
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,136 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "common.h"
|
||||
|
||||
#ifdef __ARM_FEATURE_SVE
|
||||
#include <arm_sve.h>
|
||||
#endif
|
||||
|
||||
int CNAME(BLASLONG m, BLASLONG n, FLOAT *a, BLASLONG lda, BLASLONG posX, BLASLONG posY, FLOAT *b){
|
||||
|
||||
BLASLONG i, js;
|
||||
BLASLONG X;
|
||||
|
||||
js = 0;
|
||||
FLOAT *ao;
|
||||
#ifdef DOUBLE
|
||||
svint64_t index = svindex_s64(0LL, lda);
|
||||
svbool_t pn = svwhilelt_b64(js, n);
|
||||
int n_active = svcntp_b64(svptrue_b64(), pn);
|
||||
#else
|
||||
svint32_t index = svindex_s32(0, lda);
|
||||
svbool_t pn = svwhilelt_b32(js, n);
|
||||
int n_active = svcntp_b32(svptrue_b32(), pn);
|
||||
#endif
|
||||
do
|
||||
{
|
||||
X = posX;
|
||||
|
||||
if (posX <= posY) {
|
||||
ao = a + posY + posX * lda;
|
||||
} else {
|
||||
ao = a + posX + posY * lda;
|
||||
}
|
||||
|
||||
i = 0;
|
||||
do
|
||||
{
|
||||
if (X > posY) {
|
||||
#ifdef DOUBLE
|
||||
svfloat64_t aj_vec = svld1_gather_index(pn, ao, index);
|
||||
#else
|
||||
svfloat32_t aj_vec = svld1_gather_index(pn, ao, index);
|
||||
#endif
|
||||
svst1(pn, b, aj_vec);
|
||||
ao ++;
|
||||
b += n_active;
|
||||
X ++;
|
||||
i ++;
|
||||
} else
|
||||
if (X < posY) {
|
||||
ao += lda;
|
||||
b += n_active;
|
||||
X ++;
|
||||
i ++;
|
||||
} else {
|
||||
/* I did not find a way to unroll this while preserving vector-length-agnostic code. */
|
||||
#ifdef UNIT
|
||||
int temp = 0;
|
||||
for (int j = 0; j < n_active; j++) {
|
||||
for (int k = 0 ; k < j; k++) {
|
||||
b[temp++] = *(ao+k*lda+j);
|
||||
}
|
||||
b[temp++] = ONE;
|
||||
for (int k = j+1; k < n_active; k++) {
|
||||
b[temp++] = ZERO;
|
||||
}
|
||||
}
|
||||
#else
|
||||
int temp = 0;
|
||||
for (int j = 0; j < n_active; j++) {
|
||||
for (int k = 0 ; k <= j; k++) {
|
||||
b[temp++] = *(ao+k*lda+j);
|
||||
}
|
||||
for (int k = j+1; k < n_active; k++) {
|
||||
b[temp++] = ZERO;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
ao += n_active;
|
||||
b += n_active*n_active;
|
||||
X += n_active;
|
||||
i += n_active;
|
||||
}
|
||||
} while (i < m);
|
||||
|
||||
posY += n_active;
|
||||
js += n_active;
|
||||
#ifdef DOUBLE
|
||||
pn = svwhilelt_b64(js, n);
|
||||
n_active = svcntp_b64(svptrue_b64(), pn);
|
||||
} while (svptest_any(svptrue_b64(), pn));
|
||||
#else
|
||||
pn = svwhilelt_b32(js, n);
|
||||
n_active = svcntp_b32(svptrue_b32(), pn);
|
||||
} while (svptest_any(svptrue_b32(), pn));
|
||||
#endif
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,136 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "common.h"
|
||||
|
||||
#ifdef __ARM_FEATURE_SVE
|
||||
#include <arm_sve.h>
|
||||
#endif
|
||||
|
||||
int CNAME(BLASLONG m, BLASLONG n, FLOAT *a, BLASLONG lda, BLASLONG posX, BLASLONG posY, FLOAT *b){
|
||||
|
||||
BLASLONG i, js;
|
||||
BLASLONG X;
|
||||
|
||||
FLOAT *ao;
|
||||
js = 0;
|
||||
#ifdef DOUBLE
|
||||
svbool_t pn = svwhilelt_b64(js, n);
|
||||
int n_active = svcntp_b64(svptrue_b64(), pn);
|
||||
#else
|
||||
svbool_t pn = svwhilelt_b32(js, n);
|
||||
int n_active = svcntp_b32(svptrue_b32(), pn);
|
||||
#endif
|
||||
do
|
||||
{
|
||||
X = posX;
|
||||
|
||||
if (posX <= posY) {
|
||||
ao = a + posY + posX * lda;
|
||||
} else {
|
||||
ao = a + posX + posY * lda;
|
||||
}
|
||||
|
||||
i = 0;
|
||||
do
|
||||
{
|
||||
if (X > posY) {
|
||||
ao ++;
|
||||
b += n_active;
|
||||
X ++;
|
||||
i ++;
|
||||
} else
|
||||
if (X < posY) {
|
||||
#ifdef DOUBLE
|
||||
svfloat64_t aj_vec = svld1(pn, ao);
|
||||
#else
|
||||
svfloat32_t aj_vec = svld1(pn, ao);
|
||||
#endif
|
||||
svst1(pn, b, aj_vec);
|
||||
ao += lda;
|
||||
b += n_active;
|
||||
X ++;
|
||||
i ++;
|
||||
} else {
|
||||
/* I did not find a way to unroll this while preserving vector-length-agnostic code. */
|
||||
#ifdef UNIT
|
||||
int temp = 0;
|
||||
for (int j = 0; j < n_active; j++) {
|
||||
for (int k = 0 ; k < j; k++) {
|
||||
b[temp++] = ZERO;
|
||||
}
|
||||
b[temp++] = ONE;
|
||||
for (int k = j+1; k < n_active; k++) {
|
||||
b[temp++] = *(ao+j*lda+k);
|
||||
}
|
||||
}
|
||||
#else
|
||||
int temp = 0;
|
||||
for (int j = 0; j < n_active; j++) {
|
||||
for (int k = 0 ; k < j; k++) {
|
||||
b[temp++] = ZERO;
|
||||
}
|
||||
for (int k = j; k < n_active; k++) {
|
||||
b[temp++] = *(ao+j*lda+k);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
ao += n_active * lda;
|
||||
b += n_active*n_active;
|
||||
X += n_active;
|
||||
i += n_active;
|
||||
}
|
||||
} while (i < m);
|
||||
|
||||
|
||||
posY += n_active;
|
||||
js += n_active;
|
||||
#ifdef DOUBLE
|
||||
pn = svwhilelt_b64(js, n);
|
||||
n_active = svcntp_b64(svptrue_b64(), pn);
|
||||
} while (svptest_any(svptrue_b64(), pn));
|
||||
#else
|
||||
pn = svwhilelt_b32(js, n);
|
||||
n_active = svcntp_b32(svptrue_b32(), pn);
|
||||
} while (svptest_any(svptrue_b32(), pn));
|
||||
#endif
|
||||
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,136 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "common.h"
|
||||
|
||||
#ifdef __ARM_FEATURE_SVE
|
||||
#include <arm_sve.h>
|
||||
#endif
|
||||
|
||||
int CNAME(BLASLONG m, BLASLONG n, FLOAT *a, BLASLONG lda, BLASLONG posX, BLASLONG posY, FLOAT *b){
|
||||
|
||||
BLASLONG i, js;
|
||||
BLASLONG X;
|
||||
|
||||
js = 0;
|
||||
FLOAT *ao;
|
||||
#ifdef DOUBLE
|
||||
svint64_t index = svindex_s64(0LL, lda);
|
||||
svbool_t pn = svwhilelt_b64(js, n);
|
||||
int n_active = svcntp_b64(svptrue_b64(), pn);
|
||||
#else
|
||||
svint32_t index = svindex_s32(0, lda);
|
||||
svbool_t pn = svwhilelt_b32(js, n);
|
||||
int n_active = svcntp_b32(svptrue_b32(), pn);
|
||||
#endif
|
||||
do
|
||||
{
|
||||
X = posX;
|
||||
|
||||
if (posX <= posY) {
|
||||
ao = a + posX + posY * lda;
|
||||
} else {
|
||||
ao = a + posY + posX * lda;
|
||||
}
|
||||
|
||||
i = 0;
|
||||
do
|
||||
{
|
||||
if (X < posY) {
|
||||
#ifdef DOUBLE
|
||||
svfloat64_t aj_vec = svld1_gather_index(pn, ao, index);
|
||||
#else
|
||||
svfloat32_t aj_vec = svld1_gather_index(pn, ao, index);
|
||||
#endif
|
||||
svst1(pn, b, aj_vec);
|
||||
ao ++;
|
||||
b += n_active;
|
||||
X ++;
|
||||
i ++;
|
||||
} else
|
||||
if (X > posY) {
|
||||
ao += lda;
|
||||
b += n_active;
|
||||
X ++;
|
||||
i ++;
|
||||
} else {
|
||||
/* I did not find a way to unroll this while preserving vector-length-agnostic code. */
|
||||
#ifdef UNIT
|
||||
int temp = 0;
|
||||
for (int j = 0; j < n_active; j++) {
|
||||
for (int k = 0 ; k < j; k++) {
|
||||
b[temp++] = ZERO;
|
||||
}
|
||||
b[temp++] = ONE;
|
||||
for (int k = j+1; k < n_active; k++) {
|
||||
b[temp++] = *(ao+k*lda+j);
|
||||
}
|
||||
}
|
||||
#else
|
||||
int temp = 0;
|
||||
for (int j = 0; j < n_active; j++) {
|
||||
for (int k = 0 ; k < j; k++) {
|
||||
b[temp++] = ZERO;
|
||||
}
|
||||
for (int k = j; k < n_active; k++) {
|
||||
b[temp++] = *(ao+k*lda+j);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
ao += n_active;
|
||||
b += n_active*n_active;
|
||||
X += n_active;
|
||||
i += n_active;
|
||||
}
|
||||
} while (i < m);
|
||||
|
||||
posY += n_active;
|
||||
js += n_active;
|
||||
#ifdef DOUBLE
|
||||
pn = svwhilelt_b64(js, n);
|
||||
n_active = svcntp_b64(svptrue_b64(), pn);
|
||||
} while (svptest_any(svptrue_b64(), pn));
|
||||
#else
|
||||
pn = svwhilelt_b32(js, n);
|
||||
n_active = svcntp_b32(svptrue_b32(), pn);
|
||||
} while (svptest_any(svptrue_b32(), pn));
|
||||
#endif
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,134 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "common.h"
|
||||
|
||||
#ifdef __ARM_FEATURE_SVE
|
||||
#include <arm_sve.h>
|
||||
#endif
|
||||
|
||||
int CNAME(BLASLONG m, BLASLONG n, FLOAT *a, BLASLONG lda, BLASLONG posX, BLASLONG posY, FLOAT *b){
|
||||
|
||||
BLASLONG i, js;
|
||||
BLASLONG X;
|
||||
|
||||
FLOAT *ao;
|
||||
js = 0;
|
||||
#ifdef DOUBLE
|
||||
svbool_t pn = svwhilelt_b64(js, n);
|
||||
int n_active = svcntp_b64(svptrue_b64(), pn);
|
||||
#else
|
||||
svbool_t pn = svwhilelt_b32(js, n);
|
||||
int n_active = svcntp_b32(svptrue_b32(), pn);
|
||||
#endif
|
||||
do
|
||||
{
|
||||
X = posX;
|
||||
|
||||
if (posX <= posY) {
|
||||
ao = a + posX + posY * lda;
|
||||
} else {
|
||||
ao = a + posY + posX * lda;
|
||||
}
|
||||
|
||||
i = 0;
|
||||
do
|
||||
{
|
||||
if (X < posY) {
|
||||
ao ++;
|
||||
b += n_active;
|
||||
X ++;
|
||||
i ++;
|
||||
} else
|
||||
if (X > posY) {
|
||||
#ifdef DOUBLE
|
||||
svfloat64_t aj_vec = svld1(pn, ao);
|
||||
#else
|
||||
svfloat32_t aj_vec = svld1(pn, ao);
|
||||
#endif
|
||||
svst1(pn, b, aj_vec);
|
||||
ao += lda;
|
||||
b += n_active;
|
||||
X ++;
|
||||
i ++;
|
||||
} else {
|
||||
/* I did not find a way to unroll this while preserving vector-length-agnostic code. */
|
||||
#ifdef UNIT
|
||||
int temp = 0;
|
||||
for (int j = 0; j < n_active; j++) {
|
||||
for (int k = 0 ; k < j; k++) {
|
||||
b[temp++] = *(ao+j*lda+k);
|
||||
}
|
||||
b[temp++] = ONE;
|
||||
for (int k = j+1; k < n_active; k++) {
|
||||
b[temp++] = ZERO;
|
||||
}
|
||||
}
|
||||
#else
|
||||
int temp = 0;
|
||||
for (int j = 0; j < n_active; j++) {
|
||||
for (int k = 0 ; k <= j; k++) {
|
||||
b[temp++] = *(ao+j*lda+k);
|
||||
}
|
||||
for (int k = j+1; k < n_active; k++) {
|
||||
b[temp++] = ZERO;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
ao += n_active * lda;
|
||||
b += n_active*n_active;
|
||||
X += n_active;
|
||||
i += n_active;
|
||||
}
|
||||
} while (i < m);
|
||||
|
||||
posY += n_active;
|
||||
js += n_active;
|
||||
#ifdef DOUBLE
|
||||
pn = svwhilelt_b64(js, n);
|
||||
n_active = svcntp_b64(svptrue_b64(), pn);
|
||||
} while (svptest_any(svptrue_b64(), pn));
|
||||
#else
|
||||
pn = svwhilelt_b32(js, n);
|
||||
n_active = svcntp_b32(svptrue_b32(), pn);
|
||||
} while (svptest_any(svptrue_b32(), pn));
|
||||
#endif
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -48,8 +48,8 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define pCRow2 x14
|
||||
#define pCRow3 x15
|
||||
#define pA x16
|
||||
#define alphaR x17
|
||||
#define alphaI x18
|
||||
#define alphaR x19
|
||||
#define alphaI x20
|
||||
|
||||
#define alpha0_R d10
|
||||
#define alphaV0_R v10.d[0]
|
||||
|
||||
@@ -0,0 +1,736 @@
|
||||
/***************************************************************************
|
||||
Copyright (c) 2021, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written permission.
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A00 PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
|
||||
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*****************************************************************************/
|
||||
|
||||
#include "common.h"
|
||||
#include <arm_neon.h>
|
||||
|
||||
/*******************************************************************************
|
||||
The complex GEMM kernels in OpenBLAS use static configuration of conjugation
|
||||
modes via specific macros:
|
||||
|
||||
MACRO_NAME | conjugation on matrix A | conjugation on matrix B |
|
||||
---------- | ----------------------- | ----------------------- |
|
||||
NN/NT/TN/TT | No | No |
|
||||
NR/NC/TR/TC | No | Yes |
|
||||
RN/RT/CN/CT | Yes | No |
|
||||
RR/RC/CR/CC | Yes | Yes |
|
||||
|
||||
"conjugation on matrix A" means the complex conjugates of elements from
|
||||
matrix A are used for matmul (rather than the original elements). "conjugation
|
||||
on matrix B" means the complex conjugate of each element from matrix B is taken
|
||||
for matrix multiplication, respectively.
|
||||
|
||||
Complex numbers in arrays or matrices are usually packed together as an
|
||||
array of struct (without padding):
|
||||
struct complex_number {
|
||||
FLOAT real_part;
|
||||
FLOAT imag_part;
|
||||
};
|
||||
|
||||
For a double complex array ARR[] which is usually DEFINED AS AN ARRAY OF
|
||||
DOUBLE, the real part of its Kth complex number can be accessed as
|
||||
ARR[K * 2], the imaginary part of the Kth complex number is ARR[2 * K + 1].
|
||||
|
||||
This file uses 2 ways to vectorize matrix multiplication of complex numbers:
|
||||
|
||||
(1) Expanded-form
|
||||
|
||||
During accumulation along direction K:
|
||||
|
||||
Σk(a[0][k].real b[k][n].real)
|
||||
accumulate Σk(a[0][k].imag b[k][n].real)
|
||||
-------------------> .
|
||||
| * b[k][n].real .
|
||||
| (broadcasted) .
|
||||
a[0][k].real Σk(a[v-1][k].real b[k][n].real)
|
||||
a[0][k].imag Σk(a[v-1][k].imag b[k][n].real)
|
||||
. VECTOR I
|
||||
(vec_a) .
|
||||
.
|
||||
a[v-1][k].real Σk(a[0][k].real b[k][n].imag)
|
||||
a[v-1][k].imag Σk(a[0][k].imag b[k][n].imag)
|
||||
| .
|
||||
| accumulate .
|
||||
-------------------> .
|
||||
* b[k][n].imag Σk(a[v-1][k].real b[k][n].imag)
|
||||
(broadcasted) Σk(a[v-1][k].imag b[k][n].imag)
|
||||
VECTOR II
|
||||
|
||||
After accumulation, prior to storage:
|
||||
|
||||
-1 -Σk(a[0][k].imag b[k][n].imag)
|
||||
1 Σk(a[0][k].real b[k][n].imag)
|
||||
. .
|
||||
VECTOR II permute and multiply . to get .
|
||||
. .
|
||||
-1 -Σk(a[v-1][k].imag b[k][n].imag)
|
||||
1 Σk(a[v-1][k].real b[k][n].imag)
|
||||
|
||||
then add with VECTOR I to get the result vector of elements of C.
|
||||
|
||||
2 vector registers are needed for every v elements of C, with
|
||||
v == sizeof(vector) / sizeof(complex)
|
||||
|
||||
(2) Contracted-form
|
||||
|
||||
During accumulation along direction K:
|
||||
|
||||
(the K coordinate is not shown, since the operation is identical for each k)
|
||||
|
||||
(load vector in mem) (load vector in mem)
|
||||
a[0].r a[0].i ... a[v-1].r a[v-1].i a[v].r a[v].i ... a[2v-1].r a[2v-1]i
|
||||
| |
|
||||
| unzip operation (or VLD2 in arm neon) |
|
||||
-----------------------------------------------------
|
||||
|
|
||||
|
|
||||
--------------------------------------------------
|
||||
| |
|
||||
| |
|
||||
v v
|
||||
a[0].real ... a[2v-1].real a[0].imag ... a[2v-1].imag
|
||||
| | | |
|
||||
| | * b[i].imag(broadcast) | |
|
||||
* b[i].real | -----------------------------|---- | * b[i].real
|
||||
(broadcast) | | | | (broadcast)
|
||||
| ------------------------------ | |
|
||||
+ | - | * b[i].imag(broadcast) + | + |
|
||||
v v v v
|
||||
(accumulate) (accumulate)
|
||||
c[0].real ... c[2v-1].real c[0].imag ... c[2v-1].imag
|
||||
VECTOR_REAL VECTOR_IMAG
|
||||
|
||||
After accumulation, VECTOR_REAL and VECTOR_IMAG are zipped (interleaved)
|
||||
then stored to matrix C directly.
|
||||
|
||||
For 2v elements of C, only 2 vector registers are needed, while
|
||||
4 registers are required for expanded-form.
|
||||
(v == sizeof(vector) / sizeof(complex))
|
||||
|
||||
For AArch64 zgemm, 4x4 kernel needs 32 128-bit NEON registers
|
||||
to store elements of C when using expanded-form calculation, where
|
||||
the register spilling will occur. So contracted-form operation is
|
||||
selected for 4x4 kernel. As for all other combinations of unroll parameters
|
||||
(2x4, 4x2, 2x2, and so on), expanded-form mode is used to bring more
|
||||
NEON registers into usage to hide latency of multiply-add instructions.
|
||||
******************************************************************************/
|
||||
|
||||
static inline float64x2_t set_f64x2(double lo, double hi) {
|
||||
float64x2_t ret = vdupq_n_f64(0);
|
||||
ret = vsetq_lane_f64(lo, ret, 0);
|
||||
ret = vsetq_lane_f64(hi, ret, 1);
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline float64x2x2_t expand_alpha(double alpha_r, double alpha_i) {
|
||||
float64x2x2_t ret = {{ set_f64x2(alpha_r, alpha_i), set_f64x2(-alpha_i, alpha_r) }};
|
||||
return ret;
|
||||
}
|
||||
|
||||
/*****************************************************************
|
||||
* operation: *c += alpha * c_value //complex multiplication
|
||||
* expanded_alpha: { { alpha_r, alpha_i }, { -alpha_i, alpha_r }
|
||||
* expanded_c: {{ arbr, aibr }, { arbi, aibi }}
|
||||
****************************************************************/
|
||||
static inline void store_1c(double *c, float64x2x2_t expanded_c,
|
||||
float64x2x2_t expanded_alpha) {
|
||||
float64x2_t ld = vld1q_f64(c);
|
||||
#if defined(NN) || defined(NT) || defined(TN) || defined(TT)
|
||||
double real = vgetq_lane_f64(expanded_c.val[0], 0) - vgetq_lane_f64(expanded_c.val[1], 1);
|
||||
double imag = vgetq_lane_f64(expanded_c.val[0], 1) + vgetq_lane_f64(expanded_c.val[1], 0);
|
||||
#elif defined(NR) || defined(NC) || defined(TR) || defined(TC)
|
||||
double real = vgetq_lane_f64(expanded_c.val[0], 0) + vgetq_lane_f64(expanded_c.val[1], 1);
|
||||
double imag = vgetq_lane_f64(expanded_c.val[0], 1) - vgetq_lane_f64(expanded_c.val[1], 0);
|
||||
#elif defined(RN) || defined(RT) || defined(CN) || defined(CT)
|
||||
double real = vgetq_lane_f64(expanded_c.val[0], 0) + vgetq_lane_f64(expanded_c.val[1], 1);
|
||||
double imag = -vgetq_lane_f64(expanded_c.val[0], 1) + vgetq_lane_f64(expanded_c.val[1], 0);
|
||||
#else
|
||||
double real = vgetq_lane_f64(expanded_c.val[0], 0) - vgetq_lane_f64(expanded_c.val[1], 1);
|
||||
double imag = -vgetq_lane_f64(expanded_c.val[0], 1) - vgetq_lane_f64(expanded_c.val[1], 0);
|
||||
#endif
|
||||
ld = vfmaq_n_f64(ld, expanded_alpha.val[0], real);
|
||||
vst1q_f64(c, vfmaq_n_f64(ld, expanded_alpha.val[1], imag));
|
||||
}
|
||||
|
||||
static inline void pref_c_4(const double *c) {
|
||||
__asm__ __volatile__("prfm pstl1keep,[%0]; prfm pstl1keep,[%0,#56]\n\t"::"r"(c):);
|
||||
}
|
||||
|
||||
static inline float64x2x2_t add_ec(float64x2x2_t ec1, float64x2x2_t ec2) {
|
||||
float64x2x2_t ret = {{ vaddq_f64(ec1.val[0], ec2.val[0]),
|
||||
vaddq_f64(ec1.val[1], ec2.val[1]) }};
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline float64x2x2_t update_ec(float64x2x2_t ec, float64x2_t a, float64x2_t b) {
|
||||
float64x2x2_t ret = {{ vfmaq_laneq_f64(ec.val[0], a, b, 0), vfmaq_laneq_f64(ec.val[1], a, b, 1) }};
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline float64x2x2_t init() {
|
||||
float64x2x2_t ret = {{ vdupq_n_f64(0), vdupq_n_f64(0) }};
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline void kernel_1x1(const double *sa, const double *sb, double *C,
|
||||
BLASLONG K, double alphar, double alphai) {
|
||||
|
||||
const float64x2x2_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
float64x2x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init();
|
||||
|
||||
for (; K > 3; K -= 4) {
|
||||
float64x2_t a1 = vld1q_f64(sa), a2 = vld1q_f64(sa + 2),
|
||||
a3 = vld1q_f64(sa + 4), a4 = vld1q_f64(sa + 6); sa += 8;
|
||||
float64x2_t b1 = vld1q_f64(sb), b2 = vld1q_f64(sb + 2),
|
||||
b3 = vld1q_f64(sb + 4), b4 = vld1q_f64(sb + 6); sb += 8;
|
||||
c1 = update_ec(c1, a1, b1);
|
||||
c2 = update_ec(c2, a2, b2);
|
||||
c3 = update_ec(c3, a3, b3);
|
||||
c4 = update_ec(c4, a4, b4);
|
||||
}
|
||||
c1 = add_ec(c1, c2);
|
||||
c3 = add_ec(c3, c4);
|
||||
c1 = add_ec(c1, c3);
|
||||
for (; K; K--) {
|
||||
c1 = update_ec(c1, vld1q_f64(sa), vld1q_f64(sb)); sa += 2; sb += 2;
|
||||
}
|
||||
store_1c(C, c1, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_2x1(const double *sa, const double *sb, double *C,
|
||||
BLASLONG K, double alphar, double alphai) {
|
||||
|
||||
const float64x2x2_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
float64x2x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init();
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float64x2_t a1 = vld1q_f64(sa), a2 = vld1q_f64(sa + 2),
|
||||
a3 = vld1q_f64(sa + 4), a4 = vld1q_f64(sa + 6); sa += 8;
|
||||
float64x2_t b1 = vld1q_f64(sb), b2 = vld1q_f64(sb + 2); sb += 4;
|
||||
c1 = update_ec(c1, a1, b1);
|
||||
c2 = update_ec(c2, a2, b1);
|
||||
c3 = update_ec(c3, a3, b2);
|
||||
c4 = update_ec(c4, a4, b2);
|
||||
}
|
||||
c1 = add_ec(c1, c3);
|
||||
c2 = add_ec(c2, c4);
|
||||
if (K) {
|
||||
float64x2_t b1 = vld1q_f64(sb);
|
||||
c1 = update_ec(c1, vld1q_f64(sa), b1);
|
||||
c2 = update_ec(c2, vld1q_f64(sa + 2), b1);
|
||||
}
|
||||
store_1c(C, c1, expanded_alpha);
|
||||
store_1c(C + 2, c2, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_1x2(const double *sa, const double *sb, double *C,
|
||||
BLASLONG LDC, BLASLONG K, double alphar, double alphai) {
|
||||
|
||||
const float64x2x2_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
float64x2x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init();
|
||||
|
||||
for (; K > 1; K -= 2) {
|
||||
float64x2_t a1 = vld1q_f64(sa), a2 = vld1q_f64(sa + 2); sa += 4;
|
||||
float64x2_t b1 = vld1q_f64(sb), b2 = vld1q_f64(sb + 2),
|
||||
b3 = vld1q_f64(sb + 4), b4 = vld1q_f64(sb + 6); sb += 8;
|
||||
c1 = update_ec(c1, a1, b1);
|
||||
c2 = update_ec(c2, a1, b2);
|
||||
c3 = update_ec(c3, a2, b3);
|
||||
c4 = update_ec(c4, a2, b4);
|
||||
}
|
||||
c1 = add_ec(c1, c3);
|
||||
c2 = add_ec(c2, c4);
|
||||
if (K) {
|
||||
float64x2_t a1 = vld1q_f64(sa);
|
||||
c1 = update_ec(c1, a1, vld1q_f64(sb));
|
||||
c2 = update_ec(c2, a1, vld1q_f64(sb + 2));
|
||||
}
|
||||
store_1c(C, c1, expanded_alpha);
|
||||
store_1c(C + LDC * 2, c2, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_2x2(const double *sa, const double *sb, double *C,
|
||||
BLASLONG LDC, BLASLONG K, double alphar, double alphai) {
|
||||
|
||||
const float64x2x2_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
float64x2x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init();
|
||||
|
||||
for (; K; K--) {
|
||||
float64x2_t a1 = vld1q_f64(sa), a2 = vld1q_f64(sa + 2); sa += 4;
|
||||
float64x2_t b1 = vld1q_f64(sb), b2 = vld1q_f64(sb + 2); sb += 4;
|
||||
c1 = update_ec(c1, a1, b1);
|
||||
c2 = update_ec(c2, a2, b1);
|
||||
c3 = update_ec(c3, a1, b2);
|
||||
c4 = update_ec(c4, a2, b2);
|
||||
}
|
||||
store_1c(C, c1, expanded_alpha);
|
||||
store_1c(C + 2, c2, expanded_alpha); C += LDC * 2;
|
||||
store_1c(C, c3, expanded_alpha);
|
||||
store_1c(C + 2, c4, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_4x1(const double *sa, const double *sb, double *C,
|
||||
BLASLONG K, double alphar, double alphai) {
|
||||
|
||||
const float64x2x2_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
float64x2x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init();
|
||||
pref_c_4(C);
|
||||
|
||||
for (; K; K--) {
|
||||
float64x2_t b1 = vld1q_f64(sb); sb += 2;
|
||||
c1 = update_ec(c1, vld1q_f64(sa), b1);
|
||||
c2 = update_ec(c2, vld1q_f64(sa + 2), b1);
|
||||
c3 = update_ec(c3, vld1q_f64(sa + 4), b1);
|
||||
c4 = update_ec(c4, vld1q_f64(sa + 6), b1);
|
||||
sa += 8;
|
||||
}
|
||||
store_1c(C, c1, expanded_alpha);
|
||||
store_1c(C + 2, c2, expanded_alpha);
|
||||
store_1c(C + 4, c3, expanded_alpha);
|
||||
store_1c(C + 6, c4, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_4x2(const double *sa, const double *sb, double *C,
|
||||
BLASLONG LDC, BLASLONG K, double alphar, double alphai) {
|
||||
|
||||
const float64x2x2_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
float64x2x2_t c1, c2, c3, c4, c5, c6, c7, c8;
|
||||
c1 = c2 = c3 = c4 = c5 = c6 = c7 = c8 = init();
|
||||
pref_c_4(C);
|
||||
pref_c_4(C + LDC * 2);
|
||||
|
||||
for (; K; K--) {
|
||||
float64x2_t b1 = vld1q_f64(sb), b2 = vld1q_f64(sb + 2); sb += 4;
|
||||
float64x2_t a1 = vld1q_f64(sa), a2 = vld1q_f64(sa + 2),
|
||||
a3 = vld1q_f64(sa + 4), a4 = vld1q_f64(sa + 6); sa += 8;
|
||||
c1 = update_ec(c1, a1, b1);
|
||||
c2 = update_ec(c2, a2, b1);
|
||||
c3 = update_ec(c3, a3, b1);
|
||||
c4 = update_ec(c4, a4, b1);
|
||||
c5 = update_ec(c5, a1, b2);
|
||||
c6 = update_ec(c6, a2, b2);
|
||||
c7 = update_ec(c7, a3, b2);
|
||||
c8 = update_ec(c8, a4, b2);
|
||||
}
|
||||
store_1c(C, c1, expanded_alpha);
|
||||
store_1c(C + 2, c2, expanded_alpha);
|
||||
store_1c(C + 4, c3, expanded_alpha);
|
||||
store_1c(C + 6, c4, expanded_alpha); C += LDC * 2;
|
||||
store_1c(C, c5, expanded_alpha);
|
||||
store_1c(C + 2, c6, expanded_alpha);
|
||||
store_1c(C + 4, c7, expanded_alpha);
|
||||
store_1c(C + 6, c8, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_1x4(const double *sa, const double *sb, double *C,
|
||||
BLASLONG LDC, BLASLONG K, double alphar, double alphai) {
|
||||
|
||||
const float64x2x2_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
float64x2x2_t c1, c2, c3, c4;
|
||||
c1 = c2 = c3 = c4 = init();
|
||||
|
||||
for (; K; K--) {
|
||||
float64x2_t a1 = vld1q_f64(sa); sa += 2;
|
||||
c1 = update_ec(c1, a1, vld1q_f64(sb));
|
||||
c2 = update_ec(c2, a1, vld1q_f64(sb + 2));
|
||||
c3 = update_ec(c3, a1, vld1q_f64(sb + 4));
|
||||
c4 = update_ec(c4, a1, vld1q_f64(sb + 6));
|
||||
sb += 8;
|
||||
}
|
||||
store_1c(C, c1, expanded_alpha); C += LDC * 2;
|
||||
store_1c(C, c2, expanded_alpha); C += LDC * 2;
|
||||
store_1c(C, c3, expanded_alpha); C += LDC * 2;
|
||||
store_1c(C, c4, expanded_alpha);
|
||||
}
|
||||
|
||||
static inline void kernel_2x4(const double *sa, const double *sb, double *C,
|
||||
BLASLONG LDC, BLASLONG K, double alphar, double alphai) {
|
||||
|
||||
const float64x2x2_t expanded_alpha = expand_alpha(alphar, alphai);
|
||||
float64x2x2_t c1, c2, c3, c4, c5, c6, c7, c8;
|
||||
c1 = c2 = c3 = c4 = c5 = c6 = c7 = c8 = init();
|
||||
|
||||
for (; K; K--) {
|
||||
float64x2_t a1 = vld1q_f64(sa), a2 = vld1q_f64(sa + 2); sa += 4;
|
||||
float64x2_t b1 = vld1q_f64(sb), b2 = vld1q_f64(sb + 2),
|
||||
b3 = vld1q_f64(sb + 4), b4 = vld1q_f64(sb + 6); sb += 8;
|
||||
c1 = update_ec(c1, a1, b1);
|
||||
c2 = update_ec(c2, a2, b1);
|
||||
c3 = update_ec(c3, a1, b2);
|
||||
c4 = update_ec(c4, a2, b2);
|
||||
c5 = update_ec(c5, a1, b3);
|
||||
c6 = update_ec(c6, a2, b3);
|
||||
c7 = update_ec(c7, a1, b4);
|
||||
c8 = update_ec(c8, a2, b4);
|
||||
}
|
||||
store_1c(C, c1, expanded_alpha);
|
||||
store_1c(C + 2, c2, expanded_alpha); C += LDC * 2;
|
||||
store_1c(C, c3, expanded_alpha);
|
||||
store_1c(C + 2, c4, expanded_alpha); C += LDC * 2;
|
||||
store_1c(C, c5, expanded_alpha);
|
||||
store_1c(C + 2, c6, expanded_alpha); C += LDC * 2;
|
||||
store_1c(C, c7, expanded_alpha);
|
||||
store_1c(C + 2, c8, expanded_alpha);
|
||||
}
|
||||
|
||||
#if defined(NN) || defined(NT) || defined(TN) || defined(TT)
|
||||
#define FMLA_RI "fmla "
|
||||
#define FMLA_IR "fmla "
|
||||
#define FMLA_II "fmls "
|
||||
#elif defined(NR) || defined(NC) || defined(TR) || defined(TC)
|
||||
#define FMLA_RI "fmls "
|
||||
#define FMLA_IR "fmla "
|
||||
#define FMLA_II "fmla "
|
||||
#elif defined(RN) || defined(RT) || defined(CN) || defined(CT)
|
||||
#define FMLA_RI "fmla "
|
||||
#define FMLA_IR "fmls "
|
||||
#define FMLA_II "fmla "
|
||||
#else
|
||||
#define FMLA_RI "fmls "
|
||||
#define FMLA_IR "fmls "
|
||||
#define FMLA_II "fmls "
|
||||
#endif
|
||||
#define FMLA_RR "fmla "
|
||||
|
||||
static inline void store_4c(double *C, float64x2_t up_r, float64x2_t up_i,
|
||||
float64x2_t lo_r, float64x2_t lo_i, double alphar, double alphai) {
|
||||
float64x2x2_t up = vld2q_f64(C), lo = vld2q_f64(C + 4);
|
||||
up.val[0] = vfmaq_n_f64(up.val[0], up_r, alphar);
|
||||
up.val[1] = vfmaq_n_f64(up.val[1], up_r, alphai);
|
||||
lo.val[0] = vfmaq_n_f64(lo.val[0], lo_r, alphar);
|
||||
lo.val[1] = vfmaq_n_f64(lo.val[1], lo_r, alphai);
|
||||
up.val[0] = vfmsq_n_f64(up.val[0], up_i, alphai);
|
||||
up.val[1] = vfmaq_n_f64(up.val[1], up_i, alphar);
|
||||
lo.val[0] = vfmsq_n_f64(lo.val[0], lo_i, alphai);
|
||||
lo.val[1] = vfmaq_n_f64(lo.val[1], lo_i, alphar);
|
||||
vst2q_f64(C, up);
|
||||
vst2q_f64(C + 4, lo);
|
||||
}
|
||||
|
||||
static inline void kernel_4x4(const double *sa, const double *sb, double *C,
|
||||
BLASLONG LDC, BLASLONG K, double alphar, double alphai) {
|
||||
|
||||
float64x2_t c1r, c1i, c2r, c2i;
|
||||
float64x2_t c3r, c3i, c4r, c4i;
|
||||
float64x2_t c5r, c5i, c6r, c6i;
|
||||
float64x2_t c7r, c7i, c8r, c8i;
|
||||
|
||||
const double *pref_ = C;
|
||||
pref_c_4(pref_); pref_ += LDC * 2;
|
||||
pref_c_4(pref_); pref_ += LDC * 2;
|
||||
pref_c_4(pref_); pref_ += LDC * 2;
|
||||
pref_c_4(pref_);
|
||||
|
||||
__asm__ __volatile__(
|
||||
"cmp %[K],#0\n\t"
|
||||
"movi %[c1r].16b,#0; movi %[c1i].16b,#0; movi %[c2r].16b,#0; movi %[c2i].16b,#0\n\t"
|
||||
"movi %[c3r].16b,#0; movi %[c3i].16b,#0; movi %[c4r].16b,#0; movi %[c4i].16b,#0\n\t"
|
||||
"movi %[c5r].16b,#0; movi %[c5i].16b,#0; movi %[c6r].16b,#0; movi %[c6i].16b,#0\n\t"
|
||||
"movi %[c7r].16b,#0; movi %[c7i].16b,#0; movi %[c8r].16b,#0; movi %[c8i].16b,#0\n\t"
|
||||
"beq 4f; cmp %[K],#2\n\t"
|
||||
"ld2 {v0.2d,v1.2d},[%[sa]],#32; ldp q4,q5,[%[sb]],#32\n\t"
|
||||
"ld2 {v2.2d,v3.2d},[%[sa]],#32; ldr q6,[%[sb]]; ldr d7,[%[sb],#16]\n\t"
|
||||
"ldr x0,[%[sb],#24]; add %[sb],%[sb],#32\n\t"
|
||||
"beq 2f; blt 3f\n\t"
|
||||
"1:\n\t"
|
||||
"fmov v7.d[1],x0; ldr d8,[%[sa]]\n\t"
|
||||
FMLA_RR "%[c1r].2d,v0.2d,v4.d[0]; ldr x0,[%[sa],#16]\n\t"
|
||||
FMLA_RR "%[c2r].2d,v2.2d,v4.d[0]\n\t"
|
||||
FMLA_RI "%[c1i].2d,v0.2d,v4.d[1]\n\t"
|
||||
"fmov v8.d[1],x0; ldr d9,[%[sa],#8]\n\t"
|
||||
FMLA_RI "%[c2i].2d,v2.2d,v4.d[1]; ldr x0,[%[sa],#24]\n\t"
|
||||
FMLA_II "%[c1r].2d,v1.2d,v4.d[1]\n\t"
|
||||
FMLA_II "%[c2r].2d,v3.2d,v4.d[1]\n\t"
|
||||
"fmov v9.d[1],x0; ldr d10,[%[sa],#32]\n\t"
|
||||
FMLA_IR "%[c1i].2d,v1.2d,v4.d[0]; ldr x0,[%[sa],#48]\n\t"
|
||||
FMLA_IR "%[c2i].2d,v3.2d,v4.d[0]\n\t"
|
||||
FMLA_RR "%[c3r].2d,v0.2d,v5.d[0]\n\t"
|
||||
"fmov v10.d[1],x0; ldr d11,[%[sa],#40]\n\t"
|
||||
FMLA_RR "%[c4r].2d,v2.2d,v5.d[0]; ldr x0,[%[sa],#56]\n\t"
|
||||
FMLA_RI "%[c3i].2d,v0.2d,v5.d[1]\n\t"
|
||||
FMLA_RI "%[c4i].2d,v2.2d,v5.d[1]\n\t"
|
||||
"fmov v11.d[1],x0; ldr d12,[%[sb]]\n\t"
|
||||
FMLA_II "%[c3r].2d,v1.2d,v5.d[1]; ldr x0,[%[sb],#8]\n\t"
|
||||
FMLA_II "%[c4r].2d,v3.2d,v5.d[1]\n\t"
|
||||
FMLA_IR "%[c3i].2d,v1.2d,v5.d[0]\n\t"
|
||||
"fmov v12.d[1],x0; ldr d13,[%[sb],#16]\n\t"
|
||||
FMLA_IR "%[c4i].2d,v3.2d,v5.d[0]; ldr x0,[%[sb],#24]\n\t"
|
||||
FMLA_RR "%[c5r].2d,v0.2d,v6.d[0]\n\t"
|
||||
FMLA_RR "%[c6r].2d,v2.2d,v6.d[0]\n\t"
|
||||
"fmov v13.d[1],x0; ldr d14,[%[sb],#32]\n\t"
|
||||
FMLA_RI "%[c5i].2d,v0.2d,v6.d[1]; ldr x0,[%[sb],#40]\n\t"
|
||||
FMLA_RI "%[c6i].2d,v2.2d,v6.d[1]\n\t"
|
||||
FMLA_II "%[c5r].2d,v1.2d,v6.d[1]\n\t"
|
||||
"fmov v14.d[1],x0; ldr d15,[%[sb],#48]\n\t"
|
||||
FMLA_II "%[c6r].2d,v3.2d,v6.d[1]; ldr x0,[%[sb],#56]\n\t"
|
||||
FMLA_IR "%[c5i].2d,v1.2d,v6.d[0]\n\t"
|
||||
FMLA_IR "%[c6i].2d,v3.2d,v6.d[0]\n\t"
|
||||
"fmov v15.d[1],x0; ldr d4,[%[sb],#64]\n\t"
|
||||
FMLA_RR "%[c7r].2d,v0.2d,v7.d[0]; ldr x0,[%[sb],#72]\n\t"
|
||||
FMLA_RR "%[c8r].2d,v2.2d,v7.d[0]\n\t"
|
||||
FMLA_RI "%[c7i].2d,v0.2d,v7.d[1]\n\t"
|
||||
"fmov v4.d[1],x0; ldr d5,[%[sb],#80]\n\t"
|
||||
FMLA_RI "%[c8i].2d,v2.2d,v7.d[1]; ldr x0,[%[sb],#88]\n\t"
|
||||
FMLA_II "%[c7r].2d,v1.2d,v7.d[1]\n\t"
|
||||
FMLA_II "%[c8r].2d,v3.2d,v7.d[1]\n\t"
|
||||
"fmov v5.d[1],x0; ldr d0,[%[sa],#64]\n\t"
|
||||
FMLA_IR "%[c7i].2d,v1.2d,v7.d[0]; ldr x0,[%[sa],#80]\n\t"
|
||||
FMLA_IR "%[c8i].2d,v3.2d,v7.d[0]\n\t"
|
||||
FMLA_RR "%[c1r].2d,v8.2d,v12.d[0]\n\t"
|
||||
"fmov v0.d[1],x0; ldr d1,[%[sa],#72]\n\t"
|
||||
FMLA_RR "%[c2r].2d,v10.2d,v12.d[0]; ldr x0,[%[sa],#88]\n\t"
|
||||
FMLA_RI "%[c1i].2d,v8.2d,v12.d[1]\n\t"
|
||||
FMLA_RI "%[c2i].2d,v10.2d,v12.d[1]\n\t"
|
||||
"fmov v1.d[1],x0; ldr d2,[%[sa],#96]\n\t"
|
||||
FMLA_II "%[c1r].2d,v9.2d,v12.d[1]; ldr x0,[%[sa],#112]\n\t"
|
||||
FMLA_II "%[c2r].2d,v11.2d,v12.d[1]\n\t"
|
||||
FMLA_IR "%[c1i].2d,v9.2d,v12.d[0]\n\t"
|
||||
"fmov v2.d[1],x0; ldr d3,[%[sa],#104]\n\t"
|
||||
FMLA_IR "%[c2i].2d,v11.2d,v12.d[0]; ldr x0,[%[sa],#120]\n\t"
|
||||
FMLA_RR "%[c3r].2d,v8.2d,v13.d[0]\n\t"
|
||||
FMLA_RR "%[c4r].2d,v10.2d,v13.d[0]\n\t"
|
||||
"fmov v3.d[1],x0; ldr d6,[%[sb],#96]\n\t"
|
||||
FMLA_RI "%[c3i].2d,v8.2d,v13.d[1]; ldr x0,[%[sb],#104]\n\t"
|
||||
FMLA_RI "%[c4i].2d,v10.2d,v13.d[1]\n\t"
|
||||
FMLA_II "%[c3r].2d,v9.2d,v13.d[1]\n\t"
|
||||
"fmov v6.d[1],x0; ldr d7,[%[sb],#112]\n\t"
|
||||
FMLA_II "%[c4r].2d,v11.2d,v13.d[1]; ldr x0,[%[sb],#120]\n\t"
|
||||
FMLA_IR "%[c3i].2d,v9.2d,v13.d[0]\n\t"
|
||||
FMLA_IR "%[c4i].2d,v11.2d,v13.d[0]; prfm pldl1keep,[%[sa],#256]\n\t"
|
||||
FMLA_RR "%[c5r].2d,v8.2d,v14.d[0]\n\t"
|
||||
FMLA_RR "%[c6r].2d,v10.2d,v14.d[0]; prfm pldl1keep,[%[sa],#320]\n\t"
|
||||
FMLA_RI "%[c5i].2d,v8.2d,v14.d[1]\n\t"
|
||||
FMLA_RI "%[c6i].2d,v10.2d,v14.d[1]; prfm pldl1keep,[%[sb],#256]\n\t"
|
||||
FMLA_II "%[c5r].2d,v9.2d,v14.d[1]\n\t"
|
||||
FMLA_II "%[c6r].2d,v11.2d,v14.d[1]; prfm pldl1keep,[%[sb],#320]\n\t"
|
||||
FMLA_IR "%[c5i].2d,v9.2d,v14.d[0]\n\t"
|
||||
FMLA_IR "%[c6i].2d,v11.2d,v14.d[0]; add %[sa],%[sa],#128\n\t"
|
||||
FMLA_RR "%[c7r].2d,v8.2d,v15.d[0]\n\t"
|
||||
FMLA_RR "%[c8r].2d,v10.2d,v15.d[0]; add %[sb],%[sb],#128\n\t"
|
||||
FMLA_RI "%[c7i].2d,v8.2d,v15.d[1]\n\t"
|
||||
FMLA_RI "%[c8i].2d,v10.2d,v15.d[1]; sub %[K],%[K],#2\n\t"
|
||||
FMLA_II "%[c7r].2d,v9.2d,v15.d[1]\n\t"
|
||||
FMLA_II "%[c8r].2d,v11.2d,v15.d[1]; cmp %[K],#2\n\t"
|
||||
FMLA_IR "%[c7i].2d,v9.2d,v15.d[0]\n\t"
|
||||
FMLA_IR "%[c8i].2d,v11.2d,v15.d[0]; bgt 1b; blt 3f\n\t"
|
||||
"2:\n\t"
|
||||
"fmov v7.d[1],x0; ldr d8,[%[sa]]\n\t"
|
||||
FMLA_RR "%[c1r].2d,v0.2d,v4.d[0]; ldr x0,[%[sa],#16]\n\t"
|
||||
FMLA_RR "%[c2r].2d,v2.2d,v4.d[0]\n\t"
|
||||
FMLA_RI "%[c1i].2d,v0.2d,v4.d[1]\n\t"
|
||||
"fmov v8.d[1],x0; ldr d9,[%[sa],#8]\n\t"
|
||||
FMLA_RI "%[c2i].2d,v2.2d,v4.d[1]; ldr x0,[%[sa],#24]\n\t"
|
||||
FMLA_II "%[c1r].2d,v1.2d,v4.d[1]\n\t"
|
||||
FMLA_II "%[c2r].2d,v3.2d,v4.d[1]\n\t"
|
||||
"fmov v9.d[1],x0; ldr d10,[%[sa],#32]\n\t"
|
||||
FMLA_IR "%[c1i].2d,v1.2d,v4.d[0]; ldr x0,[%[sa],#48]\n\t"
|
||||
FMLA_IR "%[c2i].2d,v3.2d,v4.d[0]\n\t"
|
||||
FMLA_RR "%[c3r].2d,v0.2d,v5.d[0]\n\t"
|
||||
"fmov v10.d[1],x0; ldr d11,[%[sa],#40]\n\t"
|
||||
FMLA_RR "%[c4r].2d,v2.2d,v5.d[0]; ldr x0,[%[sa],#56]\n\t"
|
||||
FMLA_RI "%[c3i].2d,v0.2d,v5.d[1]\n\t"
|
||||
FMLA_RI "%[c4i].2d,v2.2d,v5.d[1]\n\t"
|
||||
"fmov v11.d[1],x0; ldr d12,[%[sb]]\n\t"
|
||||
FMLA_II "%[c3r].2d,v1.2d,v5.d[1]; ldr x0,[%[sb],#8]\n\t"
|
||||
FMLA_II "%[c4r].2d,v3.2d,v5.d[1]\n\t"
|
||||
FMLA_IR "%[c3i].2d,v1.2d,v5.d[0]\n\t"
|
||||
"fmov v12.d[1],x0; ldr d13,[%[sb],#16]\n\t"
|
||||
FMLA_IR "%[c4i].2d,v3.2d,v5.d[0]; ldr x0,[%[sb],#24]\n\t"
|
||||
FMLA_RR "%[c5r].2d,v0.2d,v6.d[0]\n\t"
|
||||
FMLA_RR "%[c6r].2d,v2.2d,v6.d[0]\n\t"
|
||||
"fmov v13.d[1],x0; ldr d14,[%[sb],#32]\n\t"
|
||||
FMLA_RI "%[c5i].2d,v0.2d,v6.d[1]; ldr x0,[%[sb],#40]\n\t"
|
||||
FMLA_RI "%[c6i].2d,v2.2d,v6.d[1]\n\t"
|
||||
FMLA_II "%[c5r].2d,v1.2d,v6.d[1]\n\t"
|
||||
"fmov v14.d[1],x0; ldr d15,[%[sb],#48]\n\t"
|
||||
FMLA_II "%[c6r].2d,v3.2d,v6.d[1]; ldr x0,[%[sb],#56]\n\t"
|
||||
FMLA_IR "%[c5i].2d,v1.2d,v6.d[0]\n\t"
|
||||
FMLA_IR "%[c6i].2d,v3.2d,v6.d[0]\n\t"
|
||||
"fmov v15.d[1],x0\n\t"
|
||||
FMLA_RR "%[c7r].2d,v0.2d,v7.d[0]\n\t"
|
||||
FMLA_RR "%[c8r].2d,v2.2d,v7.d[0]\n\t"
|
||||
FMLA_RI "%[c7i].2d,v0.2d,v7.d[1]\n\t"
|
||||
FMLA_RI "%[c8i].2d,v2.2d,v7.d[1]\n\t"
|
||||
FMLA_II "%[c7r].2d,v1.2d,v7.d[1]\n\t"
|
||||
FMLA_II "%[c8r].2d,v3.2d,v7.d[1]\n\t"
|
||||
FMLA_IR "%[c7i].2d,v1.2d,v7.d[0]\n\t"
|
||||
FMLA_IR "%[c8i].2d,v3.2d,v7.d[0]\n\t"
|
||||
FMLA_RR "%[c1r].2d,v8.2d,v12.d[0]\n\t"
|
||||
FMLA_RR "%[c2r].2d,v10.2d,v12.d[0]\n\t"
|
||||
FMLA_RI "%[c1i].2d,v8.2d,v12.d[1]\n\t"
|
||||
FMLA_RI "%[c2i].2d,v10.2d,v12.d[1]\n\t"
|
||||
FMLA_II "%[c1r].2d,v9.2d,v12.d[1]\n\t"
|
||||
FMLA_II "%[c2r].2d,v11.2d,v12.d[1]\n\t"
|
||||
FMLA_IR "%[c1i].2d,v9.2d,v12.d[0]\n\t"
|
||||
FMLA_IR "%[c2i].2d,v11.2d,v12.d[0]\n\t"
|
||||
FMLA_RR "%[c3r].2d,v8.2d,v13.d[0]\n\t"
|
||||
FMLA_RR "%[c4r].2d,v10.2d,v13.d[0]\n\t"
|
||||
FMLA_RI "%[c3i].2d,v8.2d,v13.d[1]\n\t"
|
||||
FMLA_RI "%[c4i].2d,v10.2d,v13.d[1]\n\t"
|
||||
FMLA_II "%[c3r].2d,v9.2d,v13.d[1]\n\t"
|
||||
FMLA_II "%[c4r].2d,v11.2d,v13.d[1]\n\t"
|
||||
FMLA_IR "%[c3i].2d,v9.2d,v13.d[0]\n\t"
|
||||
FMLA_IR "%[c4i].2d,v11.2d,v13.d[0]\n\t"
|
||||
FMLA_RR "%[c5r].2d,v8.2d,v14.d[0]\n\t"
|
||||
FMLA_RR "%[c6r].2d,v10.2d,v14.d[0]\n\t"
|
||||
FMLA_RI "%[c5i].2d,v8.2d,v14.d[1]\n\t"
|
||||
FMLA_RI "%[c6i].2d,v10.2d,v14.d[1]\n\t"
|
||||
FMLA_II "%[c5r].2d,v9.2d,v14.d[1]\n\t"
|
||||
FMLA_II "%[c6r].2d,v11.2d,v14.d[1]\n\t"
|
||||
FMLA_IR "%[c5i].2d,v9.2d,v14.d[0]\n\t"
|
||||
FMLA_IR "%[c6i].2d,v11.2d,v14.d[0]; add %[sa],%[sa],#64\n\t"
|
||||
FMLA_RR "%[c7r].2d,v8.2d,v15.d[0]\n\t"
|
||||
FMLA_RR "%[c8r].2d,v10.2d,v15.d[0]; add %[sb],%[sb],#64\n\t"
|
||||
FMLA_RI "%[c7i].2d,v8.2d,v15.d[1]\n\t"
|
||||
FMLA_RI "%[c8i].2d,v10.2d,v15.d[1]; sub %[K],%[K],#2\n\t"
|
||||
FMLA_II "%[c7r].2d,v9.2d,v15.d[1]\n\t"
|
||||
FMLA_II "%[c8r].2d,v11.2d,v15.d[1]\n\t"
|
||||
FMLA_IR "%[c7i].2d,v9.2d,v15.d[0]\n\t"
|
||||
FMLA_IR "%[c8i].2d,v11.2d,v15.d[0]; b 4f\n\t"
|
||||
"3:\n\t"
|
||||
"fmov v7.d[1],x0\n\t"
|
||||
FMLA_RR "%[c1r].2d,v0.2d,v4.d[0]\n\t"
|
||||
FMLA_RR "%[c2r].2d,v2.2d,v4.d[0]\n\t"
|
||||
FMLA_RI "%[c1i].2d,v0.2d,v4.d[1]\n\t"
|
||||
FMLA_RI "%[c2i].2d,v2.2d,v4.d[1]\n\t"
|
||||
FMLA_II "%[c1r].2d,v1.2d,v4.d[1]\n\t"
|
||||
FMLA_II "%[c2r].2d,v3.2d,v4.d[1]\n\t"
|
||||
FMLA_IR "%[c1i].2d,v1.2d,v4.d[0]\n\t"
|
||||
FMLA_IR "%[c2i].2d,v3.2d,v4.d[0]\n\t"
|
||||
FMLA_RR "%[c3r].2d,v0.2d,v5.d[0]\n\t"
|
||||
FMLA_RR "%[c4r].2d,v2.2d,v5.d[0]\n\t"
|
||||
FMLA_RI "%[c3i].2d,v0.2d,v5.d[1]\n\t"
|
||||
FMLA_RI "%[c4i].2d,v2.2d,v5.d[1]\n\t"
|
||||
FMLA_II "%[c3r].2d,v1.2d,v5.d[1]\n\t"
|
||||
FMLA_II "%[c4r].2d,v3.2d,v5.d[1]\n\t"
|
||||
FMLA_IR "%[c3i].2d,v1.2d,v5.d[0]\n\t"
|
||||
FMLA_IR "%[c4i].2d,v3.2d,v5.d[0]\n\t"
|
||||
FMLA_RR "%[c5r].2d,v0.2d,v6.d[0]\n\t"
|
||||
FMLA_RR "%[c6r].2d,v2.2d,v6.d[0]\n\t"
|
||||
FMLA_RI "%[c5i].2d,v0.2d,v6.d[1]\n\t"
|
||||
FMLA_RI "%[c6i].2d,v2.2d,v6.d[1]\n\t"
|
||||
FMLA_II "%[c5r].2d,v1.2d,v6.d[1]\n\t"
|
||||
FMLA_II "%[c6r].2d,v3.2d,v6.d[1]\n\t"
|
||||
FMLA_IR "%[c5i].2d,v1.2d,v6.d[0]\n\t"
|
||||
FMLA_IR "%[c6i].2d,v3.2d,v6.d[0]\n\t"
|
||||
FMLA_RR "%[c7r].2d,v0.2d,v7.d[0]\n\t"
|
||||
FMLA_RR "%[c8r].2d,v2.2d,v7.d[0]\n\t"
|
||||
FMLA_RI "%[c7i].2d,v0.2d,v7.d[1]\n\t"
|
||||
FMLA_RI "%[c8i].2d,v2.2d,v7.d[1]\n\t"
|
||||
FMLA_II "%[c7r].2d,v1.2d,v7.d[1]\n\t"
|
||||
FMLA_II "%[c8r].2d,v3.2d,v7.d[1]\n\t"
|
||||
FMLA_IR "%[c7i].2d,v1.2d,v7.d[0]\n\t"
|
||||
FMLA_IR "%[c8i].2d,v3.2d,v7.d[0]; sub %[K],%[K],#1\n\t"
|
||||
"4:\n\t"
|
||||
:[c1r]"=w"(c1r), [c1i]"=w"(c1i), [c2r]"=w"(c2r), [c2i]"=w"(c2i),
|
||||
[c3r]"=w"(c3r), [c3i]"=w"(c3i), [c4r]"=w"(c4r), [c4i]"=w"(c4i),
|
||||
[c5r]"=w"(c5r), [c5i]"=w"(c5i), [c6r]"=w"(c6r), [c6i]"=w"(c6i),
|
||||
[c7r]"=w"(c7r), [c7i]"=w"(c7i), [c8r]"=w"(c8r), [c8i]"=w"(c8i),
|
||||
[K]"+r"(K), [sa]"+r"(sa), [sb]"+r"(sb)
|
||||
::"cc", "memory", "x0", "v0", "v1", "v2", "v3", "v4", "v5", "v6", "v7",
|
||||
"v8", "v9", "v10", "v11", "v12", "v13", "v14", "v15");
|
||||
|
||||
store_4c(C, c1r, c1i, c2r, c2i, alphar, alphai); C += LDC * 2;
|
||||
store_4c(C, c3r, c3i, c4r, c4i, alphar, alphai); C += LDC * 2;
|
||||
store_4c(C, c5r, c5i, c6r, c6i, alphar, alphai); C += LDC * 2;
|
||||
store_4c(C, c7r, c7i, c8r, c8i, alphar, alphai);
|
||||
}
|
||||
|
||||
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K, FLOAT alphar, FLOAT alphai,
|
||||
FLOAT *sa, FLOAT *sb, FLOAT *C, BLASLONG LDC) {
|
||||
|
||||
BLASLONG n_left = N;
|
||||
for (; n_left >= 4; n_left -= 4) {
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c_ = C;
|
||||
BLASLONG m_left = M;
|
||||
for (; m_left >= 4; m_left -= 4) {
|
||||
kernel_4x4(a_, sb, c_, LDC, K, alphar, alphai);
|
||||
a_ += 8 * K;
|
||||
c_ += 8;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
kernel_2x4(a_, sb, c_, LDC, K, alphar, alphai);
|
||||
a_ += 4 * K;
|
||||
c_ += 4;
|
||||
}
|
||||
if (m_left) {
|
||||
kernel_1x4(a_, sb, c_, LDC, K, alphar, alphai);
|
||||
}
|
||||
sb += 8 * K;
|
||||
C += 8 * LDC;
|
||||
}
|
||||
if (n_left >= 2) {
|
||||
n_left -= 2;
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c_ = C;
|
||||
BLASLONG m_left = M;
|
||||
for (; m_left >= 4; m_left -= 4) {
|
||||
kernel_4x2(a_, sb, c_, LDC, K, alphar, alphai);
|
||||
a_ += 8 * K;
|
||||
c_ += 8;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
kernel_2x2(a_, sb, c_, LDC, K, alphar, alphai);
|
||||
a_ += 4 * K;
|
||||
c_ += 4;
|
||||
}
|
||||
if (m_left) {
|
||||
kernel_1x2(a_, sb, c_, LDC, K, alphar, alphai);
|
||||
}
|
||||
sb += 4 * K;
|
||||
C += 4 * LDC;
|
||||
}
|
||||
if (n_left) {
|
||||
const FLOAT *a_ = sa;
|
||||
FLOAT *c_ = C;
|
||||
BLASLONG m_left = M;
|
||||
for (; m_left >= 4; m_left -= 4) {
|
||||
kernel_4x1(a_, sb, c_, K, alphar, alphai);
|
||||
a_ += 8 * K;
|
||||
c_ += 8;
|
||||
}
|
||||
if (m_left >= 2) {
|
||||
m_left -= 2;
|
||||
kernel_2x1(a_, sb, c_, K, alphar, alphai);
|
||||
a_ += 4 * K;
|
||||
c_ += 4;
|
||||
}
|
||||
if (m_left) {
|
||||
kernel_1x1(a_, sb, c_, K, alphar, alphai);
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -49,7 +49,7 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define pCRow3 x15
|
||||
#define pA x16
|
||||
#define alphaR x17
|
||||
#define alphaI x18
|
||||
#define alphaI x22
|
||||
#define temp x19
|
||||
#define tempOffset x20
|
||||
#define tempK x21
|
||||
|
||||
@@ -47,7 +47,6 @@ FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x, FLOAT *y, BLASLONG inc_y)
|
||||
|
||||
if ( (inc_x == 1) && (inc_y == 1) )
|
||||
{
|
||||
int n1 = n & -4;
|
||||
#if V_SIMD && !defined(DSDOT)
|
||||
const int vstep = v_nlanes_f32;
|
||||
const int unrollx4 = n & (-vstep * 4);
|
||||
@@ -84,6 +83,7 @@ FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x, FLOAT *y, BLASLONG inc_y)
|
||||
}
|
||||
dot = v_sum_f32(vsum0);
|
||||
#elif defined(DSDOT)
|
||||
int n1 = n & -4;
|
||||
for (; i < n1; i += 4)
|
||||
{
|
||||
dot += (double) y[i] * (double) x[i]
|
||||
@@ -92,6 +92,7 @@ FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x, FLOAT *y, BLASLONG inc_y)
|
||||
+ (double) y[i+3] * (double) x[i+3] ;
|
||||
}
|
||||
#else
|
||||
int n1 = n & -4;
|
||||
for (; i < n1; i += 4)
|
||||
{
|
||||
dot += y[i] * x[i]
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
/***************************************************************************
|
||||
Copyright (c) 2020, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written permission.
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
|
||||
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*****************************************************************************/
|
||||
|
||||
#include "common.h"
|
||||
|
||||
#ifdef B0
|
||||
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K, IFLOAT * A, BLASLONG lda, FLOAT alpha, IFLOAT * B, BLASLONG ldb, FLOAT * C, BLASLONG ldc)
|
||||
#else
|
||||
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K, IFLOAT * A, BLASLONG lda, FLOAT alpha, IFLOAT * B, BLASLONG ldb, FLOAT beta, FLOAT * C, BLASLONG ldc)
|
||||
#endif
|
||||
{
|
||||
//naive implemtation
|
||||
//Column major
|
||||
|
||||
BLASLONG i,j,k;
|
||||
FLOAT result=0.0;
|
||||
|
||||
for(i=0; i<M; i++){
|
||||
for(j=0; j<N; j++){
|
||||
result=0.0;
|
||||
for(k=0; k<K; k++){
|
||||
result += A[i+k*lda] * B[k+j*ldb];
|
||||
}
|
||||
#ifdef B0
|
||||
C[i+j*ldc]=alpha * result;
|
||||
#else
|
||||
C[i+j*ldc]=C[i+j*ldc] * beta + alpha * result;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
/***************************************************************************
|
||||
Copyright (c) 2020, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written permission.
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
|
||||
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*****************************************************************************/
|
||||
|
||||
#include "common.h"
|
||||
|
||||
#ifdef B0
|
||||
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K, IFLOAT * A, BLASLONG lda, FLOAT alpha, IFLOAT * B, BLASLONG ldb, FLOAT * C, BLASLONG ldc)
|
||||
#else
|
||||
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K, IFLOAT * A, BLASLONG lda, FLOAT alpha, IFLOAT * B, BLASLONG ldb, FLOAT beta, FLOAT * C, BLASLONG ldc)
|
||||
#endif
|
||||
{
|
||||
//naive implemtation
|
||||
//Column major
|
||||
|
||||
BLASLONG i,j,k;
|
||||
FLOAT result=0.0;
|
||||
|
||||
for(i=0; i<M; i++){
|
||||
for(j=0; j<N; j++){
|
||||
result=0.0;
|
||||
for(k=0; k<K; k++){
|
||||
result += A[i+k*lda] * B[k*ldb+j];
|
||||
}
|
||||
#ifdef B0
|
||||
C[i+j*ldc]=alpha * result;
|
||||
#else
|
||||
C[i+j*ldc]=C[i+j*ldc] * beta + alpha * result;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user