Compare commits
459
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c59578f314 | ||
|
|
d9786d387c | ||
|
|
b9da7dbd24 | ||
|
|
94e053ac10 | ||
|
|
646d0c9fee | ||
|
|
2c80f8c974 | ||
|
|
0ea23484c6 | ||
|
|
1aea1d6237 | ||
|
|
6a5d2142f4 | ||
|
|
f9f8e94a14 | ||
|
|
9a0f76a0a1 | ||
|
|
75a99605af | ||
|
|
9d9fcc1881 | ||
|
|
e926bb0523 | ||
|
|
e875a9cdd0 | ||
|
|
fb45e7da89 | ||
|
|
e41cb1ad7a | ||
|
|
dc32a8a90f | ||
|
|
a04ea2b2c4 | ||
|
|
f272216ae3 | ||
|
|
9b3cc7835b | ||
|
|
bef5f1c6e2 | ||
|
|
3bbd755ba7 | ||
|
|
47be0d8a52 | ||
|
|
93515c2f7a | ||
|
|
7dde52d5d2 | ||
|
|
c6e4d17819 | ||
|
|
b9ba9be508 | ||
|
|
d27e98c97a | ||
|
|
429d23f420 | ||
|
|
3f2338ba85 | ||
|
|
62dcdca823 | ||
|
|
eaeaf95e23 | ||
|
|
f1f36c02b9 | ||
|
|
9816062aaf | ||
|
|
664f17655c | ||
|
|
aec6170a8b | ||
|
|
66cc9f043d | ||
|
|
cc74393520 | ||
|
|
4bbb9fefc0 | ||
|
|
e48625414f | ||
|
|
844939a9fb | ||
|
|
f085c70784 | ||
|
|
391cbf8584 | ||
|
|
6e89813300 | ||
|
|
37e189c85d | ||
|
|
6dad37ff8d | ||
|
|
004cf0d3d0 | ||
|
|
1243314201 | ||
|
|
edad2a8b2f | ||
|
|
55d7dd89ae | ||
|
|
e19e140619 | ||
|
|
a03cd30185 | ||
|
|
904f9d60b0 | ||
|
|
ff5dc3ebc1 | ||
|
|
a5d0f89ea4 | ||
|
|
af63f2a1aa | ||
|
|
4342764c23 | ||
|
|
d9bb8f369f | ||
|
|
f5f789fc52 | ||
|
|
605b1287e3 | ||
|
|
d26960a21e | ||
|
|
16211b7170 | ||
|
|
0f9f6e4be5 | ||
|
|
3ebfc0ef65 | ||
|
|
0315003d1f | ||
|
|
75511cb67c | ||
|
|
b8dbc4a1fc | ||
|
|
e6eba9fa21 | ||
|
|
2671786e61 | ||
|
|
3c188e4c12 | ||
|
|
7086a1b075 | ||
|
|
f6d4fe703b | ||
|
|
1f1fcd4927 | ||
|
|
e3ce4623c2 | ||
|
|
86971646ed | ||
|
|
b8697b3448 | ||
|
|
d511552e64 | ||
|
|
821242ed9d | ||
|
|
8cecf899e2 | ||
|
|
3f1eac4ba0 | ||
|
|
fd1c5ca01a | ||
|
|
52178f70c7 | ||
|
|
f88aa7def7 | ||
|
|
a24cca9576 | ||
|
|
7eab365219 | ||
|
|
6137054da3 | ||
|
|
b227de9429 | ||
|
|
99c120916a | ||
|
|
1b6fc34f0c | ||
|
|
51e904df27 | ||
|
|
8b9b3f0f5e | ||
|
|
500e32818a | ||
|
|
f6d5eb7af9 | ||
|
|
9d3ae22b28 | ||
|
|
494a3f735f | ||
|
|
496af0d8bb | ||
|
|
1e48eca408 | ||
|
|
faa06bd759 | ||
|
|
81d1029950 | ||
|
|
aa6a59a32e | ||
|
|
4956446ca2 | ||
|
|
a89142fd5d | ||
|
|
afcf70dad9 | ||
|
|
c9185e91ad | ||
|
|
0dd501d794 | ||
|
|
6bf687b2ef | ||
|
|
3f6e928d34 | ||
|
|
7d4a479a29 | ||
|
|
57cdef594b | ||
|
|
f0d142c4dd | ||
|
|
e9aab19bbc | ||
|
|
b7601ea92f | ||
|
|
8f5e49556f | ||
|
|
d7b13fec90 | ||
|
|
8c3717f69a | ||
|
|
6f672df537 | ||
|
|
6bb0dbfd3c | ||
|
|
8d6238f52e | ||
|
|
adba2c3c02 | ||
|
|
e5793d8406 | ||
|
|
afbd7c2b0d | ||
|
|
c9dae4c1e0 | ||
|
|
99d05575d0 | ||
|
|
7ff3588833 | ||
|
|
b3de37c96b | ||
|
|
2bdfe31986 | ||
|
|
5741aab90b | ||
|
|
79a50d80d3 | ||
|
|
53d0be88f8 | ||
|
|
7a95460bb1 | ||
|
|
a1fd7a4658 | ||
|
|
66063d123a | ||
|
|
86d1451cbe | ||
|
|
99c6a74e7b | ||
|
|
ddfbc6499b | ||
|
|
f2a89889a4 | ||
|
|
aa967ef6ba | ||
|
|
4a888bcb73 | ||
|
|
9a00d4859c | ||
|
|
460f5e8c0b | ||
|
|
705a5f2523 | ||
|
|
6ed4cc9c86 | ||
|
|
319343a5fd | ||
|
|
ea7d134aec | ||
|
|
62944c9db0 | ||
|
|
01270a94e8 | ||
|
|
f590468b69 | ||
|
|
cd47770092 | ||
|
|
ef3315527f | ||
|
|
48f0a0f0ec | ||
|
|
cc64ce68c3 | ||
|
|
450af57a68 | ||
|
|
86ccbe8fea | ||
|
|
b95729f5b0 | ||
|
|
fdc04c0e34 | ||
|
|
f881af5bdf | ||
|
|
bc69f86dba | ||
|
|
1ff3a1a13d | ||
|
|
5b1729eb6d | ||
|
|
ee70631c4d | ||
|
|
02f5f620de | ||
|
|
1a708bac8a | ||
|
|
55b16e5923 | ||
|
|
37262654d9 | ||
|
|
75e2f12dae | ||
|
|
d073702cdf | ||
|
|
78fd789da0 | ||
|
|
f30202b705 | ||
|
|
22fc689fa7 | ||
|
|
91eb0a638c | ||
|
|
1590d8baf0 | ||
|
|
98864c7c6f | ||
|
|
db6bbc7150 | ||
|
|
ecdabf9d74 | ||
|
|
dc8b16c57c | ||
|
|
754ad2ad4f | ||
|
|
3166fffcec | ||
|
|
df29cc0205 | ||
|
|
692023e364 | ||
|
|
5a534a63e8 | ||
|
|
303903e29c | ||
|
|
18638c70ef | ||
|
|
ef27ec6bed | ||
|
|
da0e066c9e | ||
|
|
1d0ca19457 | ||
|
|
30cf14c548 | ||
|
|
b4db4a1713 | ||
|
|
43728ade59 | ||
|
|
46b963b9a0 | ||
|
|
822aae6cab | ||
|
|
dccbf18c1f | ||
|
|
0cfb587fde | ||
|
|
92fcffff54 | ||
|
|
5a07c1b61c | ||
|
|
11986454b3 | ||
|
|
946a2cffec | ||
|
|
ef1c06f5eb | ||
|
|
b7542ffb3d | ||
|
|
5d29f88fed | ||
|
|
61db4e8191 | ||
|
|
bf0d7eaacc | ||
|
|
1da181dac6 | ||
|
|
4389e1de70 | ||
|
|
1defad49b6 | ||
|
|
be4ddc752f | ||
|
|
7fe8bd8046 | ||
|
|
7d431f3bb0 | ||
|
|
0e28b427f3 | ||
|
|
92b4d1b6f3 | ||
|
|
7d7a6c6708 | ||
|
|
d0a6e36896 | ||
|
|
6e3fb2ce52 | ||
|
|
efe63e7970 | ||
|
|
1ef6319990 | ||
|
|
1d6aa0dc31 | ||
|
|
7a1d23400f | ||
|
|
1cc377ef61 | ||
|
|
0acb60aab3 | ||
|
|
9701a80a9f | ||
|
|
4121a22c02 | ||
|
|
1690982cf1 | ||
|
|
5613deb794 | ||
|
|
ea82d802e6 | ||
|
|
e5ba61c344 | ||
|
|
387be46c42 | ||
|
|
445b11148f | ||
|
|
db00d5c2c9 | ||
|
|
33560437f5 | ||
|
|
e3cb067bf4 | ||
|
|
74d9fe2832 | ||
|
|
aa1cebd45b | ||
|
|
b5f2a50fe9 | ||
|
|
7da983ebac | ||
|
|
986ba29493 | ||
|
|
37f7a2e00c | ||
|
|
08381cd2f0 | ||
|
|
35e8eeaad3 | ||
|
|
720654ace1 | ||
|
|
0ae18524cd | ||
|
|
20699b1812 | ||
|
|
9e42e40884 | ||
|
|
e955736005 | ||
|
|
59da821b0d | ||
|
|
cb4e4ce8bb | ||
|
|
1a9cf8e291 | ||
|
|
27e35d639d | ||
|
|
69d92490c1 | ||
|
|
ebc3eaf80b | ||
|
|
0d6b7fe07b | ||
|
|
2ddcdafc0b | ||
|
|
2ef9819803 | ||
|
|
601bdde8ec | ||
|
|
d53d2b11a9 | ||
|
|
bc3b7e749a | ||
|
|
80995622dd | ||
|
|
dafb996425 | ||
|
|
b6aff4754a | ||
|
|
861b3db733 | ||
|
|
71261a7b3f | ||
|
|
1f3b81e562 | ||
|
|
413e609f9c | ||
|
|
a10f535803 | ||
|
|
2b4eaad2a0 | ||
|
|
5ffbf38b41 | ||
|
|
331b9ef11f | ||
|
|
a18a4ee08a | ||
|
|
60d03c3600 | ||
|
|
d5a5c7d319 | ||
|
|
3c9858cfa0 | ||
|
|
14594773a0 | ||
|
|
a8a2238848 | ||
|
|
8870cfc750 | ||
|
|
d40e19ef41 | ||
|
|
566e315f4f | ||
|
|
8742434212 | ||
|
|
8ea938f03a | ||
|
|
67c0675cf0 | ||
|
|
01657b356f | ||
|
|
70ecde3e49 | ||
|
|
3628f35251 | ||
|
|
d2906e8787 | ||
|
|
f298361f98 | ||
|
|
4001d7a74f | ||
|
|
55e853a698 | ||
|
|
c077708852 | ||
|
|
45e9820118 | ||
|
|
f8a9c067d8 | ||
|
|
76f1be470c | ||
|
|
0e7b11fbc3 | ||
|
|
ca1aefc5f9 | ||
|
|
366847fc10 | ||
|
|
10bd0ec4c3 | ||
|
|
5bb7ef1466 | ||
|
|
4cd575c20f | ||
|
|
6f225daf94 | ||
|
|
55a10c748d | ||
|
|
5133aac055 | ||
|
|
faa18750d6 | ||
|
|
7acf919836 | ||
|
|
93cd7b9238 | ||
|
|
d49df4c579 | ||
|
|
7ffce1c788 | ||
|
|
88c583ed49 | ||
|
|
d3e4b41136 | ||
|
|
6735872092 | ||
|
|
6137236c0a | ||
|
|
fa021e1887 | ||
|
|
bdcb9b7252 | ||
|
|
533cab235f | ||
|
|
31bb6ca7df | ||
|
|
5e5f9a39ad | ||
|
|
770ad6883d | ||
|
|
10ba0e6044 | ||
|
|
3149408165 | ||
|
|
e07bea17c8 | ||
|
|
e04df1941d | ||
|
|
31150eb1e6 | ||
|
|
0a53d91789 | ||
|
|
aafd3cb0db | ||
|
|
01cc6df92e | ||
|
|
e776297bf9 | ||
|
|
05d7c18894 | ||
|
|
4d08156266 | ||
|
|
6de062cfc2 | ||
|
|
52ec7faf31 | ||
|
|
1ffea2b8c1 | ||
|
|
a3af4cadcc | ||
|
|
d1de282a4e | ||
|
|
e5aebeaf93 | ||
|
|
a514760e06 | ||
|
|
d7d1088d21 | ||
|
|
a9a6edaf17 | ||
|
|
2d46f1ec65 | ||
|
|
c040d5ed86 | ||
|
|
6939a43c3b | ||
|
|
20ae36ba75 | ||
|
|
b53d18b3ad | ||
|
|
7e612b640f | ||
|
|
e384396a51 | ||
|
|
02bc005306 | ||
|
|
a18a53605e | ||
|
|
618bcbd7c0 | ||
|
|
badf4c09e2 | ||
|
|
879497990f | ||
|
|
2283fcbbe7 | ||
|
|
67fd33e729 | ||
|
|
f4383d0235 | ||
|
|
7beba94023 | ||
|
|
b183182e61 | ||
|
|
5c8cf37d83 | ||
|
|
275eb6f7f3 | ||
|
|
e4344def6a | ||
|
|
80951a2acc | ||
|
|
772741e2b1 | ||
|
|
54f7b76f20 | ||
|
|
579eda3778 | ||
|
|
83a788c387 | ||
|
|
5e3a9922bd | ||
|
|
e548bda1ba | ||
|
|
cd02751b12 | ||
|
|
5766adbcad | ||
|
|
1f2bffb4fe | ||
|
|
067e43c1e1 | ||
|
|
0ff51a40f1 | ||
|
|
0b2b583223 | ||
|
|
fed16d638c | ||
|
|
d39b77748f | ||
|
|
371663f0c2 | ||
|
|
097d2d98fd | ||
|
|
5b0884d8e7 | ||
|
|
c7b0304ba3 | ||
|
|
652bf6b51b | ||
|
|
9fa64b9a3d | ||
|
|
b8163b65cb | ||
|
|
ac2c66321d | ||
|
|
cfa28bcf71 | ||
|
|
6bc4276f90 | ||
|
|
bc52252cd5 | ||
|
|
5aff62eb96 | ||
|
|
e155bc0061 | ||
|
|
e1d2411545 | ||
|
|
cbecf98308 | ||
|
|
1607a49cb9 | ||
|
|
e85efb8d86 | ||
|
|
825d3ad12e | ||
|
|
5f0735832b | ||
|
|
c3c857c95e | ||
|
|
7ab8dc125d | ||
|
|
705259c344 | ||
|
|
a683287006 | ||
|
|
b185c9a4ce | ||
|
|
4af187080a | ||
|
|
b0bd49a064 | ||
|
|
7e44f62a09 | ||
|
|
8da0a1fb9c | ||
|
|
7d35bf61ba | ||
|
|
8c0b13c41c | ||
|
|
9c0965b884 | ||
|
|
ea85b6696f | ||
|
|
4867c421ed | ||
|
|
71c6016206 | ||
|
|
8882409131 | ||
|
|
92fe96b460 | ||
|
|
682f61e8b8 | ||
|
|
83d3e0ed1a | ||
|
|
1b591ea4ed | ||
|
|
f4ee3aec88 | ||
|
|
e01b1094de | ||
|
|
643a0b53b0 | ||
|
|
d7b0fccbb4 | ||
|
|
2346d0bdc4 | ||
|
|
8211db6203 | ||
|
|
3d5010bf37 | ||
|
|
96f34621fb | ||
|
|
d539685c49 | ||
|
|
9bfc3612f9 | ||
|
|
47a66aef0f | ||
|
|
20f5ed1a94 | ||
|
|
c889558317 | ||
|
|
4ae3e37b45 | ||
|
|
b3d0bc40e9 | ||
|
|
ba9d2d29f3 | ||
|
|
fc516af155 | ||
|
|
2b5d8c789d | ||
|
|
1b88c9c742 | ||
|
|
b4fc09e9e1 | ||
|
|
8e50b8d525 | ||
|
|
7f89c6f353 | ||
|
|
1ee8879c78 | ||
|
|
edaa73fd24 | ||
|
|
501728a354 | ||
|
|
107c883c8a | ||
|
|
05dbb54362 | ||
|
|
4609732e69 | ||
|
|
bf98e448eb | ||
|
|
0bc19a1335 | ||
|
|
426b5f23ed | ||
|
|
4328c91e27 | ||
|
|
c794d0a4ce | ||
|
|
a4f5fec46e | ||
|
|
ca542f319f | ||
|
|
18f9582f3e | ||
|
|
4e2a8c18e5 | ||
|
|
30970460b8 | ||
|
|
b0a00fbd62 | ||
|
|
ccfd0170fb | ||
|
|
ef0b883dff | ||
|
|
e76c39099a | ||
|
|
202a7a0e2a | ||
|
|
de91afd2ae | ||
|
|
0203657f40 | ||
|
|
e82bcd2740 | ||
|
|
731f4dd686 | ||
|
|
53d3bb50cc | ||
|
|
08a00326a4 | ||
|
|
89898fc499 | ||
|
|
22c6607db9 | ||
|
|
ca22e28ca1 |
+5
-3
@@ -89,14 +89,16 @@ task:
|
||||
type: text/plain
|
||||
|
||||
macos_instance:
|
||||
image: ghcr.io/cirruslabs/macos-sonoma-xcode:latest
|
||||
image: ghcr.io/cirruslabs/macos-tahoe-xcode:latest
|
||||
task:
|
||||
name: AppleM1/LLVM armv7-androidndk xbuild
|
||||
compile_script:
|
||||
- brew install --cask android-ndk
|
||||
- export ANDROID_NDK_HOME="/opt/homebrew/share/android-ndk"
|
||||
- export CC=/opt/homebrew/share/android-ndk/toolchains/llvm/prebuilt/darwin-x86_64/bin/armv7a-linux-androideabi23-clang
|
||||
- make TARGET=ARMV7 ARM_SOFTFP_ABI=1 NUM_THREADS=32 HOSTCC=clang NOFORTRAN=1 RANLIB="ls -l"
|
||||
- export AR=/opt/homebrew/share/android-ndk/toolchains/llvm/prebuilt/darwin-x86_64/bin/llvm-ar
|
||||
- export RANLIB=/opt/homebrew/share/android-ndk/toolchains/llvm/prebuilt/darwin-x86_64/bin/llvm-ranlib
|
||||
- make TARGET=ARMV7 ARM_SOFTFP_ABI=1 NUM_THREADS=32 HOSTCC=clang NOFORTRAN=1
|
||||
always:
|
||||
config_artifacts:
|
||||
path: "*conf*"
|
||||
@@ -151,7 +153,7 @@ FreeBSD_task:
|
||||
image_family: freebsd-14-3
|
||||
install_script:
|
||||
- pkg update -f && pkg upgrade -y && pkg install -y gmake gcc
|
||||
- ln -s /usr/local/lib/gcc13/libgfortran.so.5.0.0 /usr/lib/libgfortran.so
|
||||
- ln -s /usr/local/lib/gcc14/libgfortran.so.5.0.0 /usr/lib/libgfortran.so
|
||||
compile_script:
|
||||
- gmake CC=clang FC=gfortran USE_OPENMP=1 CPP_THREAD_SAFETY_TEST=1
|
||||
|
||||
|
||||
@@ -99,6 +99,7 @@ jobs:
|
||||
run: |
|
||||
export CPPFLAGS="-I/opt/homebrew/opt/llvm/include"
|
||||
export CC="/opt/homebrew/opt/llvm/bin/clang"
|
||||
export RANLIB=llvm-ranlib
|
||||
case "${{ matrix.build }}" in
|
||||
"make")
|
||||
make -j$(nproc) DYNAMIC_ARCH=1 USE_OPENMP=${{matrix.openmp}} INTERFACE64=${{matrix.ilp64}} FC="ccache ${{ matrix.fortran }}"
|
||||
|
||||
+36
-3
@@ -9,7 +9,7 @@ project(OpenBLAS C ASM)
|
||||
|
||||
set(OpenBLAS_MAJOR_VERSION 0)
|
||||
set(OpenBLAS_MINOR_VERSION 3)
|
||||
set(OpenBLAS_PATCH_VERSION 30.dev)
|
||||
set(OpenBLAS_PATCH_VERSION 32.dev)
|
||||
|
||||
set(OpenBLAS_VERSION "${OpenBLAS_MAJOR_VERSION}.${OpenBLAS_MINOR_VERSION}.${OpenBLAS_PATCH_VERSION}")
|
||||
|
||||
@@ -308,8 +308,8 @@ if (USE_OPENMP)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Fix "Argument list too long" for macOS with POWERPC or Intel CPUs
|
||||
if(APPLE AND (NOT CMAKE_HOST_SYSTEM_PROCESSOR STREQUAL "arm64"))
|
||||
# Fix "Argument list too long" for macOS - mostly seen with older OS versions on POWERPC or Intel CPUs
|
||||
if(APPLE)
|
||||
# Use response files
|
||||
set(CMAKE_C_USE_RESPONSE_FILE_FOR_OBJECTS 1)
|
||||
# Always build static library first
|
||||
@@ -708,6 +708,39 @@ if(NOT NO_LAPACKE)
|
||||
COMMAND ${CMAKE_COMMAND} -E copy ${CMAKE_CURRENT_SOURCE_DIR}/lapack-netlib/LAPACKE/include/lapacke_mangling_with_flags.h.in "${CMAKE_BINARY_DIR}/lapacke_mangling.h"
|
||||
)
|
||||
install (FILES ${CMAKE_BINARY_DIR}/lapacke_mangling.h DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
|
||||
if (NOT (x${SYMBOLPREFIX}${SYMBOLSUFFIX} STREQUAL "x"))
|
||||
message (STATUS "Generating lapacke.h in ${CMAKE_INSTALL_INCLUDEDIR}")
|
||||
set(LAPACKE_H ${CMAKE_BINARY_DIR}/generated/lapacke.h)
|
||||
file(READ ${CMAKE_CURRENT_SOURCE_DIR}/lapack-netlib/LAPACKE/include/lapacke.h LAPACKE_H_CONTENTS)
|
||||
if (NOT ${SYMBOLPREFIX} STREQUAL "")
|
||||
string(REGEX REPLACE "(LAPACKE_*)" " ${SYMBOLPREFIX}\\1" LAPACKE_H_CONTENTS_NEW "${LAPACKE_H_CONTENTS}")
|
||||
string(REPLACE "_ ${SYMBOLPREFIX}LAPACKE_H_" "_LAPACKE_H_" LAPACKE_H_CONTENTS ${LAPACKE_H_CONTENTS_NEW})
|
||||
string(REPLACE "${SYMBOLPREFIX}LAPACKE_malloc" "LAPACKE_malloc" LAPACKE_H_CONTENTS_NEW ${LAPACKE_H_CONTENTS})
|
||||
string(REPLACE "${SYMBOLPREFIX}LAPACKE_free" "LAPACKE_free" LAPACKE_H_CONTENTS ${LAPACKE_H_CONTENTS_NEW})
|
||||
set(LAPACKE_H_CONTENTS_NEW ${LAPACKE_H_CONTENTS})
|
||||
endif()
|
||||
if (NOT ${SYMBOLSUFFIX} STREQUAL "")
|
||||
string(REGEX REPLACE "(${SYMBOLPREFIX}LAPACKE_[a-z1-9]*[^ (]*)" "\\1${SYMBOLSUFFIX}" LAPACKE_H_CONTENTS_NEW "${LAPACKE_H_CONTENTS}")
|
||||
string(REPLACE "#define${SYMBOLSUFFIX}" "#define" LAPACKE_H_CONTENTS ${LAPACKE_H_CONTENTS_NEW})
|
||||
string(REPLACE "LAPACKE_malloc${SYMBOLSUFFIX}" "LAPACKE_malloc" LAPACKE_H_CONTENTS_NEW ${LAPACKE_H_CONTENTS})
|
||||
string(REPLACE "LAPACKE_free${SYMBOLSUFFIX}" "LAPACKE_free" LAPACKE_H_CONTENTS ${LAPACKE_H_CONTENTS_NEW})
|
||||
set(LAPACKE_H_CONTENTS_NEW ${LAPACKE_H_CONTENTS})
|
||||
endif()
|
||||
file(WRITE ${LAPACKE_H} "${LAPACKE_H_CONTENTS_NEW}")
|
||||
install (FILES ${LAPACKE_H} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
|
||||
message (STATUS "Generating lapack.h in ${CMAKE_INSTALL_INCLUDEDIR}")
|
||||
set(LAPACK_H ${CMAKE_BINARY_DIR}/generated/lapack.h)
|
||||
file(READ ${CMAKE_CURRENT_SOURCE_DIR}/lapack-netlib/LAPACKE/include/lapack.h LAPACK_H_CONTENTS)
|
||||
if (NOT ${SYMBOLPREFIX} STREQUAL "")
|
||||
string(REGEX REPLACE "(LAPACK_[a-z1-9]*[ \(][.\)]*)" "${SYMBOLPREFIX}\\1" LAPACK_H_CONTENTS_NEW "${LAPACK_H_CONTENTS}")
|
||||
set(LAPACK_H_CONTENTS ${LAPACK_H_CONTENTS_NEW})
|
||||
endif()
|
||||
if (NOT ${SYMBOLSUFFIX} STREQUAL "")
|
||||
string(REGEX REPLACE "(${SYMBOLPREFIX}LAPACK_[a-z1-9]*)([ \(].\)" "\\1${SYMBOLSUFFIX}\\2" LAPACK_H_CONTENTS_NEW "${LAPACK_H_CONTENTS}")
|
||||
endif()
|
||||
file(WRITE ${LAPACK_H} "${LAPACK_H_CONTENTS_NEW}")
|
||||
install (FILES ${LAPACK_H} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Install pkg-config files
|
||||
|
||||
@@ -29,6 +29,9 @@
|
||||
* Annop Wongwathanarat <annop.wongwathanarat@arm.com>
|
||||
* Optimizations and other improvements targeting AArch64
|
||||
|
||||
* Anna Mayne <anna.mayne@arm.com>
|
||||
* Optimizations and other improvements targeting AArch64
|
||||
|
||||
## Previous Developers
|
||||
|
||||
* Zaheer Chothia <zaheer.chothia@gmail.com>
|
||||
@@ -267,3 +270,8 @@ In chronological order:
|
||||
* [2025-05-29] Optimise axpby kernel for RISCV64_ZVL256B
|
||||
* [2025-06-05] Optimise hbmv kernel for RISCV64_ZVL256B
|
||||
|
||||
* Anna Mayne <anna.mayne@arm.com>
|
||||
* [2025-11-19] Update thread throttling profile for SGEMV on NEOVERSEV1 and NEOVERSEV2
|
||||
|
||||
* Fadi Arafeh <fadi.arafeh@arm.com>
|
||||
* [2026-03-05] Accelerate SVE128 SBGEMM/BGEMM
|
||||
|
||||
+199
@@ -1,4 +1,203 @@
|
||||
OpenBLAS ChangeLog
|
||||
====================================================================
|
||||
Version 0.3.32
|
||||
23-Mar-2026
|
||||
|
||||
general:
|
||||
- Moved the preliminary support for a Web Assembly target to its own WASM
|
||||
architecture and WASM128_GENERIC target
|
||||
- Fixed a potential performance difference between dedicated compilation for
|
||||
a target and its representation in DYNAMIC_ARCH builds by making additional
|
||||
cpu-specific parameters available to the DYNAMIC_ARCH configuration
|
||||
- Fixed the reimplementation of LAPACK ?GESV to conform to the reference (i.e.
|
||||
compute the LU factorization even when NRHS is zero)
|
||||
- Improved the error message that is displayed when the compile-time allocation
|
||||
of memory buffers is exceeded
|
||||
- Fixed a problem with non-serialized accesses to parallelized SYRK by concurrent
|
||||
callers
|
||||
- Fixed an ABI mismatch in the internal version of CDOT/ZDOT used by the C fallback
|
||||
versions of the LAPACK source
|
||||
- Improved the f_check script for detecting the Fortran compiler to handle embedded
|
||||
dashes in path names
|
||||
- Fixed several memory access issues in the utests that were detected by Address
|
||||
Sanitizer
|
||||
- Fixed Makefile errors in cases where only a subset of precision types was selected
|
||||
- Fixed missing function errors in Makefile builds without LAPACK or without threads
|
||||
- Fixed a syntax error in the benchmarks Makefile
|
||||
- Fixed compiler warnings in the CBLAS testsuite
|
||||
- Fixed the OpenMP compiler option used with the Intel Ifx compiler
|
||||
- Updated the README sections on supported cpus and operating systems, and added
|
||||
notes pertaining to JAVA
|
||||
- Updated the documentation page for supported BLAS-like extensions
|
||||
- included fixes from the Reference-LAPACK project:
|
||||
- Improved step length selection in the fallback path of ?LAED4
|
||||
(Reference-LAPACK PR 1191)
|
||||
- Rounding up of LWORK and removal of redundant type conversions in the GVD
|
||||
functions (Reference-LAPACK PR 1202)
|
||||
- internal errors were getting ignored in calculation of selected eigenvalues
|
||||
(Reference-LAPACK PR 1204)
|
||||
|
||||
arm64:
|
||||
- Fixed a potential miscompilation of the SDOT/DDOT/DSDOT kernels
|
||||
- Fixed DYNAMIC_ARCH compilation with CMake and compilers lacking SVE support
|
||||
- Improved the performance of BGEMM and SBGEMM kernels for Neoverse V2
|
||||
- Added optimized SSUM and DSUM kernels for Neoverse N1
|
||||
- Added preliminary support for Neoverse V3 cpus as NEOVERSEV2
|
||||
- Added cpu autodetection of Cortex A725 and X925 cpus
|
||||
- Fixed a CMake build problem with flang on Mac OS
|
||||
- Fixed build problems with gcc versions 12 and earlier that do not support fp16
|
||||
- Fixed compilation of GEMM kernels for VORTEXM4/ARMV9SME without multithreading
|
||||
- Fixed the optimized CDOT/ZDOT kernel to compile with LLVM under Windows on Arm
|
||||
- Renamed the copy of the DllMain function used in static linking on MS Windows to
|
||||
OpenBLASDllMain to avoid symbol name conflicts with other libraries
|
||||
|
||||
ioongarch64:
|
||||
- fixed POTRF returning wrong results on LA464 due to a wrong parameter setting
|
||||
|
||||
power:
|
||||
- Fixed compilation problems caused by missing support for half-precision floats (FP16)
|
||||
- Fixed a potential miscompilation of the POWER10 DGEMV kernel by limiting its optimization
|
||||
level
|
||||
- Fixed a SCAL issue on PPCG4/PPC970 running Linux
|
||||
- Worked around a SCAL issue on PPC970 running FreeBSD by switching to the generic C kernels
|
||||
|
||||
riscv64:
|
||||
- Optimized the CROT/ZROT kernel for vector length 128 in the non-unit stride path
|
||||
- Improved SBGEMM/SHGEMM and related helper functions for type conversion
|
||||
- Fixed probing for BFLOAT16 support in DYNAMIC_ARCH cpu detection at runtime
|
||||
|
||||
x86_64:
|
||||
- Fixed a potential miscompilation (by gcc 15.x) of the AVX512 SGEMM kernel for "small"
|
||||
matrix sizes
|
||||
- Fixed the SROT and DROT kernels for Haswell to have consistent (FMA) rounding
|
||||
in the main loop and tail call
|
||||
- Added automatic detection of Intel Arrow Lake H/U, Panther Lake and Jasper Lake
|
||||
- Added automatic detection of Intel Emerald Rapids and upcoming cpu models
|
||||
- Updated the cache size translation table in the cpu model autodetection code
|
||||
- Improved cpu detection fallback to also include Nehalem as a non-AVX option
|
||||
- Fixed a Makefile build issue with clang and the SkylakeX SGEMM kernel
|
||||
- Renamed the copy of the DllMain function used in static linking on MS Windows to
|
||||
OpenBLASDllMain to avoid symbol name conflicts with other libraries
|
||||
|
||||
wasm:
|
||||
- Added optimized intrinsics kernels for SGEMM and DGEMM as well as DOT, ROT and SUM
|
||||
|
||||
====================================================================
|
||||
Version 0.3.31
|
||||
15-Jan-2026
|
||||
|
||||
general:
|
||||
- reverted a matrix partitioning optimization from 0.3.30 that could lead to
|
||||
race conditions and subsequent invalid results in GEMM
|
||||
- added the bfloat16 extensions BGEMM and BGEMV
|
||||
- added a BLAS interface for the ?GEMM_BATCH extensions
|
||||
- added the BLAS extensions ?GEMM_BATCH_STRIDED and their CBLAS interface
|
||||
- added the basic infrastructure for half-precision float (FP16) format
|
||||
using SH prefix
|
||||
- reimplemented the LAPACK SLAED3/DLAED3 function using multithreading, thereby
|
||||
improving the performance of the SSYEVD/DSYEVD eigensolver for symmetric matrices
|
||||
on all platforms
|
||||
- limited the number of retries for initial memory allocation to avoid infinite
|
||||
hanging on low-memory systems
|
||||
- fixed a thread lockup situation encountered with python 3.9 or older and numpy
|
||||
- introduced a problem size threshold for multithreading in STRMV/DTRMV
|
||||
- introduced a problem size threshold for multithreading in CHER/CHER2/CHPR/CHPR2
|
||||
and ZHER/ZHER2/ZHPR/ZHPR2
|
||||
- improved the problem size thresholds for multithreading in SGER/DGER
|
||||
- improved autodetection of the Fortran compiler
|
||||
- fixed passing of the INTERFACE64=1 option to the flang-new compiler
|
||||
- fixed a potential deadlock in multithreaded code after calling fork()
|
||||
- fixed builds using CMake on FreeBSD
|
||||
- fixed builds using CMake from within Cygwin on Windows
|
||||
- fixed builds using CMake and the NVHPC compiler on ARM64
|
||||
- fixed CMake build error from misdetecting compiler or OpenMP versions
|
||||
- improved contents of the CMake-generated OpenBLASConfig.cmake file
|
||||
- added support for cross-compilation to RISCV targets via CMake
|
||||
- fixed cross-compilation to x86 targets from non-x86 architectures
|
||||
- fixed failure to install cblas.h if NO_CBLAS=0 was specified
|
||||
- fixed missing user-defined pre- and postfixes on functions in lapack.h,lapacke.h
|
||||
- included fixes from the Reference-LAPACK project:
|
||||
- fix ordering bug in ?LAED/?LASD (Reference-LAPACK PR 1140)
|
||||
- revert changes in ?GEEV from PR 1129 (Reference-LAPACK PR 1142)
|
||||
- fix workspace allocation in LAPACKE_?TRSEN (Reference-LAPACK PR 1144)
|
||||
|
||||
riscv:
|
||||
- added optimized SBGEMM kernels for ZVL128B and ZVL256B targets
|
||||
- added optimized SHGEMM kernels for ZVL128B and ZVL256B targets
|
||||
- added optimized SBGEMV and SHGEMV kernels for ZVL128B/ZVL256B
|
||||
- improved performance of the GEMV kernel for ZVL256B
|
||||
- improved the performance of the CROT and ZROT kernels for ZVL128B and x280
|
||||
- improved the detection of RVV1.0 capability
|
||||
- improved performance of the matrix packing helper functions for ZVL128B and ZVL256B
|
||||
- improved performance of OMATCOPY for ZVL128B and ZVL256B
|
||||
|
||||
arm:
|
||||
- fixed spurious executable stack in the getarch utility
|
||||
|
||||
arm64:
|
||||
- fixed spurious executable stack in the getarch utility
|
||||
- fixed compiler warnings arising from the timer macro RPCC
|
||||
- fixed cache size detection for Qualcomm Oryon under Windows on Arm
|
||||
- fixed argument handling in the default SVE kernel for SDOT/DDOT
|
||||
- building the BFLOAT16 kernels is now enabled by default
|
||||
- improved the overall performance of GEMM,SYMM and HEMM on A64FX
|
||||
- improved the performance of SDOT/DDOT on A64FX
|
||||
- improved the multithreading performance of SDOT/DDOT on A64FX by
|
||||
introduction of a throttling table matching thread count to problem size
|
||||
- improved the performance of SGER/DGER on A64FX and NEOVERSEV1
|
||||
- improved the multithreading performance of GEMM on A64FX and NEOVERSEV1
|
||||
- improved the performance of the GEMV kernel for SVE-capable targets
|
||||
- improved the multithreading performance of SGEMM on NEOVERSEV1 and V2
|
||||
- added optimized SAXPY/DAXPY SVE kernels for A64FX and NEOVERSEV1
|
||||
- added optimized BGEMM and BGEMV kernels for NEOVERSEV1
|
||||
- added an optimized BGEMM kernel for NEOVERSEN2
|
||||
- added support for the NEOVERSEV2 cpu
|
||||
- added dedicated support for the Apple M4 cpu as VORTEXM4
|
||||
- added optimized SGEMM/SSYMM/STRMM/SSYRK/SSYR2K for SME-capable targets
|
||||
(ARMV9SME and VORTEXM4)
|
||||
- improved the precision of the SNRM2 kernel
|
||||
- added cpu autodetection and compiler settings for Ampere One processors
|
||||
- fixed cpu autodetection for Apple M systems running Linux
|
||||
- fixed building on MacOS with AppleClang,gfortran and xcode v16 or newer
|
||||
- fixed several errors in the C code replacements for the complex and double
|
||||
precision complex LAPACK functions that get used (only) when compiling with
|
||||
Microsoft C and NOFORTRAN=1 under MS Windows
|
||||
|
||||
power:
|
||||
- added initial support for the POWER11 architecture
|
||||
- improved performance of DGEMM and DGEMV on POWER10
|
||||
- fixed the default compiler flags to use "-O3" instead of the possibly unsafe
|
||||
"-Ofast"
|
||||
- fixed building under MacOS (for old G4 Macs) with CMake
|
||||
- fixed potential miscompilation of DGEMV and other assembly kernels by gcc15.1
|
||||
- fixed compilation with recent versions of flang
|
||||
|
||||
loongarch64:
|
||||
- fixed warnings and potential inaccuracies arising from incorrect saving of registers
|
||||
- fixed enumeration of logical cores on big NUMA servers
|
||||
- fixed building with LLVM and the INTERFACE64=1 option
|
||||
|
||||
x86:
|
||||
- fixed building the GEMM3M kernels for the GENERIC target
|
||||
- fixed several errors in the C code replacements for the complex and double
|
||||
precision complex LAPACK functions that get used (only) when compiling with
|
||||
Microsoft C and NOFORTRAN=1 under MS Windows
|
||||
|
||||
x86_64:
|
||||
- added cpu autodetection for Intel Lunar Lake (Core Ultra 200V)
|
||||
- changed all ?MIN and ?MAX assembly kernels to use unaligned operations
|
||||
- fixed several errors in the C code replacements for the complex and double
|
||||
precision complex LAPACK functions that get used (only) when compiling with
|
||||
Microsoft C and NOFORTRAN=1 under MS Windows
|
||||
- fixed potential crashes in builds for Cooper Lake, Sapphire Rapids or Zen5 cpus
|
||||
under MS Windows
|
||||
|
||||
zarch:
|
||||
- added support for building with CMake
|
||||
|
||||
sparc:
|
||||
- fixed a potential crash in the DNRM2 kernel
|
||||
|
||||
====================================================================
|
||||
Version 0.3.30
|
||||
19-Jun-2025
|
||||
|
||||
+21
-6
@@ -1,16 +1,31 @@
|
||||
pipeline {
|
||||
agent {
|
||||
docker {
|
||||
image 'osuosl/ubuntu-ppc64le:18.04'
|
||||
}
|
||||
}
|
||||
agent none
|
||||
stages {
|
||||
stage('Build') {
|
||||
stage('GCC build') {
|
||||
agent {
|
||||
docker {
|
||||
image 'osuosl/ubuntu-ppc64le:18.04' // gcc 7, gfortran 7
|
||||
}
|
||||
}
|
||||
steps {
|
||||
checkout scm
|
||||
sh 'sudo apt update'
|
||||
sh 'sudo apt install gfortran -y'
|
||||
sh 'make clean && make'
|
||||
}
|
||||
}
|
||||
stage('Clang build') {
|
||||
agent {
|
||||
docker {
|
||||
image 'osuosl/ubuntu-ppc64le:20.04' // clang 10, gfortran 9
|
||||
}
|
||||
}
|
||||
steps {
|
||||
checkout scm
|
||||
sh 'sudo apt update'
|
||||
sh 'sudo apt install -y clang gfortran'
|
||||
sh 'make clean && make CC=clang'
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -61,6 +61,11 @@ endif
|
||||
ifeq ($(CORE), ARMV9SME)
|
||||
CCOMMON_OPT += -march=armv9-a+sve2+sme
|
||||
FCOMMON_OPT += -march=armv9-a+sve2
|
||||
ifdef OS_WINDOWS
|
||||
ifeq ($(C_COMPILER), CLANG)
|
||||
CCOMMON_OPT += --aarch64-stack-hazard-size=0
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(CORE), CORTEXA53)
|
||||
@@ -303,6 +308,20 @@ FCOMMON_OPT += -march=armv8.3-a
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(CORE), VORTEXM4)
|
||||
ifneq ($(C_COMPILER), GCC)
|
||||
ifeq ($(APPLECLANG),1)
|
||||
CCOMMON_OPT += -march=armv8.4-a+sme
|
||||
else
|
||||
CCOMMON_OPT += -march=armv8.4-a+sme
|
||||
override LDFLAGS += -lclang_rt_builtins-aarch64
|
||||
endif
|
||||
else
|
||||
CCOMMON_OPT += -march=armv8.4-a
|
||||
endif
|
||||
FCOMMON_OPT += -march=armv8.4-a
|
||||
endif
|
||||
|
||||
ifeq (1, $(filter 1,$(GCCVERSIONGTEQ9) $(ISCLANG)))
|
||||
ifeq ($(CORE), TSV110)
|
||||
CCOMMON_OPT += -march=armv8.2-a -mtune=tsv110
|
||||
|
||||
+20
-2
@@ -93,9 +93,27 @@ endif
|
||||
|
||||
ifneq ($(OSNAME), AIX)
|
||||
ifneq ($(NO_LAPACKE), 1)
|
||||
@cp $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapacke.h lapacke_h.tmp
|
||||
ifdef SYMBOLPREFIX
|
||||
@sed 's/LAPACKE_[a-z1-9].[^() ]*/$(SYMBOLPREFIX)&/g' lapacke_h.tmp > lapacke.tmp2
|
||||
@mv lapacke.tmp2 lapacke_h.tmp
|
||||
endif
|
||||
ifdef SYMBOLSUFFIX
|
||||
@sed 's/LAPACKE_[a-z1-9].[^() ]*/&$(SYMBOLSUFFIX)/g' lapacke_h.tmp > lapacke.tmp2
|
||||
@mv lapacke.tmp2 lapacke_h.tmp
|
||||
endif
|
||||
@-install -m644 lapacke_h.tmp "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapacke.h"
|
||||
@echo Copying LAPACKE header files to $(DESTDIR)$(OPENBLAS_INCLUDE_DIR)
|
||||
@-install -m644 $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapack.h "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapack.h"
|
||||
@-install -m644 $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapacke.h "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapacke.h"
|
||||
@cp $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapack.h lapack_h.tmp
|
||||
ifdef SYMBOLPREFIX
|
||||
@sed 's/LAPACK_[a-z1-9]*(\.\.\.)/$(SYMBOLPREFIX)&/g' lapack_h.tmp > lapack.tmp2
|
||||
@mv lapack.tmp2 lapack_h.tmp
|
||||
endif
|
||||
ifdef SYMBOLSUFFIX
|
||||
@sed 's/\(#define $(SYMBOLPREFIX)LAPACK_[a-z1-9].*\)\((...)\)/\1$(SYMBOLSUFFIX)\2/g' lapack_h.tmp > lapack.tmp2
|
||||
@mv lapack.tmp2 lapack_h.tmp
|
||||
endif
|
||||
@-install -m644 lapack_h.tmp "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapack.h"
|
||||
@-install -m644 $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapacke_config.h "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapacke_config.h"
|
||||
@-install -m644 $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapacke_mangling_with_flags.h.in "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapacke_mangling.h"
|
||||
@-install -m644 $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapacke_utils.h "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapacke_utils.h"
|
||||
|
||||
+1
-1
@@ -3,7 +3,7 @@
|
||||
#
|
||||
|
||||
# This library's version
|
||||
VERSION = 0.3.30.dev
|
||||
VERSION = 0.3.32.dev
|
||||
|
||||
# If you set this prefix, the library name will be lib$(LIBNAMESUFFIX)openblas.a
|
||||
# and lib$(LIBNAMESUFFIX)openblas.so, with a matching soname in the shared library
|
||||
|
||||
+8
-1
@@ -331,6 +331,7 @@ HAVE_SSE5=
|
||||
HAVE_AVX=
|
||||
HAVE_AVX2=
|
||||
HAVE_FMA3=
|
||||
HAVE_SME=
|
||||
include $(TOPDIR)/Makefile_kernel.conf
|
||||
endif
|
||||
|
||||
@@ -427,7 +428,7 @@ ifndef MACOSX_DEPLOYMENT_TARGET
|
||||
ifeq ($(ARCH), arm64)
|
||||
export MACOSX_DEPLOYMENT_TARGET=11.0
|
||||
export NO_SVE = 1
|
||||
export NO_SME = 1
|
||||
# export NO_SME = 1
|
||||
else
|
||||
export MACOSX_DEPLOYMENT_TARGET=10.8
|
||||
endif
|
||||
@@ -721,6 +722,11 @@ DYNAMIC_CORE += A64FX
|
||||
endif
|
||||
ifneq ($(NO_SME), 1)
|
||||
DYNAMIC_CORE += ARMV9SME
|
||||
ifeq ($(OSNAME), Darwin)
|
||||
ifneq ($(C_COMPILER), GCC)
|
||||
DYNAMIC_CORE += VORTEXM4
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
DYNAMIC_CORE += THUNDERX
|
||||
DYNAMIC_CORE += THUNDERX2T99
|
||||
@@ -1896,6 +1902,7 @@ ifndef NO_MSA
|
||||
export HAVE_MSA
|
||||
export MSA_FLAGS
|
||||
endif
|
||||
export HAVE_SME
|
||||
export KERNELDIR
|
||||
export FUNCTION_PROFILE
|
||||
export TARGET_CORE
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
CCOMMON_OPT += -msimd128
|
||||
@@ -61,6 +61,9 @@ endif
|
||||
ifeq ($(CORE), SKYLAKEX)
|
||||
ifndef NO_AVX512
|
||||
CCOMMON_OPT += -march=skylake-avx512
|
||||
ifeq ($(C_COMPILER), CLANG)
|
||||
CCOMMON_OPT += -mllvm -exhaustive-register-search
|
||||
endif
|
||||
ifneq ($(F_COMPILER), NAG)
|
||||
FCOMMON_OPT += -march=skylake-avx512
|
||||
endif
|
||||
@@ -93,6 +96,7 @@ ifeq ($(C_COMPILER), GCC)
|
||||
endif
|
||||
endif
|
||||
else ifeq ($(C_COMPILER), CLANG)
|
||||
CCOMMON_OPT += -mllvm -exhaustive-register-search
|
||||
# cooperlake support was added in clang 9
|
||||
ifeq ($(CLANGVERSIONGTEQ9), 1)
|
||||
CCOMMON_OPT += -march=cooperlake
|
||||
@@ -135,6 +139,7 @@ ifeq ($(C_COMPILER), GCC)
|
||||
endif
|
||||
endif
|
||||
else ifeq ($(C_COMPILER), CLANG)
|
||||
CCOMMON_OPT += -mllvm -exhaustive-register-search
|
||||
# sapphire rapids support was added in clang 12
|
||||
ifeq ($(CLANGVERSIONGTEQ12), 1)
|
||||
CCOMMON_OPT += -march=sapphirerapids
|
||||
|
||||
@@ -148,11 +148,12 @@ Please read `GotoBLAS_01Readme.txt` for older CPU models already supported by th
|
||||
- **Intel Haswell**: Optimized Level-3 and Level-2 BLAS with AVX2 and FMA on x86-64.
|
||||
- **Intel Skylake-X**: Optimized Level-3 and Level-2 BLAS with AVX512 and FMA on x86-64.
|
||||
- **Intel Cooper Lake**: as Skylake-X with improved BFLOAT16 support.
|
||||
- **Intel Sapphire Rapids**: as Cooper Lake with improved BFLOAT16 SBGEMM kernel.
|
||||
- **AMD Bobcat**: Used GotoBLAS2 Barcelona codes.
|
||||
- **AMD Bulldozer**: x86-64 ?GEMM FMA4 kernels. (Thanks to Werner Saar)
|
||||
- **AMD PILEDRIVER**: Uses Bulldozer codes with some optimizations.
|
||||
- **AMD STEAMROLLER**: Uses Bulldozer codes with some optimizations.
|
||||
- **AMD ZEN**: Uses Haswell codes with some optimizations for Zen 2/3 (use SkylakeX for Zen4)
|
||||
- **AMD ZEN**: Uses Haswell codes with some optimizations for Zen 2/3, SkylakeX for Zen4, Cooperlake for Zen5
|
||||
|
||||
#### MIPS32
|
||||
|
||||
@@ -186,9 +187,13 @@ Please read `GotoBLAS_01Readme.txt` for older CPU models already supported by th
|
||||
- **EMAG 8180**: preliminary support based on A57
|
||||
- **Neoverse N1**: (AWS Graviton2) preliminary support
|
||||
- **Neoverse V1**: (AWS Graviton3) optimized Level-3 BLAS
|
||||
- **Neoverse N2**: preliminary support
|
||||
- **Neoverse V2**: preliminary support
|
||||
- **Apple Vortex**: preliminary support based on ThunderX2/3
|
||||
- **Apple VortexM4**: preliminary support based on ThunderX2/3, SME kernels for SGEMM,SSYMM,STRMM,SSYRK,SSYR2K
|
||||
- **A64FX**: preliminary support, optimized Level-3 BLAS
|
||||
- **ARMV8SVE**: any ARMV8 cpu with SVE extensions
|
||||
- **ARMV9SME**: any ARMV9 cpu with SVE and SME extensions
|
||||
|
||||
#### PPC/PPC64
|
||||
|
||||
@@ -249,9 +254,15 @@ e.g.:
|
||||
```
|
||||
The old-style TARGET=LOONGSON3R5 is still supported
|
||||
|
||||
#### WASM
|
||||
Not a cpu target in the strict sense, but portable WebAssembly for browser-based applications and the like. See emscripten.org for the compiler and related information
|
||||
|
||||
- **WASM128_GENERIC**: Optimized SGEMM,DGEMM, DAXPY, SSUM/DSUM, SDOT/DDOT and SROT/DROT
|
||||
|
||||
|
||||
### Support for multiple targets in a single library
|
||||
|
||||
OpenBLAS can be built for multiple targets with runtime detection of the target cpu by specifiying `DYNAMIC_ARCH=1` in Makefile.rule, on the gmake command line or as `-DDYNAMIC_ARCH=TRUE` in cmake.
|
||||
OpenBLAS can be built for multiple targets with runtime detection of the target cpu by specifying `DYNAMIC_ARCH=1` in Makefile.rule, on the gmake command line or as `-DDYNAMIC_ARCH=TRUE` in cmake.
|
||||
|
||||
For **x86_64**, the list of targets this activates contains Prescott, Core2, Nehalem, Barcelona, Sandybridge, Bulldozer, Piledriver, Steamroller, Excavator, Haswell, Zen, SkylakeX, Cooper Lake, Sapphire Rapids. For cpu generations not included in this list, the corresponding older model is used. If you also specify `DYNAMIC_OLDER=1`, specific support for Penryn, Dunnington, Opteron, Opteron/SSE3, Bobcat, Atom and Nano is added. Finally there is an option `DYNAMIC_LIST` that allows to specify an individual list of targets to include instead of the default.
|
||||
|
||||
@@ -277,23 +288,29 @@ Please note that it is not possible to combine support for different architectur
|
||||
### Supported OS
|
||||
|
||||
- **GNU/Linux**
|
||||
- **MinGW or Visual Studio (CMake)/Windows**: Please read <https://github.com/xianyi/OpenBLAS/wiki/How-to-use-OpenBLAS-in-Microsoft-Visual-Studio>.
|
||||
- **Darwin/macOS/OSX/iOS**: Experimental. Although GotoBLAS2 already supports Darwin, we are not OSX/iOS experts.
|
||||
- **MinGW or Visual Studio (CMake)/Windows**: Please read <https://github.com/OpenMathLib/OpenBLAS/docs/nstall.md#visual-studio-native-windows-abi>.
|
||||
- **Darwin/macOS/OSX/iOS**: Already supported on PPC and x86 by the original GotoBLAS, now also on ARM64 but we are not OSX/iOS experts.
|
||||
- **FreeBSD**: Supported by the community. We don't actively test the library on this OS.
|
||||
- **OpenBSD**: Supported by the community. We don't actively test the library on this OS.
|
||||
- **NetBSD**: Supported by the community. We don't actively test the library on this OS.
|
||||
- **DragonFly BSD**: Supported by the community. We don't actively test the library on this OS.
|
||||
- **Android**: Supported by the community. Please read <https://github.com/xianyi/OpenBLAS/wiki/How-to-build-OpenBLAS-for-Android>.
|
||||
- **AIX**: Supported on PPC up to POWER10
|
||||
- **Android**: Supported by the community. Please read <https://github.com/OpenMathLib/OpenBLAS/docs/install.md#android>.
|
||||
- **AIX**: Supported on PPC up to POWER10 but testing is increasingly problematic due to lack of publicly available systems
|
||||
- **Haiku**: Supported by the community. We don't actively test the library on this OS.
|
||||
- **SunOS**: Supported by the community. We don't actively test the library on this OS.
|
||||
- **Cortex-M**: Supported by the community. Please read <https://github.com/xianyi/OpenBLAS/wiki/How-to-use-OpenBLAS-on-Cortex-M>.
|
||||
- **Cortex-M**: Supported by the community. Please read <https://github.com/OpenMathLib/OpenBLAS/docs/install.md#cortex-m>.
|
||||
|
||||
## Usage
|
||||
|
||||
Statically link with `libopenblas.a` or dynamically link with `-lopenblas` if OpenBLAS was
|
||||
compiled as a shared library.
|
||||
|
||||
### Considerations for using the library from Java
|
||||
|
||||
The default stack size of only 1MB may be too small, especially if you built OpenBLAS to support larger matrix sizes than provided for by the default settings. Use the -Xss option to request a larger stack size if you encounter problems.
|
||||
|
||||
When a Windows build of OpenBLAS was created using the MINGW gfortran (for the LAPACK parts), the java application may hang on startup due to a deadlock between the gfortran runtime library initialization and any pipes created by a Win11/SBT/Play Framework environment. Use -Djdk.console=jdk.internal.le to work around this.
|
||||
|
||||
### Setting the number of threads using environment variables
|
||||
|
||||
Environment variables are used to specify a maximum number of threads.
|
||||
|
||||
@@ -111,6 +111,7 @@ THUNDERX2T99
|
||||
TSV110
|
||||
THUNDERX3T110
|
||||
VORTEX
|
||||
VORTEXM4
|
||||
A64FX
|
||||
ARMV8SVE
|
||||
ARMV9SME
|
||||
@@ -152,3 +153,7 @@ EV6
|
||||
14.CSKY
|
||||
CSKY
|
||||
CK860FV
|
||||
|
||||
15. WebAssembly/Emscripten:
|
||||
WASM128_GENERIC
|
||||
|
||||
|
||||
@@ -91,6 +91,7 @@ jobs:
|
||||
openblas_utest.exe
|
||||
|
||||
- job: Windows_mingw_gmake
|
||||
timeoutInMinutes: 100
|
||||
pool:
|
||||
vmImage: 'windows-latest'
|
||||
steps:
|
||||
@@ -185,6 +186,7 @@ jobs:
|
||||
variables:
|
||||
LD_LIBRARY_PATH: /usr/local/opt/llvm/lib
|
||||
LIBRARY_PATH: /usr/local/opt/llvm/lib
|
||||
RANLIB: touch
|
||||
steps:
|
||||
- script: |
|
||||
brew update
|
||||
@@ -197,6 +199,7 @@ jobs:
|
||||
variables:
|
||||
LD_LIBRARY_PATH: /usr/local/opt/llvm/lib
|
||||
LIBRARY_PATH: /usr/local/opt/llvm/lib
|
||||
RANLIB: touch
|
||||
steps:
|
||||
- script: |
|
||||
brew update
|
||||
@@ -240,6 +243,7 @@ jobs:
|
||||
LD_LIBRARY_PATH: /usr/local/opt/llvm/lib
|
||||
MACOS_HPCKIT_URL: https://registrationcenter-download.intel.com/akdlm/IRC_NAS/edb4dc2f-266f-47f2-8d56-21bc7764e119/m_HPCKit_p_2023.2.0.49443.dmg
|
||||
LIBRARY_PATH: /usr/local/opt/llvm/lib
|
||||
RANLIB: touch
|
||||
MACOS_FORTRAN_COMPONENTS: intel.oneapi.mac.ifort-compiler
|
||||
steps:
|
||||
- script: |
|
||||
|
||||
+1
-1
@@ -3155,7 +3155,7 @@ bgemv.$(SUFFIX) : gemv.c
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -DBGEMM -UCOMPLEX -UDOUBLE -o $(@F) $^
|
||||
sbgemv.$(SUFFIX) : gemv.c
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -UCOMPLEX -UDOUBLE -o $(@F) $^
|
||||
endif ()
|
||||
endif
|
||||
|
||||
zgemv.$(SUFFIX) : gemv.c
|
||||
$(CC) $(CFLAGS) -c -DCOMPLEX -DDOUBLE -o $(@F) $^
|
||||
|
||||
@@ -23,6 +23,7 @@ config="$2"
|
||||
compiler_name="$3"
|
||||
shift 3
|
||||
flags="$*"
|
||||
is_ios=false
|
||||
|
||||
# First, we need to know the target OS and compiler name
|
||||
{
|
||||
@@ -78,6 +79,7 @@ case "$data" in *OS_CYGWIN_NT*) os=CYGWIN_NT ;; esac
|
||||
case "$data" in *OS_INTERIX*) os=Interix ;; esac
|
||||
case "$data" in *OS_ANDROID*) os=Android ;; esac
|
||||
case "$data" in *OS_HAIKU*) os=Haiku ;; esac
|
||||
case "$data" in *OS_IOS*) is_ios=true ;; esac
|
||||
|
||||
case "$data" in
|
||||
*ARCH_X86_64*) architecture=x86_64 ;;
|
||||
@@ -95,6 +97,7 @@ case "$data" in
|
||||
*ARCH_RISCV64*) architecture=riscv64 ;;
|
||||
*ARCH_LOONGARCH64*) architecture=loongarch64 ;;
|
||||
*ARCH_CSKY*) architecture=csky ;;
|
||||
*ARCH_WASM*) architecture=wasm ;;
|
||||
esac
|
||||
|
||||
defined=0
|
||||
@@ -128,7 +131,7 @@ case "$architecture" in
|
||||
defined=1
|
||||
;;
|
||||
arm|arm64) defined=1 ;;
|
||||
zarch|e2k|alpha|ia64|riscv64|loonarch64)
|
||||
zarch|e2k|alpha|ia64|riscv64|loongarch64|wasm)
|
||||
defined=1
|
||||
BINARY=64
|
||||
;;
|
||||
@@ -252,6 +255,7 @@ case "$data" in
|
||||
*ARCH_ZARCH*) architecture=zarch ;;
|
||||
*ARCH_LOONGARCH64*) architecture=loongarch64 ;;
|
||||
*ARCH_CSKY*) architecture=csky ;;
|
||||
*ARCH_WASM*) architecture=wasm ;;
|
||||
esac
|
||||
|
||||
binformat='bin32'
|
||||
@@ -335,7 +339,14 @@ if [ "$architecture" = "arm64" ]; then
|
||||
fi
|
||||
|
||||
no_sme=0
|
||||
is_appleclang=0
|
||||
if [ "$architecture" = "arm64" ]; then
|
||||
if [ "$compiler" = "CLANG" ]; then
|
||||
data=`$compiler_name --version`
|
||||
case "$data" in Apple*)
|
||||
is_appleclang=1
|
||||
esac
|
||||
fi
|
||||
tmpd=$(mktemp -d 2>/dev/null || mktemp -d -t 'OBC')
|
||||
tmpf="$tmpd/a.S"
|
||||
printf ".text \n.global sme_test\n\nsme_test:\nsmstart\nsmstop\nret\n">> "$tmpf"
|
||||
@@ -410,6 +421,8 @@ fi
|
||||
[ "$os" = "Android" ] && [ "$hostos" = "Linux" ] && [ -n "$TERMUX_APP_PID" ] \
|
||||
&& cross=0
|
||||
|
||||
[ "$is_ios" = true ] && cross=1
|
||||
|
||||
[ "$USE_OPENMP" != 1 ] && openmp=''
|
||||
|
||||
linker_L=""
|
||||
@@ -469,6 +482,7 @@ done
|
||||
[ "$no_avx512bf" -eq 1 ] && printf "NO_AVX512BF16=1\n"
|
||||
[ "$no_avx2" -eq 1 ] && printf "NO_AVX2=1\n"
|
||||
[ "$oldgcc" -eq 1 ] && printf "OLDGCC=1\n"
|
||||
[ "$is_appleclang" -eq 1 ] && printf "APPLECLANG=1\n"
|
||||
exit 0
|
||||
}
|
||||
|
||||
@@ -499,6 +513,7 @@ done
|
||||
[ "$no_avx512bf" -eq 1 ] && printf "NO_AVX512BF16=1\n"
|
||||
[ "$no_avx2" -eq 1 ] && printf "NO_AVX2=1\n"
|
||||
[ "$oldgcc" -eq 1 ] && printf "OLDGCC=1\n"
|
||||
[ "$is_appleclang" -eq 1 ] && printf "APPLECLANG=1\n"
|
||||
[ "$no_lsx" -eq 1 ] && printf "NO_LSX=1\n"
|
||||
[ "$no_lasx" -eq 1 ] && printf "NO_LASX=1\n"
|
||||
} >> "$makefile"
|
||||
|
||||
+7
-2
@@ -40,14 +40,19 @@ if (DYNAMIC_ARCH)
|
||||
endif ()
|
||||
if (${CMAKE_C_COMPILER_VERSION} VERSION_GREATER_EQUAL 14) # SME ACLE supported in GCC >= 14
|
||||
set(DYNAMIC_CORE ${DYNAMIC_CORE} ARMV9SME)
|
||||
endif()
|
||||
if (${CMAKE_C_COMPILER_ID} MATCHES "Clang" AND ${CMAKE_SYSTEM_NAME} STREQUAL "Darwin")
|
||||
set(DYNAMIC_CORE ${DYNAMIC_CORE} VORTEXM4)
|
||||
endif()
|
||||
elseif (${CMAKE_C_COMPILER_ID} MATCHES "Clang")
|
||||
if (${CMAKE_C_COMPILER_VERSION} VERSION_GREATER_EQUAL 11) # SVE ACLE supported in LLVM >= 11
|
||||
set(DYNAMIC_CORE ${DYNAMIC_CORE} NEOVERSEV1 NEOVERSEN2 ARMV8SVE A64FX)
|
||||
endif ()
|
||||
if (${CMAKE_C_COMPILER_VERSION} VERSION_GREATER_EQUAL 19) # SME ACLE supported in LLVM >= 19
|
||||
set(DYNAMIC_CORE ${DYNAMIC_CORE} ARMV9SME)
|
||||
if (NOT ${CMAKE_SYSTEM_NAME} STREQUAL "Windows")
|
||||
if (${CMAKE_C_COMPILER_VERSION} VERSION_GREATER_EQUAL 19 OR (${CMAKE_C_COMPILER_ID} MATCHES AppleClang AND ${CMAKE_C_COMPILER_VERSION} VERSION_GREATER_EQUAL 17) ) # SME ACLE supported in LLVM >= 19 and AppleClang >= 17
|
||||
set(DYNAMIC_CORE ${DYNAMIC_CORE} ARMV9SME VORTEXM4)
|
||||
endif()
|
||||
endif()
|
||||
endif ()
|
||||
if (DYNAMIC_LIST)
|
||||
set(DYNAMIC_CORE ARMV8 ${DYNAMIC_LIST})
|
||||
|
||||
@@ -315,7 +315,24 @@ if (${CORE} STREQUAL ARMV9SME)
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -tp=host")
|
||||
else ()
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -march=armv9-a+sme")
|
||||
if (${OSNAME} STREQUAL Windows AND ${CMAKE_C_COMPILER_ID} MATCHES "Clang" )
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} --aarch64-stack-hazard-size=0")
|
||||
endif ()
|
||||
endif ()
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (${CORE} STREQUAL VORTEXM4)
|
||||
if (NOT DYNAMIC_ARCH)
|
||||
if (${CMAKE_C_COMPILER_ID} STREQUAL "NVC" AND NOT NO_SVE)
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -tp=host")
|
||||
else ()
|
||||
if (${CMAKE_C_COMPILER_ID} STREQUAL "AppleClang")
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -march=armv8.4-a+sme -mcpu=apple-m4")
|
||||
else ()
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -march=armv8.4-a -mcpu=apple-m4")
|
||||
endif ()
|
||||
endif ()
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
|
||||
+1
-1
@@ -128,7 +128,7 @@ if (${F_COMPILER} STREQUAL "INTEL" OR CMAKE_Fortran_COMPILER_ID MATCHES "Intel")
|
||||
endif ()
|
||||
set(FCOMMON_OPT "${FCOMMON_OPT} -recursive -fp-model=consistent")
|
||||
if (USE_OPENMP)
|
||||
set(OpenMP_Fortran_FLAGS "-openmp" CACHE STRING "OpenMP Fortran compiler flags")
|
||||
set(OpenMP_Fortran_FLAGS "-qopenmp" CACHE STRING "OpenMP Fortran compiler flags")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
|
||||
+8
-6
@@ -71,7 +71,7 @@ set(SLASRC
|
||||
slaqr0.f slaqr1.f slaqr2.f slaqr3.f slaqr4.f slaqr5.f
|
||||
slaqtr.f slar1v.f slar2v.f ilaslr.f ilaslc.f
|
||||
slarf.f slarfb.f slarfb_gett.f slarfg.f slarfgp.f slarft.f slarfx.f slarfy.f slargv.f
|
||||
slarrv.f slartv.f
|
||||
slarf1f.f slarf1l.f slarrv.f slartv.f
|
||||
slarz.f slarzb.f slarzt.f slasy2.f
|
||||
slasyf.f slasyf_rook.f slasyf_rk.f slasyf_aa.f
|
||||
slatbs.f slatdf.f slatps.f slatrd.f slatrs.f slatrz.f
|
||||
@@ -178,6 +178,7 @@ set(CLASRC
|
||||
claqz0.f claqz1.f claqz2.f claqz3.f
|
||||
claqsp.f claqsy.f clar1v.f clar2v.f ilaclr.f ilaclc.f
|
||||
clarf.f clarfb.f clarfb_gett.f clarfg.f clarfgp.f clarft.f
|
||||
clarf1f.f clarf1l.f
|
||||
clarfx.f clarfy.f clargv.f clarnv.f clarrv.f clartg.f90 clartv.f
|
||||
clarz.f clarzb.f clarzt.f clascl.f claset.f clasr.f classq.f90
|
||||
clasyf.f clasyf_rook.f clasyf_rk.f clasyf_aa.f
|
||||
@@ -262,7 +263,7 @@ set(DLASRC
|
||||
dlaqr0.f dlaqr1.f dlaqr2.f dlaqr3.f dlaqr4.f dlaqr5.f
|
||||
dlaqtr.f dlar1v.f dlar2v.f iladlr.f iladlc.f
|
||||
dlarf.f dlarfb.f dlarfb_gett.f dlarfg.f dlarfgp.f dlarft.f dlarfx.f dlarfy.f
|
||||
dlargv.f dlarrv.f dlartv.f
|
||||
dlarf1f.f dlarf1l.f dlargv.f dlarrv.f dlartv.f
|
||||
dlarz.f dlarzb.f dlarzt.f dlasy2.f
|
||||
dlasyf.f dlasyf_rook.f dlasyf_rk.f dlasyf_aa.f
|
||||
dlatbs.f dlatdf.f dlatps.f dlatrd.f dlatrs.f dlatrz.f
|
||||
@@ -371,7 +372,7 @@ set(ZLASRC
|
||||
zlaqr0.f zlaqr1.f zlaqr2.f zlaqr3.f zlaqr4.f zlaqr5.f
|
||||
zlaqsp.f zlaqsy.f zlar1v.f zlar2v.f ilazlr.f ilazlc.f
|
||||
zlarcm.f zlarf.f zlarfb.f zlarfb_gett.f
|
||||
zlarfg.f zlarfgp.f zlarft.f
|
||||
zlarfg.f zlarfgp.f zlarft.f zlarf1f.f zlarf1l.f
|
||||
zlarfx.f zlarfy.f zlargv.f zlarnv.f zlarrv.f zlartg.f90 zlartv.f
|
||||
zlarz.f zlarzb.f zlarzt.f zlascl.f zlaset.f zlasr.f
|
||||
zlassq.f90 zlasyf.f zlasyf_rook.f zlasyf_rk.f zlasyf_aa.f
|
||||
@@ -575,7 +576,7 @@ set(SLASRC
|
||||
slaqr0.c slaqr1.c slaqr2.c slaqr3.c slaqr4.c slaqr5.c
|
||||
slaqtr.c slar1v.c slar2v.c ilaslr.c ilaslc.c
|
||||
slarf.c slarfb.c slarfb_gett.c slarfg.c slarfgp.c slarft.c slarfx.c slarfy.c slargv.c
|
||||
slarrv.c slartv.c
|
||||
slarf1f.c slarf1l.c slarrv.c slartv.c
|
||||
slarz.c slarzb.c slarzt.c slasy2.c
|
||||
slasyf.c slasyf_rook.c slasyf_rk.c slasyf_aa.c
|
||||
slatbs.c slatdf.c slatps.c slatrd.c slatrs.c slatrz.c
|
||||
@@ -681,6 +682,7 @@ set(CLASRC
|
||||
claqr0.c claqr1.c claqr2.c claqr3.c claqr4.c claqr5.c
|
||||
claqsp.c claqsy.c clar1v.c clar2v.c ilaclr.c ilaclc.c
|
||||
clarf.c clarfb.c clarfb_gett.c clarfg.c clarfgp.c clarft.c
|
||||
clarf1f.c clarf1l.c
|
||||
clarfx.c clarfy.c clargv.c clarnv.c clarrv.c clartg.c clartv.c
|
||||
clarz.c clarzb.c clarzt.c clascl.c claset.c clasr.c classq.c
|
||||
clasyf.c clasyf_rook.c clasyf_rk.c clasyf_aa.c
|
||||
@@ -764,7 +766,7 @@ set(DLASRC
|
||||
dlaqr0.c dlaqr1.c dlaqr2.c dlaqr3.c dlaqr4.c dlaqr5.c
|
||||
dlaqtr.c dlar1v.c dlar2v.c iladlr.c iladlc.c
|
||||
dlarf.c dlarfb.c dlarfb_gett.c dlarfg.c dlarfgp.c dlarft.c dlarfx.c dlarfy.c
|
||||
dlargv.c dlarrv.c dlartv.c
|
||||
dlarf1f.c dlarf1l.c dlargv.c dlarrv.c dlartv.c
|
||||
dlarz.c dlarzb.c dlarzt.c dlasy2.c
|
||||
dlasyf.c dlasyf_rook.c dlasyf_rk.c dlasyf_aa.c
|
||||
dlatbs.c dlatdf.c dlatps.c dlatrd.c dlatrs.c dlatrz.c
|
||||
@@ -871,7 +873,7 @@ set(ZLASRC
|
||||
zlaqhb.c zlaqhe.c zlaqhp.c zlaqp2.c zlaqp2rk.c zlaqp3rk.c zlaqps.c zlaqsb.c
|
||||
zlaqr0.c zlaqr1.c zlaqr2.c zlaqr3.c zlaqr4.c zlaqr5.c
|
||||
zlaqsp.c zlaqsy.c zlar1v.c zlar2v.c ilazlr.c ilazlc.c
|
||||
zlarcm.c zlarf.c zlarfb.c zlarfb_gett.c
|
||||
zlarcm.c zlarf.c zlarfb.c zlarfb_gett.c zlarf1f.c zlarf1l.c
|
||||
zlarfg.c zlarfgp.c zlarft.c
|
||||
zlarfx.c zlarfy.c zlargv.c zlarnv.c zlarrv.c zlartg.c zlartv.c
|
||||
zlarz.c zlarzb.c zlarzt.c zlascl.c zlaset.c zlasr.c
|
||||
|
||||
+16
-1
@@ -98,6 +98,10 @@ if (${COMPILER_ID} STREQUAL "GNU")
|
||||
set(COMPILER_ID "GCC")
|
||||
endif ()
|
||||
|
||||
if (HOST_OS STREQUAL "EMSCRIPTEN")
|
||||
set (ARCH wasm)
|
||||
endif()
|
||||
|
||||
string(TOUPPER ${ARCH} UC_ARCH)
|
||||
file(WRITE ${TARGET_CONF_TEMP}
|
||||
"#define OS_${HOST_OS}\t1\n"
|
||||
@@ -1255,7 +1259,7 @@ endif ()
|
||||
set(ZGEMM_UNROLL_M 4)
|
||||
set(ZGEMM_UNROLL_N 4)
|
||||
set(SYMV_P 16)
|
||||
elseif ("${TCORE}" STREQUAL "VORTEX")
|
||||
elseif ("${TCORE}" STREQUAL "VORTEX" OR "${TCORE}" STREQUAL "VORTEXM4")
|
||||
file(APPEND ${TARGET_CONF_TEMP}
|
||||
"#define ARMV8\n"
|
||||
"#define L1_CODE_SIZE\t32768\n"
|
||||
@@ -1500,6 +1504,15 @@ endif ()
|
||||
"#define DTB_DEFAULT_ENTRIES 128\n"
|
||||
"#define DTB_SIZE 4096\n"
|
||||
"#define L2_ASSOCIATIVE 4\n")
|
||||
elseif ("${TCORE}" STREQUAL "WASM128_GENERIC")
|
||||
file(APPEND ${TARGET_CONF_TEMP}
|
||||
"#define L1_DATA_SIZE 32768\n"
|
||||
"#define L1_DATA_LINESIZE 32\n"
|
||||
"#define L2_SIZE 1048576\n"
|
||||
"#define L2_LINESIZE 32 \n"
|
||||
"#define DTB_DEFAULT_ENTRIES 128\n"
|
||||
"#define DTB_SIZE 4096\n"
|
||||
"#define L2_ASSOCIATIVE 4\n")
|
||||
elseif ("${TCORE}" STREQUAL "LA64_GENERIC")
|
||||
file(APPEND ${TARGET_CONF_TEMP}
|
||||
"#define DTB_DEFAULT_ENTRIES 64\n")
|
||||
@@ -1639,6 +1652,8 @@ else(NOT CMAKE_CROSSCOMPILING)
|
||||
unset (HAVE_VFP)
|
||||
unset (HAVE_VFPV3)
|
||||
unset (HAVE_VFPV4)
|
||||
unset (HAVE_SVE)
|
||||
unset (HAVE_SME)
|
||||
message(STATUS "Running getarch")
|
||||
|
||||
# use the cmake binary w/ the -E param to run a shell command in a cross-platform way
|
||||
|
||||
@@ -367,11 +367,21 @@ if (${TARGET} STREQUAL NEOVERSEV1)
|
||||
endif()
|
||||
if (${TARGET} STREQUAL ARMV9SME)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=armv9-a+sme -O3")
|
||||
if (${CMAKE_SYSTEM_NAME} STREQUAL Windows AND ${CMAKE_C_COMPILER_ID} MATCHES "Clang")
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} --aarch64-stack-hazard-size=0")
|
||||
endif()
|
||||
endif()
|
||||
if (${TARGET} STREQUAL VORTEXM4)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=armv8.4-a+sme -O3")
|
||||
if (${CMAKE_SYSTEM_NAME} STREQUAL Windows AND ${CMAKE_C_COMPILER_ID} MATCHES "Clang")
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} --aarch64-stack-hazard-size=0")
|
||||
endif()
|
||||
endif()
|
||||
if (${TARGET} STREQUAL A64FX)
|
||||
if (${CMAKE_C_COMPILER_ID} STREQUAL "PGI" AND NOT NO_SVE)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -Msve-intrinsics -march=armv8.2-a+sve -mtune=a64fx")
|
||||
else ()
|
||||
set (GCC_VERSION 0.0)
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -dumpversion OUTPUT_VARIABLE GCC_VERSION)
|
||||
if (${GCC_VERSION} VERSION_GREATER 10.4 OR ${GCC_VERSION} VERSION_EQUAL 10.4)
|
||||
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=armv8.2-a+sve -mtune=a64fx")
|
||||
@@ -869,6 +879,10 @@ if (DEFINED ARCH)
|
||||
set(USE_GEMM3M 1)
|
||||
endif ()
|
||||
|
||||
if (EMSCRIPTEN)
|
||||
set(USE_GEMM3M 0)
|
||||
endif ()
|
||||
|
||||
if (${CORE} STREQUAL "generic")
|
||||
set(USE_GEMM3M 0)
|
||||
endif ()
|
||||
|
||||
@@ -40,6 +40,8 @@ if(CMAKE_CL_64 OR MINGW64)
|
||||
else()
|
||||
set(X86_64 1)
|
||||
endif()
|
||||
elseif(OS_EMSCRIPTEN)
|
||||
set(WASM 1)
|
||||
elseif(MINGW OR (MSVC AND NOT CMAKE_CROSSCOMPILING))
|
||||
set(X86 1)
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "ppc.*|power.*|Power.*" OR (CMAKE_SYSTEM_NAME MATCHES "Darwin" AND CMAKE_OSX_ARCHITECTURES MATCHES "ppc.*"))
|
||||
@@ -145,6 +147,15 @@ endif()
|
||||
endif()
|
||||
|
||||
if (ARM64)
|
||||
if (NOT NO_SVE)
|
||||
file(WRITE ${PROJECT_BINARY_DIR}/sve.c "#include <arm_sve.h>\n\n int main(void){}\n")
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -march=armv8-a+sve -c -o ${PROJECT_BINARY_DIR}/sve.o ${PROJECT_BINARY_DIR}/sve.c OUTPUT_QUIET ERROR_QUIET RESULT_VARIABLE NO_SVE)
|
||||
if (NO_SVE EQUAL 1)
|
||||
set (CCOMMON_OPT "${CCOMMON_OPT} -DNO_SVE")
|
||||
endif()
|
||||
file(REMOVE "${PROJECT_BINARY_DIR}/sve.c" "${PROJECT_BINARY_DIR}/sve.o")
|
||||
endif()
|
||||
|
||||
if (NOT NO_SME)
|
||||
file(WRITE ${PROJECT_BINARY_DIR}/sme.c ".text \n.global sme_test\n\nsme_test:\nsmstart\nsmstop\nret\n")
|
||||
execute_process(COMMAND ${CMAKE_C_COMPILER} -march=armv9-a+sve2+sme -c -v -o ${PROJECT_BINARY_DIR}/sme.o ${PROJECT_BINARY_DIR}/sme.c OUTPUT_QUIET ERROR_QUIET RESULT_VARIABLE NO_SME)
|
||||
|
||||
+1
-1
@@ -51,7 +51,7 @@ macro(ParseMakefileVars MAKEFILE_IN)
|
||||
if (${OSNAME} STREQUAL Windows)
|
||||
set (OSNAME WINNT)
|
||||
endif ()
|
||||
message(STATUS OS ${OSNAME} COMPILER ${C_COMPILER})
|
||||
#message(STATUS OS ${OSNAME} COMPILER ${C_COMPILER})
|
||||
set (IfElse 0)
|
||||
set (ElseSeen 0)
|
||||
set (SkipIfs 0)
|
||||
|
||||
@@ -362,18 +362,6 @@ typedef int blasint;
|
||||
#define MAX_CPU_NUMBER 2
|
||||
#endif
|
||||
|
||||
#if defined(OS_SUNOS)
|
||||
#define YIELDING thr_yield()
|
||||
#endif
|
||||
|
||||
#if defined(OS_WINDOWS)
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
#define YIELDING YieldProcessor()
|
||||
#else
|
||||
#define YIELDING SwitchToThread()
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#if defined(ARMV7) || defined(ARMV6) || defined(ARMV8) || defined(ARMV5)
|
||||
#define YIELDING __asm__ __volatile__ ("nop;nop;nop;nop;nop;nop;nop;nop; \n");
|
||||
#endif
|
||||
@@ -398,14 +386,28 @@ typedef int blasint;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
||||
#ifdef __EMSCRIPTEN__
|
||||
#if defined(ARCH_WASM)
|
||||
#ifndef YIELDING
|
||||
#define YIELDING
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
#undef YIELDING // MSVC doesn't support assembly code
|
||||
#define YIELDING YieldProcessor()
|
||||
#endif
|
||||
|
||||
#ifndef YIELDING
|
||||
#if defined(OS_SUNOS)
|
||||
#define YIELDING thr_yield()
|
||||
|
||||
#elif defined(OS_WINDOWS)
|
||||
#define YIELDING SwitchToThread()
|
||||
|
||||
#else // assume POSIX.1-2008
|
||||
#define YIELDING sched_yield()
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/***
|
||||
To alloc job_t on heap or stack.
|
||||
@@ -498,6 +500,10 @@ please https://github.com/xianyi/OpenBLAS/issues/246
|
||||
#include "common_csky.h"
|
||||
#endif
|
||||
|
||||
#ifdef ARCH_WASM
|
||||
#include "common_wasm.h"
|
||||
#endif
|
||||
|
||||
#ifndef ASSEMBLER
|
||||
#ifdef OS_WINDOWSSTORE
|
||||
typedef char env_var_t[MAX_PATH];
|
||||
@@ -765,7 +771,7 @@ static __inline int readenv_atoi(char *env) {
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
#ifdef OS_WINDOWS
|
||||
#if defined(OS_WINDOWS) && !defined(OS_CYGWIN_NT)
|
||||
static __inline int readenv_atoi(char *env) {
|
||||
env_var_t p;
|
||||
return readenv(p,env) ? 0 : atoi(p);
|
||||
|
||||
@@ -110,6 +110,31 @@ void ssyrk_direct_alpha_betaLT(BLASLONG N, BLASLONG K,
|
||||
float beta,
|
||||
float * C, BLASLONG strideC);
|
||||
|
||||
void ssyr2k_direct_alpha_betaUN(BLASLONG N, BLASLONG K,
|
||||
float alpha,
|
||||
float * A, BLASLONG strideA,
|
||||
float * B, BLASLONG strideB,
|
||||
float beta,
|
||||
float * R, BLASLONG strideR);
|
||||
void ssyr2k_direct_alpha_betaUT(BLASLONG N, BLASLONG K,
|
||||
float alpha,
|
||||
float * A, BLASLONG strideA,
|
||||
float * B, BLASLONG strideB,
|
||||
float beta,
|
||||
float * R, BLASLONG strideR);
|
||||
void ssyr2k_direct_alpha_betaLN(BLASLONG N, BLASLONG K,
|
||||
float alpha,
|
||||
float * A, BLASLONG strideA,
|
||||
float * B, BLASLONG strideB,
|
||||
float beta,
|
||||
float * R, BLASLONG strideR);
|
||||
void ssyr2k_direct_alpha_betaLT(BLASLONG N, BLASLONG K,
|
||||
float alpha,
|
||||
float * A, BLASLONG strideA,
|
||||
float * B, BLASLONG strideB,
|
||||
float beta,
|
||||
float * R, BLASLONG strideR);
|
||||
|
||||
int sgemm_direct_performant(BLASLONG M, BLASLONG N, BLASLONG K);
|
||||
|
||||
int shgemm_beta(BLASLONG, BLASLONG, BLASLONG, float,
|
||||
|
||||
@@ -3159,6 +3159,8 @@ typedef struct {
|
||||
#define NEG_TCOPY ZNEG_TCOPY
|
||||
#define LARF_L ZLARF_L
|
||||
#define LARF_R ZLARF_R
|
||||
#define LAED3_SINGLE dlaed3_single
|
||||
#define LAED3_PARALLEL dlaed3_parallel
|
||||
#else
|
||||
#define GETF2 CGETF2
|
||||
#define GETRF CGETRF
|
||||
@@ -3180,6 +3182,8 @@ typedef struct {
|
||||
#define NEG_TCOPY CNEG_TCOPY
|
||||
#define LARF_L CLARF_L
|
||||
#define LARF_R CLARF_R
|
||||
#define LAED3_SINGLE slaed3_single
|
||||
#define LAED3_PARALLEL slaed3_parallel
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
||||
@@ -47,6 +47,9 @@
|
||||
typedef struct {
|
||||
int dtb_entries;
|
||||
int switch_ratio;
|
||||
int divide_rate;
|
||||
int divide_limit;
|
||||
int preferred_size;
|
||||
int offsetA, offsetB, align;
|
||||
#if BUILD_HFLOAT16 == 1
|
||||
int shgemm_p, shgemm_q, shgemm_r;
|
||||
@@ -257,6 +260,7 @@ int (*shgemv_t) (BLASLONG, BLASLONG, float, hfloat16 *, BLASLONG, hfloat16 *, BL
|
||||
#endif
|
||||
#ifdef ARCH_ARM64
|
||||
void (*sgemm_direct) (BLASLONG, BLASLONG, BLASLONG, float *, BLASLONG , float *, BLASLONG , float * , BLASLONG);
|
||||
int (*sgemm_direct_performant) (BLASLONG M, BLASLONG N, BLASLONG K);
|
||||
void (*sgemm_direct_alpha_beta) (BLASLONG, BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float * , BLASLONG);
|
||||
void (*ssymm_direct_alpha_betaLU) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float * , BLASLONG);
|
||||
void (*ssymm_direct_alpha_betaLL) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float * , BLASLONG);
|
||||
@@ -268,6 +272,10 @@ int (*shgemv_t) (BLASLONG, BLASLONG, float, hfloat16 *, BLASLONG, hfloat16 *, BL
|
||||
void (*ssyrk_direct_alpha_betaUT) (BLASLONG, BLASLONG, float, float *, BLASLONG, float, float *, BLASLONG);
|
||||
void (*ssyrk_direct_alpha_betaLN) (BLASLONG, BLASLONG, float, float *, BLASLONG, float, float *, BLASLONG);
|
||||
void (*ssyrk_direct_alpha_betaLT) (BLASLONG, BLASLONG, float, float *, BLASLONG, float, float *, BLASLONG);
|
||||
void (*ssyr2k_direct_alpha_betaUN) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float *, BLASLONG);
|
||||
void (*ssyr2k_direct_alpha_betaUT) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float *, BLASLONG);
|
||||
void (*ssyr2k_direct_alpha_betaLN) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float *, BLASLONG);
|
||||
void (*ssyr2k_direct_alpha_betaLT) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float *, BLASLONG);
|
||||
#endif
|
||||
|
||||
|
||||
|
||||
+9
-1
@@ -60,6 +60,10 @@
|
||||
#define SSYRK_DIRECT_ALPHA_BETA_UT ssyrk_direct_alpha_betaUT
|
||||
#define SSYRK_DIRECT_ALPHA_BETA_LN ssyrk_direct_alpha_betaLN
|
||||
#define SSYRK_DIRECT_ALPHA_BETA_LT ssyrk_direct_alpha_betaLT
|
||||
#define SSYR2K_DIRECT_ALPHA_BETA_UN ssyr2k_direct_alpha_betaUN
|
||||
#define SSYR2K_DIRECT_ALPHA_BETA_UT ssyr2k_direct_alpha_betaUT
|
||||
#define SSYR2K_DIRECT_ALPHA_BETA_LN ssyr2k_direct_alpha_betaLN
|
||||
#define SSYR2K_DIRECT_ALPHA_BETA_LT ssyr2k_direct_alpha_betaLT
|
||||
|
||||
#define SGEMM_ONCOPY sgemm_oncopy
|
||||
#define SGEMM_OTCOPY sgemm_otcopy
|
||||
@@ -227,7 +231,7 @@
|
||||
#define SGEMM_DIRECT_PERFORMANT gotoblas -> sgemm_direct_performant
|
||||
#define SGEMM_DIRECT gotoblas -> sgemm_direct
|
||||
#elif ARCH_ARM64
|
||||
#define SGEMM_DIRECT_PERFORMANT sgemm_direct_performant
|
||||
#define SGEMM_DIRECT_PERFORMANT gotoblas -> sgemm_direct_performant
|
||||
#define SGEMM_DIRECT gotoblas -> sgemm_direct
|
||||
#define SGEMM_DIRECT_ALPHA_BETA gotoblas -> sgemm_direct_alpha_beta
|
||||
#define SSYMM_DIRECT_ALPHA_BETA_LU gotoblas -> ssymm_direct_alpha_betaLU
|
||||
@@ -240,6 +244,10 @@
|
||||
#define SSYRK_DIRECT_ALPHA_BETA_UT gotoblas -> ssyrk_direct_alpha_betaUT
|
||||
#define SSYRK_DIRECT_ALPHA_BETA_LN gotoblas -> ssyrk_direct_alpha_betaLN
|
||||
#define SSYRK_DIRECT_ALPHA_BETA_LT gotoblas -> ssyrk_direct_alpha_betaLT
|
||||
#define SSYR2K_DIRECT_ALPHA_BETA_UN gotoblas -> ssyr2k_direct_alpha_betaUN
|
||||
#define SSYR2K_DIRECT_ALPHA_BETA_UT gotoblas -> ssyr2k_direct_alpha_betaUT
|
||||
#define SSYR2K_DIRECT_ALPHA_BETA_LN gotoblas -> ssyr2k_direct_alpha_betaLN
|
||||
#define SSYR2K_DIRECT_ALPHA_BETA_LT gotoblas -> ssyr2k_direct_alpha_betaLT
|
||||
#endif
|
||||
|
||||
#define SGEMM_ONCOPY gotoblas -> sgemm_oncopy
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
/*****************************************************************************
|
||||
Copyright (c) 2011-2014, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
met:
|
||||
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written
|
||||
permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
|
||||
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
**********************************************************************************/
|
||||
|
||||
/*********************************************************************/
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
/* Redistribution and use in source and binary forms, with or */
|
||||
/* without modification, are permitted provided that the following */
|
||||
/* conditions are met: */
|
||||
/* */
|
||||
/* 1. Redistributions of source code must retain the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer. */
|
||||
/* */
|
||||
/* 2. Redistributions in binary form must reproduce the above */
|
||||
/* copyright notice, this list of conditions and the following */
|
||||
/* disclaimer in the documentation and/or other materials */
|
||||
/* provided with the distribution. */
|
||||
/* */
|
||||
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
|
||||
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
|
||||
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
|
||||
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
|
||||
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
|
||||
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
|
||||
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
|
||||
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
|
||||
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
|
||||
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
|
||||
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
|
||||
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
|
||||
/* POSSIBILITY OF SUCH DAMAGE. */
|
||||
/* */
|
||||
/* The views and conclusions contained in the software and */
|
||||
/* documentation are those of the authors and should not be */
|
||||
/* interpreted as representing official policies, either expressed */
|
||||
/* or implied, of The University of Texas at Austin. */
|
||||
/*********************************************************************/
|
||||
|
||||
#ifndef COMMON_WASM
|
||||
#define COMMON_WASM
|
||||
|
||||
#define MB __sync_synchronize()
|
||||
#define WMB __sync_synchronize()
|
||||
#define RMB __sync_synchronize()
|
||||
|
||||
#ifndef ASSEMBLER
|
||||
|
||||
|
||||
static inline int blas_quickdivide(blasint x, blasint y){
|
||||
return x / y;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
#define BUFFER_SIZE ( 16 << 20)
|
||||
#define SEEK_ADDRESS
|
||||
|
||||
#endif
|
||||
|
||||
+31
-3
@@ -82,6 +82,7 @@ size_t length64=sizeof(value64);
|
||||
#define CPU_AMPERE1 25
|
||||
// Apple
|
||||
#define CPU_VORTEX 13
|
||||
#define CPU_VORTEXM4 26
|
||||
// Fujitsu
|
||||
#define CPU_A64FX 15
|
||||
// Phytium
|
||||
@@ -113,7 +114,8 @@ static char *cpuname[] = {
|
||||
"FT2000",
|
||||
"CORTEXA76",
|
||||
"NEOVERSEV2",
|
||||
"AMPERE1"
|
||||
"AMPERE1",
|
||||
"VORTEXM4",
|
||||
};
|
||||
|
||||
static char *cpuname_lower[] = {
|
||||
@@ -143,7 +145,7 @@ static char *cpuname_lower[] = {
|
||||
"cortexa76",
|
||||
"neoversev2",
|
||||
"ampere1",
|
||||
"ampere1a"
|
||||
"vortexm4"
|
||||
};
|
||||
|
||||
static int cpulowperf=0;
|
||||
@@ -321,6 +323,8 @@ int detect(void)
|
||||
return CPU_CORTEXX2;
|
||||
else if (strstr(cpu_part, "0xd4f")) //NVIDIA Grace et al.
|
||||
return CPU_NEOVERSEV2;
|
||||
else if (strstr(cpu_part, "0xd87") || strstr(cpu_part, "0xd85") || strstr(cpu_part, "0xd83")) // X925/A725
|
||||
return CPU_NEOVERSEV2;
|
||||
else if (strstr(cpu_part, "0xd0b"))
|
||||
return CPU_CORTEXA76;
|
||||
}
|
||||
@@ -402,7 +406,8 @@ int detect(void)
|
||||
if (value64 ==131287967|| value64 == 458787763 ) return CPU_VORTEX; //A12/M1
|
||||
if (value64 == 3660830781) return CPU_VORTEX; //A15/M2
|
||||
if (value64 == 2271604202) return CPU_VORTEX; //A16/M3
|
||||
if (value64 == 1867590060) return CPU_VORTEX; //M4
|
||||
if (value64 == 1867590060) return CPU_VORTEXM4; //M4
|
||||
if (value64 == 492472296) return CPU_VORTEXM4; //M5
|
||||
#else
|
||||
#ifdef OS_WINDOWS
|
||||
HKEY reghandle;
|
||||
@@ -749,6 +754,29 @@ void get_cpuconfig(void)
|
||||
break;
|
||||
case CPU_VORTEX:
|
||||
printf("#define VORTEX \n");
|
||||
#ifdef __APPLE__
|
||||
length64 = sizeof(value64);
|
||||
sysctlbyname("hw.l1icachesize",&value64,&length64,NULL,0);
|
||||
printf("#define L1_CODE_SIZE %lld \n",value64);
|
||||
length64 = sizeof(value64);
|
||||
sysctlbyname("hw.cachelinesize",&value64,&length64,NULL,0);
|
||||
printf("#define L1_CODE_LINESIZE %lld \n",value64);
|
||||
printf("#define L1_DATA_LINESIZE %lld \n",value64);
|
||||
length64 = sizeof(value64);
|
||||
sysctlbyname("hw.l1dcachesize",&value64,&length64,NULL,0);
|
||||
printf("#define L1_DATA_SIZE %lld \n",value64);
|
||||
length64 = sizeof(value64);
|
||||
sysctlbyname("hw.l2cachesize",&value64,&length64,NULL,0);
|
||||
printf("#define L2_SIZE %lld \n",value64);
|
||||
#endif
|
||||
printf("#define DTB_DEFAULT_ENTRIES 64 \n");
|
||||
printf("#define DTB_SIZE 4096 \n");
|
||||
break;
|
||||
case CPU_VORTEXM4:
|
||||
printf("#define VORTEXM4 \n");
|
||||
#ifdef __clang__
|
||||
printf("#define HAVE_SME 1 \n");
|
||||
#endif
|
||||
#ifdef __APPLE__
|
||||
length64 = sizeof(value64);
|
||||
sysctlbyname("hw.l1icachesize",&value64,&length64,NULL,0);
|
||||
|
||||
+40
-41
@@ -1,5 +1,5 @@
|
||||
/*****************************************************************************
|
||||
Copyright (c) 2011-2014, The OpenBLAS Project
|
||||
Copyright (c) 2011-2026, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
@@ -13,9 +13,9 @@ met:
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written
|
||||
permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
@@ -109,7 +109,7 @@ int detect(void){
|
||||
return CPU_1004K;
|
||||
} else if (strstr(p, " 24K")) {
|
||||
return CPU_24K;
|
||||
} else
|
||||
} else
|
||||
return CPU_UNKNOWN;
|
||||
}
|
||||
#endif
|
||||
@@ -136,6 +136,40 @@ void get_subdirname(void){
|
||||
printf("mips");
|
||||
}
|
||||
|
||||
int get_feature(char *search) {
|
||||
|
||||
#ifdef __linux
|
||||
FILE *infile;
|
||||
char buffer[2048], *p, *t;
|
||||
p = (char *)NULL;
|
||||
|
||||
infile = fopen("/proc/cpuinfo", "r");
|
||||
|
||||
while (fgets(buffer, sizeof(buffer), infile)) {
|
||||
|
||||
if (!strncmp("Features", buffer, 8) ||
|
||||
!strncmp("ASEs implemented", buffer, 16)) {
|
||||
p = strchr(buffer, ':') + 2;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
fclose(infile);
|
||||
|
||||
if (p == NULL)
|
||||
return 0;
|
||||
|
||||
t = strtok(p, " ");
|
||||
while (t = strtok(NULL, " ")) {
|
||||
if (strstr(t, search)) {
|
||||
return (1);
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
return (0);
|
||||
}
|
||||
|
||||
void get_cpuconfig(void){
|
||||
if(detect()==CPU_P5600){
|
||||
printf("#define P5600\n");
|
||||
@@ -165,7 +199,7 @@ void get_cpuconfig(void){
|
||||
}else{
|
||||
printf("#define UNKNOWN\n");
|
||||
}
|
||||
#ifndef NO_MSA
|
||||
#ifndef NO_MSA
|
||||
if (get_feature("msa")) printf("#define HAVE_MSA\n");
|
||||
#endif
|
||||
}
|
||||
@@ -181,38 +215,3 @@ void get_libname(void){
|
||||
printf("mips\n");
|
||||
}
|
||||
}
|
||||
|
||||
int get_feature(char *search)
|
||||
{
|
||||
|
||||
#ifdef __linux
|
||||
FILE *infile;
|
||||
char buffer[2048], *p,*t;
|
||||
p = (char *) NULL ;
|
||||
|
||||
infile = fopen("/proc/cpuinfo", "r");
|
||||
|
||||
while (fgets(buffer, sizeof(buffer), infile))
|
||||
{
|
||||
|
||||
if (!strncmp("Features", buffer, 8) || !strncmp("ASEs implemented", buffer, 16))
|
||||
{
|
||||
p = strchr(buffer, ':') + 2;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
fclose(infile);
|
||||
|
||||
if( p == NULL ) return 0;
|
||||
|
||||
t = strtok(p," ");
|
||||
while( t = strtok(NULL," "))
|
||||
{
|
||||
if (strstr(t, search)) { return(1); }
|
||||
}
|
||||
|
||||
#endif
|
||||
return(0);
|
||||
}
|
||||
|
||||
|
||||
+39
-40
@@ -1,5 +1,5 @@
|
||||
/*****************************************************************************
|
||||
Copyright (c) 2011-2014, The OpenBLAS Project
|
||||
Copyright (c) 2011-2026, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
@@ -13,9 +13,9 @@ met:
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written
|
||||
permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
@@ -145,13 +145,47 @@ void get_subarchitecture(void){
|
||||
printf("SICORTEX");
|
||||
}else{
|
||||
printf("MIPS64_GENERIC");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void get_subdirname(void){
|
||||
printf("mips64");
|
||||
}
|
||||
|
||||
int get_feature(char *search) {
|
||||
|
||||
#ifdef __linux
|
||||
FILE *infile;
|
||||
char buffer[2048], *p, *t;
|
||||
p = (char *)NULL;
|
||||
|
||||
infile = fopen("/proc/cpuinfo", "r");
|
||||
|
||||
while (fgets(buffer, sizeof(buffer), infile)) {
|
||||
|
||||
if (!strncmp("Features", buffer, 8) ||
|
||||
!strncmp("ASEs implemented", buffer, 16)) {
|
||||
p = strchr(buffer, ':') + 2;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
fclose(infile);
|
||||
|
||||
if (p == NULL)
|
||||
return 0;
|
||||
|
||||
t = strtok(p, " ");
|
||||
while (t = strtok(NULL, " ")) {
|
||||
if (strstr(t, search)) {
|
||||
return (1);
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
return (0);
|
||||
}
|
||||
|
||||
void get_cpuconfig(void){
|
||||
if(detect()==CPU_LOONGSON3R3) {
|
||||
printf("#define LOONGSON3R3\n");
|
||||
@@ -228,38 +262,3 @@ void get_libname(void){
|
||||
printf("mips64_generic\n");
|
||||
}
|
||||
}
|
||||
|
||||
int get_feature(char *search)
|
||||
{
|
||||
|
||||
#ifdef __linux
|
||||
FILE *infile;
|
||||
char buffer[2048], *p,*t;
|
||||
p = (char *) NULL ;
|
||||
|
||||
infile = fopen("/proc/cpuinfo", "r");
|
||||
|
||||
while (fgets(buffer, sizeof(buffer), infile))
|
||||
{
|
||||
|
||||
if (!strncmp("Features", buffer, 8) || !strncmp("ASEs implemented", buffer, 16))
|
||||
{
|
||||
p = strchr(buffer, ':') + 2;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
fclose(infile);
|
||||
|
||||
if( p == NULL ) return 0;
|
||||
|
||||
t = strtok(p," ");
|
||||
while( t = strtok(NULL," "))
|
||||
{
|
||||
if (strstr(t, search)) { return(1); }
|
||||
}
|
||||
|
||||
#endif
|
||||
return(0);
|
||||
}
|
||||
|
||||
|
||||
+1694
-1755
File diff suppressed because it is too large
Load Diff
@@ -178,7 +178,10 @@ ARCH_CSKY
|
||||
#endif
|
||||
|
||||
#if defined(__EMSCRIPTEN__)
|
||||
ARCH_RISCV64
|
||||
ARCH_WASM
|
||||
OS_WINDOWS
|
||||
#endif
|
||||
|
||||
#if defined(TARGET_OS_IPHONE)
|
||||
OS_IOS
|
||||
#endif
|
||||
|
||||
+8
-15
@@ -23,17 +23,10 @@ typedef struct { real r, i; } complex;
|
||||
typedef struct { doublereal r, i; } doublecomplex;
|
||||
#ifdef _MSC_VER
|
||||
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
|
||||
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
|
||||
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
|
||||
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
|
||||
#else
|
||||
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
|
||||
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
|
||||
#endif
|
||||
#define pCf(z) (*_pCf(z))
|
||||
#define pCd(z) (*_pCd(z))
|
||||
typedef int logical;
|
||||
typedef short int shortlogical;
|
||||
typedef char logical1;
|
||||
@@ -440,12 +433,12 @@ static real c_b43 = (float)1.;
|
||||
extern /* Subroutine */ int ctest_(integer*, complex*, complex*, complex*, real*);
|
||||
static complex mwpcs[5], mwpct[5];
|
||||
extern /* Subroutine */ int itest1_(integer*, integer*), stest1_(real*,real*,real*,real*);
|
||||
extern /* Subroutine */ int cscaltest_(integer*, complex*, complex*, integer*);
|
||||
extern /* Subroutine */ void cscaltest_(integer*, complex*, complex*, integer*);
|
||||
static complex cx[8];
|
||||
extern real scnrm2test_(integer*, complex*, integer*);
|
||||
static integer np1;
|
||||
extern integer icamaxtest_(integer*, complex*, integer*);
|
||||
extern /* Subroutine */ int csscaltest_(integer*, real*, complex*, integer*);
|
||||
extern /* Subroutine */ void csscaltest_(integer*, real*, complex*, integer*);
|
||||
extern real scasumtest_(integer*, complex*, integer*);
|
||||
static integer len;
|
||||
|
||||
@@ -468,7 +461,7 @@ static real c_b43 = (float)1.;
|
||||
i__1 = len;
|
||||
for (i__ = 1; i__ <= i__1; ++i__) {
|
||||
i__2 = i__ - 1;
|
||||
i__3 = i__ + (np1 + combla_1.incx * 5 << 3) - 49;
|
||||
i__3 = i__ + ((np1 + combla_1.incx * 5) << 3) - 49;
|
||||
cx[i__2].r = cv[i__3].r, cx[i__2].i = cv[i__3].i;
|
||||
/* L20: */
|
||||
}
|
||||
@@ -483,13 +476,13 @@ static real c_b43 = (float)1.;
|
||||
} else if (combla_1.icase == 8) {
|
||||
/* .. CSCAL .. */
|
||||
cscaltest_(&combla_1.n, &ca, cx, &combla_1.incx);
|
||||
ctest_(&len, cx, &ctrue5[(np1 + combla_1.incx * 5 << 3) - 48],
|
||||
&ctrue5[(np1 + combla_1.incx * 5 << 3) - 48], sfac);
|
||||
ctest_(&len, cx, &ctrue5[((np1 + combla_1.incx * 5) << 3) - 48],
|
||||
&ctrue5[((np1 + combla_1.incx * 5) << 3) - 48], sfac);
|
||||
} else if (combla_1.icase == 9) {
|
||||
/* .. CSSCALTEST .. */
|
||||
csscaltest_(&combla_1.n, &sa, cx, &combla_1.incx);
|
||||
ctest_(&len, cx, &ctrue6[(np1 + combla_1.incx * 5 << 3) - 48],
|
||||
&ctrue6[(np1 + combla_1.incx * 5 << 3) - 48], sfac);
|
||||
ctest_(&len, cx, &ctrue6[((np1 + combla_1.incx * 5) << 3) - 48],
|
||||
&ctrue6[((np1 + combla_1.incx * 5) << 3) - 48], sfac);
|
||||
} else if (combla_1.icase == 10) {
|
||||
/* .. ICAMAXTEST .. */
|
||||
i__1 = icamaxtest_(&combla_1.n, cx, &combla_1.incx);
|
||||
@@ -737,7 +730,7 @@ static real c_b43 = (float)1.;
|
||||
static complex ctemp;
|
||||
extern /* Subroutine */ int ctest_(integer*, complex*, complex*, complex*, real*);
|
||||
static integer ksize;
|
||||
extern /* Subroutine */ int cdotctest_(integer*, complex*, integer*, complex*, integer*,complex*), ccopytest_(integer*, complex*, integer*, complex*, integer*), cdotutest_(integer*, complex*, integer*, complex*, integer*, complex*),
|
||||
extern /* Subroutine */ void cdotctest_(integer*, complex*, integer*, complex*, integer*,complex*), ccopytest_(integer*, complex*, integer*, complex*, integer*), cdotutest_(integer*, complex*, integer*, complex*, integer*, complex*),
|
||||
cswaptest_(integer*, complex*, integer*, complex*, integer*), caxpytest_(integer*, complex*, complex*, integer*, complex*, integer*);
|
||||
static integer ki, kn;
|
||||
static complex cx[7], cy[7];
|
||||
|
||||
+32
-46
@@ -23,17 +23,12 @@ typedef struct { real r, i; } complex;
|
||||
typedef struct { doublereal r, i; } doublecomplex;
|
||||
#ifdef _MSC_VER
|
||||
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
|
||||
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
|
||||
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
|
||||
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
|
||||
#else
|
||||
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
|
||||
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
|
||||
#endif
|
||||
#define pCf(z) (*_pCf(z))
|
||||
#define pCd(z) (*_pCd(z))
|
||||
typedef int logical;
|
||||
typedef short int shortlogical;
|
||||
typedef char logical1;
|
||||
@@ -319,7 +314,7 @@ static logical c_false = FALSE_;
|
||||
static char snamet[12];
|
||||
static real thresh;
|
||||
static logical rorder;
|
||||
extern /* Subroutine */ void cc2chke_(char*, ftnlen);
|
||||
extern /* Subroutine */ void cc2chke_(char*);
|
||||
static integer layout;
|
||||
static logical ltestt, tsterr;
|
||||
static complex alf[7];
|
||||
@@ -712,7 +707,7 @@ L100:
|
||||
ftnlen)12);
|
||||
/* Test error exits. */
|
||||
if (tsterr) {
|
||||
cc2chke_(snames[isnum - 1], (ftnlen)12);
|
||||
cc2chke_(snames[isnum - 1]);
|
||||
}
|
||||
/* Test computations. */
|
||||
infoc_1.infot = 0;
|
||||
@@ -892,8 +887,8 @@ L240:
|
||||
static integer ia, ib, ic;
|
||||
static logical banded;
|
||||
static integer nc, nd, im, in, kl, ml, nk, nl, ku, ix, iy, ms, lx, ly, ns;
|
||||
extern /* Subroutine */ int ccgbmv_(integer*, char*, integer*, integer*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void ccgemv_(integer*, char*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void ccgbmv_(integer*, char*, integer*, integer*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*);
|
||||
extern /* Subroutine */ void ccgemv_(integer*, char*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*);
|
||||
extern logical lceres_(char*, char*, integer*, integer*, complex*, complex*, integer*, ftnlen, ftnlen);
|
||||
static char ctrans[14];
|
||||
static real errmax;
|
||||
@@ -1142,8 +1137,7 @@ L240:
|
||||
}
|
||||
ccgemv_(iorder, trans, &m, &n, &alpha,
|
||||
&aa[1], &lda, &xx[1], &incx,
|
||||
&beta, &yy[1], &incy, (ftnlen)
|
||||
1);
|
||||
&beta, &yy[1], &incy);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1158,8 +1152,7 @@ L240:
|
||||
}
|
||||
ccgbmv_(iorder, trans, &m, &n, &kl, &
|
||||
ku, &alpha, &aa[1], &lda, &xx[
|
||||
1], &incx, &beta, &yy[1], &
|
||||
incy, (ftnlen)1);
|
||||
1], &incx, &beta, &yy[1], &incy);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -1347,10 +1340,10 @@ L140:
|
||||
static integer nc, ik, in;
|
||||
static logical packed;
|
||||
static integer nk, ks, ix, iy, ns, lx, ly;
|
||||
extern /* Subroutine */ void cchbmv_(integer*, char*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cchemv_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cchbmv_(integer*, char*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*);
|
||||
extern /* Subroutine */ void cchemv_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*);
|
||||
extern logical lceres_(char*, char*, integer*, integer*, complex*, complex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cchpmv_(integer*, char*, integer*, complex*, complex*, complex*, integer*, complex*, complex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cchpmv_(integer*, char*, integer*, complex*, complex*, complex*, integer*, complex*, complex*, integer*);
|
||||
static real errmax;
|
||||
static complex transl;
|
||||
static integer laa, lda;
|
||||
@@ -1566,7 +1559,7 @@ L140:
|
||||
}
|
||||
cchemv_(iorder, uplo, &n, &alpha, &aa[1],
|
||||
&lda, &xx[1], &incx, &beta, &yy[1]
|
||||
, &incy, (ftnlen)1);
|
||||
, &incy);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1581,7 +1574,7 @@ L140:
|
||||
}
|
||||
cchbmv_(iorder, uplo, &n, &k, &alpha, &aa[
|
||||
1], &lda, &xx[1], &incx, &beta, &
|
||||
yy[1], &incy, (ftnlen)1);
|
||||
yy[1], &incy);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1596,7 +1589,7 @@ L140:
|
||||
}
|
||||
cchpmv_(iorder, uplo, &n, &alpha, &aa[1],
|
||||
&xx[1], &incx, &beta, &yy[1], &
|
||||
incy, (ftnlen)1);
|
||||
incy);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -1792,15 +1785,15 @@ L130:
|
||||
static logical packed;
|
||||
static integer nk, ks, ix, ns, lx;
|
||||
extern logical lceres_(char*, char*, integer*, integer*, complex*, complex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cctbmv_(integer*, char*, char*, char*, integer*, integer*, complex*, integer*, complex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cctbsv_(integer*, char*, char*, char*, integer*, integer*, complex*, integer*, complex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cctbmv_(integer*, char*, char*, char*, integer*, integer*, complex*, integer*, complex*, integer*);
|
||||
extern /* Subroutine */ void cctbsv_(integer*, char*, char*, char*, integer*, integer*, complex*, integer*, complex*, integer*);
|
||||
static char ctrans[14];
|
||||
extern /* Subroutine */ void cctpmv_(integer*, char*, char*, char*, integer*, complex*, complex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cctpmv_(integer*, char*, char*, char*, integer*, complex*, complex*, integer*);
|
||||
static real errmax;
|
||||
extern /* Subroutine */ void cctrmv_(integer*, char*, char*, char*, integer*, complex*, integer*, complex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cctpsv_(integer*, char*, char*, char*, integer*, complex*, complex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cctrmv_(integer*, char*, char*, char*, integer*, complex*, integer*, complex*, integer*);
|
||||
extern /* Subroutine */ void cctpsv_(integer*, char*, char*, char*, integer*, complex*, complex*, integer*);
|
||||
static complex transl;
|
||||
extern /* Subroutine */ void cctrsv_(integer*, char*, char*, char*, integer*, complex*, integer*, complex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cctrsv_(integer*, char*, char*, char*, integer*, complex*, integer*, complex*, integer*);
|
||||
static char transs[1];
|
||||
static integer laa, icd, lda;
|
||||
extern logical lce_(complex*, complex*, integer*);
|
||||
@@ -2010,8 +2003,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cctrmv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
aa[1], &lda, &xx[1], &incx);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2025,8 +2017,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cctbmv_(iorder, uplo, trans, diag, &n, &k,
|
||||
&aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
&aa[1], &lda, &xx[1], &incx);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2040,8 +2031,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cctpmv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &xx[1], &incx, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
aa[1], &xx[1], &incx);
|
||||
}
|
||||
} else if (s_cmp(sname + 9, "sv", (ftnlen)2, (
|
||||
ftnlen)2) == 0) {
|
||||
@@ -2058,8 +2048,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cctrsv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
aa[1], &lda, &xx[1], &incx);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2073,8 +2062,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cctbsv_(iorder, uplo, trans, diag, &n, &k,
|
||||
&aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
&aa[1], &lda, &xx[1], &incx);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2088,8 +2076,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cctpsv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &xx[1], &incx, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
aa[1], &xx[1], &incx);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2634,10 +2621,10 @@ L150:
|
||||
static char uplo[1];
|
||||
static integer i__, j, n;
|
||||
extern /* Subroutine */ int cmake_(char*, char*, char*, integer*, integer*, complex*, integer*, complex*, integer*, integer*, integer*, logical*, complex*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void ccher_(integer*, char*, integer*, real*, complex*, integer*, complex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void ccher_(integer*, char*, integer*, real*, complex*, integer*, complex*, integer*);
|
||||
static complex alpha, w[1];
|
||||
static logical isame[13];
|
||||
extern /* Subroutine */ void cchpr_(integer*, char*, integer*, real*, complex*, integer*, complex*, ftnlen);
|
||||
extern /* Subroutine */ void cchpr_(integer*, char*, integer*, real*, complex*, integer*, complex*);
|
||||
extern /* Subroutine */ int cmvch_(char*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*, complex*, real*, complex*, real*, real*, logical*, integer*, logical*, ftnlen);
|
||||
static integer nargs;
|
||||
static logical reset;
|
||||
@@ -2812,7 +2799,7 @@ L150:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
ccher_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[
|
||||
1], &lda, (ftnlen)1);
|
||||
1], &lda);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2825,8 +2812,7 @@ L150:
|
||||
al__1.aunit = *ntra;
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cchpr_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[
|
||||
1], (ftnlen)1);
|
||||
cchpr_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[1]);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -3005,8 +2991,8 @@ L130:
|
||||
static integer incxs, incys;
|
||||
static logical upper;
|
||||
static char uplos[1];
|
||||
extern /* Subroutine */ void ccher2_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cchpr2_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, ftnlen);
|
||||
extern /* Subroutine */ void ccher2_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, integer*);
|
||||
extern /* Subroutine */ void cchpr2_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*);
|
||||
static integer ia, ja, ic, nc, jj, lj, in;
|
||||
static logical packed;
|
||||
static integer ix, iy, ns, lx, ly;
|
||||
@@ -3202,7 +3188,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
ccher2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
|
||||
yy[1], &incy, &aa[1], &lda, (ftnlen)1);
|
||||
yy[1], &incy, &aa[1], &lda);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -3216,7 +3202,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cchpr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
|
||||
yy[1], &incy, &aa[1], (ftnlen)1);
|
||||
yy[1], &incy, &aa[1]);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
+35
-40
@@ -23,17 +23,12 @@ typedef struct { real r, i; } complex;
|
||||
typedef struct { doublereal r, i; } doublecomplex;
|
||||
#ifdef _MSC_VER
|
||||
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
|
||||
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
|
||||
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
|
||||
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
|
||||
#else
|
||||
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
|
||||
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
|
||||
#endif
|
||||
#define pCf(z) (*_pCf(z))
|
||||
#define pCd(z) (*_pCd(z))
|
||||
typedef int logical;
|
||||
typedef short int shortlogical;
|
||||
typedef char logical1;
|
||||
@@ -284,10 +279,10 @@ int /* Main program */ main(void)
|
||||
real r__1;
|
||||
|
||||
/* Local variables */
|
||||
integer nalf, idim[9];
|
||||
logical same;
|
||||
integer nbet, ntra;
|
||||
logical rewi;
|
||||
static integer nalf, idim[9];
|
||||
static logical same;
|
||||
static integer nbet, ntra;
|
||||
static logical rewi;
|
||||
extern /* Subroutine */ int cchk1_(char *, real *, real *, integer *,
|
||||
integer *, logical *, logical *, logical *, integer *, integer *,
|
||||
integer *, complex *, integer *, complex *, integer *, complex *,
|
||||
@@ -311,35 +306,35 @@ int /* Main program */ main(void)
|
||||
integer *, complex *, integer *, complex *, integer *, complex *,
|
||||
complex *, complex *, complex *, complex *, complex *, complex *,
|
||||
complex *, complex *, real *, complex *, integer *);
|
||||
complex c__[4225] /* was [65][65] */;
|
||||
real g[65];
|
||||
integer i__, j, n;
|
||||
logical fatal;
|
||||
complex w[130];
|
||||
static complex c__[4225] /* was [65][65] */;
|
||||
static real g[65];
|
||||
static integer i__, j, n;
|
||||
static logical fatal;
|
||||
static complex w[130];
|
||||
extern /* Subroutine */ int cmmch_(char *, char *, integer *, integer *,
|
||||
integer *, complex *, complex *, integer *, complex *, integer *,
|
||||
complex *, complex *, integer *, complex *, real *, complex *,
|
||||
integer *, real *, real *, logical *, integer *, logical *);
|
||||
extern real sdiff_(real *, real *);
|
||||
logical trace;
|
||||
integer nidim;
|
||||
char snaps[32];
|
||||
integer isnum;
|
||||
logical ltest[9];
|
||||
complex aa[4225], ab[8450] /* was [65][130] */, bb[4225], cc[4225], as[
|
||||
static logical trace;
|
||||
static integer nidim;
|
||||
static char snaps[32];
|
||||
static integer isnum;
|
||||
static logical ltest[9];
|
||||
static complex aa[4225], ab[8450] /* was [65][130] */, bb[4225], cc[4225], as[
|
||||
4225], bs[4225], cs[4225], ct[65];
|
||||
logical sfatal, corder;
|
||||
char snamet[12], transa[1], transb[1];
|
||||
real thresh;
|
||||
logical rorder;
|
||||
extern /* Subroutine */ int cc3chke_(char *);
|
||||
integer layout;
|
||||
logical ltestt, tsterr;
|
||||
complex alf[7];
|
||||
static logical sfatal, corder;
|
||||
static char snamet[12], transa[1], transb[1];
|
||||
static real thresh;
|
||||
static logical rorder;
|
||||
extern /* Subroutine */ void cc3chke_(char *);
|
||||
static integer layout;
|
||||
static logical ltestt, tsterr;
|
||||
static complex alf[7];
|
||||
extern logical lce_(complex *, complex *, integer *);
|
||||
complex bet[7];
|
||||
real eps, err;
|
||||
char tmpchar;
|
||||
static complex bet[7];
|
||||
static real eps, err;
|
||||
static char tmpchar;
|
||||
|
||||
/* Test program for the COMPLEX Level 3 Blas. */
|
||||
|
||||
@@ -856,7 +851,7 @@ L230:
|
||||
*, char *, char *, integer *, integer *, integer *, complex *,
|
||||
integer *, integer *, complex *, integer *);
|
||||
integer ia, ib, ma, mb, na, nb, nc, ik, im, in;
|
||||
extern /* Subroutine */ int ccgemm_(integer *, char *, char *, integer *,
|
||||
extern /* Subroutine */ void ccgemm_(integer *, char *, char *, integer *,
|
||||
integer *, integer *, complex *, complex *, integer *, complex *,
|
||||
integer *, complex *, complex *, integer *);
|
||||
integer ks, ms, ns;
|
||||
@@ -1268,13 +1263,13 @@ L130:
|
||||
*, char *, char *, integer *, integer *, complex *, integer *,
|
||||
integer *, complex *, integer *);
|
||||
integer ia, ib, na, nc, im, in;
|
||||
extern /* Subroutine */ int cchemm_(integer *, char *, char *, integer *,
|
||||
extern /* Subroutine */ void cchemm_(integer *, char *, char *, integer *,
|
||||
integer *, complex *, complex *, integer *, complex *, integer *,
|
||||
complex *, complex *, integer *);
|
||||
integer ms, ns;
|
||||
extern logical lceres_(char *, char *, integer *, integer *, complex *,
|
||||
complex *, integer *);
|
||||
extern /* Subroutine */ int ccsymm_(integer *, char *, char *, integer *,
|
||||
extern /* Subroutine */ void ccsymm_(integer *, char *, char *, integer *,
|
||||
integer *, complex *, complex *, integer *, complex *, integer *,
|
||||
complex *, complex *, integer *);
|
||||
real errmax;
|
||||
@@ -1668,11 +1663,11 @@ L120:
|
||||
integer ia, na, nc, im, in, ms, ns;
|
||||
extern logical lceres_(char *, char *, integer *, integer *, complex *,
|
||||
complex *, integer *);
|
||||
extern /* Subroutine */ int cctrmm_(integer *, char *, char *, char *,
|
||||
extern /* Subroutine */ void cctrmm_(integer *, char *, char *, char *,
|
||||
char *, integer *, integer *, complex *, complex *, integer *,
|
||||
complex *, integer *);
|
||||
char tranas[1], transa[1];
|
||||
extern /* Subroutine */ int cctrsm_(integer *, char *, char *, char *,
|
||||
extern /* Subroutine */ void cctrsm_(integer *, char *, char *, char *,
|
||||
char *, integer *, integer *, complex *, complex *, integer *,
|
||||
complex *, integer *);
|
||||
real errmax;
|
||||
@@ -2143,7 +2138,7 @@ L160:
|
||||
integer *, char *, integer *, char *, char *, integer *, integer *
|
||||
, real *, integer *, real *, integer *);
|
||||
integer ia, ib, jc, ma, na, nc, ik, in, jj, lj, ks;
|
||||
extern /* Subroutine */ int ccherk_(integer *, char *, char *, integer *,
|
||||
extern /* Subroutine */ void ccherk_(integer *, char *, char *, integer *,
|
||||
integer *, real *, complex *, integer *, real *, complex *,
|
||||
integer *);
|
||||
integer ns;
|
||||
@@ -2151,7 +2146,7 @@ L160:
|
||||
extern logical lceres_(char *, char *, integer *, integer *, complex *,
|
||||
complex *, integer *);
|
||||
real errmax;
|
||||
extern /* Subroutine */ int ccsyrk_(integer *, char *, char *, integer *,
|
||||
extern /* Subroutine */ void ccsyrk_(integer *, char *, char *, integer *,
|
||||
integer *, complex *, complex *, integer *, complex *, complex *,
|
||||
integer *);
|
||||
char transs[1], transt[1];
|
||||
@@ -2643,12 +2638,12 @@ L130:
|
||||
complex *, integer *);
|
||||
real errmax;
|
||||
char transs[1], transt[1];
|
||||
extern /* Subroutine */ int ccher2k_(integer *, char *, char *, integer *,
|
||||
extern /* Subroutine */ void ccher2k_(integer *, char *, char *, integer *,
|
||||
integer *, complex *, complex *, integer *, complex *, integer *,
|
||||
real *, complex *, integer *);
|
||||
integer laa, lbb, lda, lcc, ldb, ldc;
|
||||
extern logical lce_(complex *, complex *, integer *);
|
||||
extern /* Subroutine */ int ccsyr2k_(integer *, char *, char *, integer *,
|
||||
extern /* Subroutine */ void ccsyr2k_(integer *, char *, char *, integer *,
|
||||
integer *, complex *, complex *, integer *, complex *, integer *,
|
||||
complex *, complex *, integer *);
|
||||
complex als;
|
||||
|
||||
+1
-1
@@ -54,7 +54,7 @@ void F77_drot( const int *N, double *X, const int *incX, double *Y,
|
||||
}
|
||||
|
||||
void F77_drotm(const int *N, double *X, const int *incX, double *Y,
|
||||
const int *incY, const double *dparam)
|
||||
const int *incY, double *dparam)
|
||||
{
|
||||
cblas_drotm(*N, X, *incX, Y, *incY, dparam);
|
||||
return;
|
||||
|
||||
+13
-8
@@ -332,7 +332,8 @@ static doublereal c_b34 = 1.;
|
||||
|
||||
/* Local variables */
|
||||
static integer k;
|
||||
extern /* Subroutine */ int drotgtest_(doublereal*,doublereal*,doublereal*,doublereal*), stest1_(doublereal*,doublereal*,doublereal*,doublereal*);
|
||||
extern /* Subroutine */ void drotgtest_(doublereal*,doublereal*,doublereal*,doublereal*);
|
||||
extern int stest1_(doublereal*,doublereal*,doublereal*,doublereal*);
|
||||
static doublereal sa, sb, sc, ss;
|
||||
|
||||
/* .. Parameters .. */
|
||||
@@ -404,7 +405,8 @@ L40:
|
||||
static integer i__;
|
||||
extern doublereal dnrm2test_(integer*, doublereal*, integer*);
|
||||
static doublereal stemp[1], strue[8];
|
||||
extern /* Subroutine */ int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*), dscaltest_(integer*,doublereal*,doublereal*,integer*);
|
||||
extern /* Subroutine */ int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*);
|
||||
extern void dscaltest_(integer*,doublereal*,doublereal*,integer*);
|
||||
extern doublereal dasumtest_(integer*,doublereal*,integer*);
|
||||
extern /* Subroutine */ int itest1_(integer*,integer*), stest1_(doublereal*,doublereal*,doublereal*,doublereal*);
|
||||
static doublereal sx[8];
|
||||
@@ -430,7 +432,7 @@ L40:
|
||||
/* .. Set vector arguments .. */
|
||||
i__1 = len;
|
||||
for (i__ = 1; i__ <= i__1; ++i__) {
|
||||
sx[i__ - 1] = dv[i__ + (np1 + combla_1.incx * 5 << 3) - 49];
|
||||
sx[i__ - 1] = dv[i__ + ((np1 + combla_1.incx * 5) << 3) - 49];
|
||||
/* L20: */
|
||||
}
|
||||
|
||||
@@ -450,7 +452,7 @@ L40:
|
||||
, sx, &combla_1.incx);
|
||||
i__1 = len;
|
||||
for (i__ = 1; i__ <= i__1; ++i__) {
|
||||
strue[i__ - 1] = dtrue5[i__ + (np1 + combla_1.incx * 5 <<
|
||||
strue[i__ - 1] = dtrue5[i__ + ((np1 + combla_1.incx * 5) <<
|
||||
3) - 49];
|
||||
/* L40: */
|
||||
}
|
||||
@@ -517,8 +519,10 @@ L40:
|
||||
static integer lenx, leny;
|
||||
extern doublereal ddottest_(integer*,doublereal*,integer*,doublereal*,integer*);
|
||||
static integer i__, j, ksize;
|
||||
extern /* Subroutine */ int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*), dcopytest_(integer*,doublereal*,integer*,doublereal*,integer*), dswaptest_(integer*,doublereal*,integer*,doublereal*,integer*),
|
||||
daxpytest_(integer*,doublereal*,doublereal*,integer*,doublereal*,integer*), stest1_(doublereal*,doublereal*,doublereal*,doublereal*);
|
||||
extern /* Subroutine */ int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*);
|
||||
extern void dcopytest_(integer*,doublereal*,integer*,doublereal*,integer*), dswaptest_(integer*,doublereal*,integer*,doublereal*,integer*),
|
||||
daxpytest_(integer*,doublereal*,doublereal*,integer*,doublereal*,integer*);
|
||||
extern int stest1_(doublereal*,doublereal*,doublereal*,doublereal*);
|
||||
static integer ki, kn, mx, my;
|
||||
static doublereal sx[7], sy[7], stx[7], sty[7];
|
||||
|
||||
@@ -618,9 +622,10 @@ L40:
|
||||
;
|
||||
|
||||
/* Local variables */
|
||||
extern /* Subroutine */ int drottest_(integer*,doublereal*,integer*,doublereal*,integer*,doublereal*,doublereal*);
|
||||
extern /* Subroutine */ void drottest_(integer*,doublereal*,integer*,doublereal*,integer*,doublereal*,doublereal*);
|
||||
static integer i__, k, ksize;
|
||||
extern /* Subroutine */int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*), drotmtest_(integer*,doublereal*,integer*,doublereal*,integer*,doublereal*);
|
||||
extern /* Subroutine */int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*);
|
||||
extern void drotmtest_(integer*,doublereal*,integer*,doublereal*,integer*,doublereal*);
|
||||
static integer ki, kn;
|
||||
static doublereal dparam[5], sx[10], sy[10], stx[10], sty[10];
|
||||
|
||||
|
||||
+32
-53
@@ -21,19 +21,6 @@ typedef float real;
|
||||
typedef double doublereal;
|
||||
typedef struct { real r, i; } complex;
|
||||
typedef struct { doublereal r, i; } doublecomplex;
|
||||
#ifdef _MSC_VER
|
||||
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
|
||||
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
|
||||
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
|
||||
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
|
||||
#else
|
||||
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
|
||||
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
|
||||
#endif
|
||||
#define pCf(z) (*_pCf(z))
|
||||
#define pCd(z) (*_pCd(z))
|
||||
typedef int logical;
|
||||
typedef short int shortlogical;
|
||||
typedef char logical1;
|
||||
@@ -318,7 +305,7 @@ static logical c_false = FALSE_;
|
||||
static char snamet[12];
|
||||
static doublereal thresh;
|
||||
static logical rorder;
|
||||
extern /* Subroutine */ void cd2chke_(char*, ftnlen);
|
||||
extern /* Subroutine */ void cd2chke_(char*);
|
||||
static integer layout;
|
||||
static logical ltestt, tsterr;
|
||||
static doublereal alf[7];
|
||||
@@ -706,7 +693,7 @@ L100:
|
||||
ftnlen)12);
|
||||
/* Test error exits. */
|
||||
if (tsterr) {
|
||||
cd2chke_(snames[isnum - 1], (ftnlen)12);
|
||||
cd2chke_(snames[isnum - 1]);
|
||||
}
|
||||
/* Test computations. */
|
||||
infoc_1.infot = 0;
|
||||
@@ -885,8 +872,8 @@ L240:
|
||||
static integer ia, ib, ic;
|
||||
static logical banded;
|
||||
static integer nc, nd, im, in, kl, ml, nk, nl, ku, ix, iy, ms, lx, ly, ns;
|
||||
extern /* Subroutine */ void cdgbmv_(integer*, char*, integer*, integer*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cdgemv_(integer*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cdgbmv_(integer*, char*, integer*, integer*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
|
||||
extern /* Subroutine */ void cdgemv_(integer*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
|
||||
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
static char ctrans[14];
|
||||
static doublereal errmax, transl;
|
||||
@@ -1118,8 +1105,7 @@ L240:
|
||||
}
|
||||
cdgemv_(iorder, trans, &m, &n, &alpha,
|
||||
&aa[1], &lda, &xx[1], &incx,
|
||||
&beta, &yy[1], &incy, (ftnlen)
|
||||
1);
|
||||
&beta, &yy[1], &incy);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1135,7 +1121,7 @@ L240:
|
||||
cdgbmv_(iorder, trans, &m, &n, &kl, &
|
||||
ku, &alpha, &aa[1], &lda, &xx[
|
||||
1], &incx, &beta, &yy[1], &
|
||||
incy, (ftnlen)1);
|
||||
incy);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -1329,10 +1315,10 @@ L140:
|
||||
static logical packed;
|
||||
static integer nk, ks, ix, iy, ns, lx, ly;
|
||||
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdsbmv_(integer*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cdspmv_(integer*, char*, integer*, doublereal*, doublereal*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cdsbmv_(integer*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
|
||||
extern /* Subroutine */ void cdspmv_(integer*, char*, integer*, doublereal*, doublereal*, doublereal*, integer*, doublereal*, doublereal*, integer*);
|
||||
static doublereal errmax, transl;
|
||||
extern /* Subroutine */ void cdsymv_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cdsymv_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
|
||||
static integer laa, lda;
|
||||
extern logical lde_(doublereal*, doublereal*, integer*);
|
||||
static doublereal als, bls, err;
|
||||
@@ -1536,7 +1522,7 @@ L140:
|
||||
}
|
||||
cdsymv_(iorder, uplo, &n, &alpha, &aa[1],
|
||||
&lda, &xx[1], &incx, &beta, &yy[1]
|
||||
, &incy, (ftnlen)1);
|
||||
, &incy);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1551,7 +1537,7 @@ L140:
|
||||
}
|
||||
cdsbmv_(iorder, uplo, &n, &k, &alpha, &aa[
|
||||
1], &lda, &xx[1], &incx, &beta, &
|
||||
yy[1], &incy, (ftnlen)1);
|
||||
yy[1], &incy);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1566,7 +1552,7 @@ L140:
|
||||
}
|
||||
cdspmv_(iorder, uplo, &n, &alpha, &aa[1],
|
||||
&xx[1], &incx, &beta, &yy[1], &
|
||||
incy, (ftnlen)1);
|
||||
incy);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -1770,15 +1756,15 @@ L130:
|
||||
static logical packed;
|
||||
static integer nk, ks, ix, ns, lx;
|
||||
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdtbmv_(integer*, char*, char*, char*, integer*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdtbsv_(integer*, char*, char*, char*, integer*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdtbmv_(integer*, char*, char*, char*, integer*, integer*, doublereal*, integer*, doublereal*, integer*);
|
||||
extern /* Subroutine */ void cdtbsv_(integer*, char*, char*, char*, integer*, integer*, doublereal*, integer*, doublereal*, integer*);
|
||||
static char ctrans[14];
|
||||
static doublereal errmax;
|
||||
extern /* Subroutine */ void cdtpmv_(integer*, char*, char*, char*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdtrmv_(integer*, char*, char*, char*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdtpmv_(integer*, char*, char*, char*, integer*, doublereal*, doublereal*, integer*);
|
||||
extern /* Subroutine */ void cdtrmv_(integer*, char*, char*, char*, integer*, doublereal*, integer*, doublereal*, integer*);
|
||||
static doublereal transl;
|
||||
extern /* Subroutine */ void cdtpsv_(integer*, char*, char*, char*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdtrsv_(integer*, char*, char*, char*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdtpsv_(integer*, char*, char*, char*, integer*, doublereal*, doublereal*, integer*);
|
||||
extern /* Subroutine */ void cdtrsv_(integer*, char*, char*, char*, integer*, doublereal*, integer*, doublereal*, integer*);
|
||||
static char transs[1];
|
||||
static integer laa, icd, lda;
|
||||
extern logical lde_(doublereal*, doublereal*, integer*);
|
||||
@@ -1978,8 +1964,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdtrmv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
aa[1], &lda, &xx[1], &incx);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1993,8 +1978,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdtbmv_(iorder, uplo, trans, diag, &n, &k,
|
||||
&aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
&aa[1], &lda, &xx[1], &incx);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2008,8 +1992,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdtpmv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &xx[1], &incx, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
aa[1], &xx[1], &incx);
|
||||
}
|
||||
} else if (s_cmp(sname + 9, "sv", (ftnlen)2, (
|
||||
ftnlen)2) == 0) {
|
||||
@@ -2026,8 +2009,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdtrsv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
aa[1], &lda, &xx[1], &incx);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2041,8 +2023,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdtbsv_(iorder, uplo, trans, diag, &n, &k,
|
||||
&aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
&aa[1], &lda, &xx[1], &incx);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2056,8 +2037,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdtpsv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &xx[1], &incx, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
aa[1], &xx[1], &incx);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2587,11 +2567,11 @@ L150:
|
||||
static logical isame[13];
|
||||
extern /* Subroutine */ int dmvch_(char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, doublereal*, doublereal*, doublereal*, doublereal*, doublereal*, logical*, integer*, logical*, ftnlen);
|
||||
static integer nargs;
|
||||
extern /* Subroutine */ void cdspr_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, ftnlen);
|
||||
extern /* Subroutine */ void cdspr_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*);
|
||||
static logical reset;
|
||||
static char cuplo[14];
|
||||
static integer incxs;
|
||||
extern /* Subroutine */ void cdsyr_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cdsyr_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*);
|
||||
static logical upper;
|
||||
static char uplos[1];
|
||||
static integer ia, ja, ic, nc, jj, lj, in;
|
||||
@@ -2751,7 +2731,7 @@ L150:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdsyr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]
|
||||
, &lda, (ftnlen)1);
|
||||
, &lda);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2764,8 +2744,7 @@ L150:
|
||||
al__1.aunit = *ntra;
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdspr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]
|
||||
, (ftnlen)1);
|
||||
cdspr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -2948,8 +2927,8 @@ L130:
|
||||
static integer incxs, incys;
|
||||
static logical upper;
|
||||
static char uplos[1];
|
||||
extern /* Subroutine */ void cdspr2_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, ftnlen);
|
||||
extern /* Subroutine */ void cdsyr2_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cdspr2_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*);
|
||||
extern /* Subroutine */ void cdsyr2_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, integer*);
|
||||
static integer ia, ja, ic, nc, jj, lj, in;
|
||||
static logical packed;
|
||||
static integer ix, iy, ns, lx, ly;
|
||||
@@ -3132,7 +3111,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdsyr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
|
||||
yy[1], &incy, &aa[1], &lda, (ftnlen)1);
|
||||
yy[1], &incy, &aa[1], &lda);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -3146,7 +3125,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdspr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
|
||||
yy[1], &incy, &aa[1], (ftnlen)1);
|
||||
yy[1], &incy, &aa[1]);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
+14
-32
@@ -21,19 +21,6 @@ typedef float real;
|
||||
typedef double doublereal;
|
||||
typedef struct { real r, i; } complex;
|
||||
typedef struct { doublereal r, i; } doublecomplex;
|
||||
#ifdef _MSC_VER
|
||||
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
|
||||
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
|
||||
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
|
||||
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
|
||||
#else
|
||||
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
|
||||
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
|
||||
#endif
|
||||
#define pCf(z) (*_pCf(z))
|
||||
#define pCd(z) (*_pCd(z))
|
||||
typedef int logical;
|
||||
typedef short int shortlogical;
|
||||
typedef char logical1;
|
||||
@@ -309,7 +296,7 @@ static logical c_false = FALSE_;
|
||||
static char snamet[12], transa[1], transb[1];
|
||||
static doublereal thresh;
|
||||
static logical rorder;
|
||||
extern /* Subroutine */ void cd3chke_(char*, ftnlen);
|
||||
extern /* Subroutine */ void cd3chke_(char*);
|
||||
static integer layout;
|
||||
static logical ltestt, tsterr;
|
||||
static doublereal alf[7];
|
||||
@@ -658,7 +645,7 @@ L80:
|
||||
ftnlen)12);
|
||||
/* Test error exits. */
|
||||
if (tsterr) {
|
||||
cd3chke_(snames[isnum - 1], (ftnlen)12);
|
||||
cd3chke_(snames[isnum - 1]);
|
||||
}
|
||||
/* Test computations. */
|
||||
infoc_1.infot = 0;
|
||||
@@ -807,7 +794,7 @@ L230:
|
||||
static logical reset;
|
||||
extern /* Subroutine */ void dprcn1_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, integer*, doublereal*, integer*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
static integer ia, ib, ma, mb, na, nb, nc, ik, im, in;
|
||||
extern /* Subroutine */ void cdgemm_(integer*, char*, char*, integer*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdgemm_(integer*, char*, char*, integer*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
|
||||
static integer ks, ms, ns;
|
||||
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
static char tranas[1], tranbs[1], transa[1], transb[1];
|
||||
@@ -1012,8 +999,7 @@ L230:
|
||||
}
|
||||
cdgemm_(iorder, transa, transb, &m, &n, &k, &
|
||||
alpha, &aa[1], &lda, &bb[1], &ldb, &
|
||||
beta, &cc[1], &ldc, (ftnlen)1, (
|
||||
ftnlen)1);
|
||||
beta, &cc[1], &ldc);
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
@@ -1204,7 +1190,7 @@ L130:
|
||||
extern /* Subroutine */ void dprcn2_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, doublereal*, integer*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
static integer ia, ib, na, nc, im, in, ms, ns;
|
||||
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdsymm_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdsymm_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
|
||||
static doublereal errmax;
|
||||
static integer laa, lbb, lda, lcc, ldb, ldc;
|
||||
extern logical lde_(doublereal*, doublereal*, integer*);
|
||||
@@ -1386,8 +1372,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdsymm_(iorder, side, uplo, &m, &n, &alpha, &aa[1]
|
||||
, &lda, &bb[1], &ldb, &beta, &cc[1], &ldc,
|
||||
(ftnlen)1, (ftnlen)1);
|
||||
, &lda, &bb[1], &ldb, &beta, &cc[1], &ldc);
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
@@ -1580,9 +1565,9 @@ L120:
|
||||
extern /* Subroutine */ void dprcn3_(integer*, integer*, char*, integer*, char*, char*, char*, char*, integer*, integer*, doublereal*, integer*, integer*, ftnlen, ftnlen, ftnlen, ftnlen, ftnlen);
|
||||
static integer ia, na, nc, im, in, ms, ns;
|
||||
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdtrmm_(integer*, char*, char*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdtrmm_(integer*, char*, char*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*);
|
||||
static char tranas[1], transa[1];
|
||||
extern /* Subroutine */ void cdtrsm_(integer*, char*, char*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdtrsm_(integer*, char*, char*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*);
|
||||
static doublereal errmax;
|
||||
static integer laa, icd, lbb, lda, ldb;
|
||||
extern logical lde_(doublereal*, doublereal*, integer*);
|
||||
@@ -1762,8 +1747,7 @@ L120:
|
||||
}
|
||||
cdtrmm_(iorder, side, uplo, transa, diag,
|
||||
&m, &n, &alpha, &aa[1], &lda, &bb[
|
||||
1], &ldb, (ftnlen)1, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
1], &ldb);
|
||||
} else if (s_cmp(sname + 9, "sm", (ftnlen)2, (
|
||||
ftnlen)2) == 0) {
|
||||
if (*trace) {
|
||||
@@ -1780,8 +1764,7 @@ L120:
|
||||
}
|
||||
cdtrsm_(iorder, side, uplo, transa, diag,
|
||||
&m, &n, &alpha, &aa[1], &lda, &bb[
|
||||
1], &ldb, (ftnlen)1, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
1], &ldb);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -2038,7 +2021,7 @@ L160:
|
||||
static integer ia, ib, jc, ma, na, nc, ik, in, jj, lj, ks, ns;
|
||||
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
static doublereal errmax;
|
||||
extern /* Subroutine */ void cdsyrk_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdsyrk_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, doublereal*, integer*);
|
||||
static char transs[1];
|
||||
static integer laa, lda, lcc, ldc;
|
||||
extern logical lde_(doublereal*, doublereal*, integer*);
|
||||
@@ -2199,8 +2182,7 @@ L160:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cdsyrk_(iorder, uplo, trans, &n, &k, &alpha, &aa[
|
||||
1], &lda, &beta, &cc[1], &ldc, (ftnlen)1,
|
||||
(ftnlen)1);
|
||||
1], &lda, &beta, &cc[1], &ldc);
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
@@ -2420,7 +2402,7 @@ L130:
|
||||
static char transs[1];
|
||||
static integer laa, lbb, lda, lcc, ldb, ldc;
|
||||
extern logical lde_(doublereal*, doublereal*, integer*);
|
||||
extern /* Subroutine */ void cdsyr2k_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cdsyr2k_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
|
||||
static doublereal als;
|
||||
static integer ict, icu;
|
||||
static doublereal err;
|
||||
@@ -2604,7 +2586,7 @@ L130:
|
||||
}
|
||||
cdsyr2k_(iorder, uplo, trans, &n, &k, &alpha, &aa[
|
||||
1], &lda, &bb[1], &ldb, &beta, &cc[1], &
|
||||
ldc, (ftnlen)1, (ftnlen)1);
|
||||
ldc);
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
|
||||
+10
-6
@@ -342,7 +342,8 @@ static real c_b34 = (float)1.;
|
||||
|
||||
/* Local variables */
|
||||
static integer k;
|
||||
extern /* Subroutine */ int srotgtest_(real*,real*,real*,real*), stest1_(real*,real*,real*,real*);
|
||||
extern /* Subroutine */ void srotgtest_(real*,real*,real*,real*);
|
||||
extern int stest1_(real*,real*,real*,real*);
|
||||
static real sa, sb, sc, ss;
|
||||
|
||||
/* .. Parameters .. */
|
||||
@@ -435,7 +436,8 @@ L40:
|
||||
static integer i__;
|
||||
extern real snrm2test_(integer*,real*,integer*);
|
||||
static real stemp[1], strue[8];
|
||||
extern /* Subroutine */ int stest_(integer*, real*,real*,real*,real*), sscaltest_(integer*,real*,real*,integer*);
|
||||
extern /* Subroutine */ int stest_(integer*, real*,real*,real*,real*);
|
||||
extern void sscaltest_(integer*,real*,real*,integer*);
|
||||
extern real sasumtest_(integer*,real*,integer*);
|
||||
extern /* Subroutine */ int itest1_(integer*,integer*), stest1_(real*,real*,real*,real*);
|
||||
static real sx[8];
|
||||
@@ -462,7 +464,7 @@ L40:
|
||||
/* .. Set vector arguments .. */
|
||||
i__1 = len;
|
||||
for (i__ = 1; i__ <= i__1; ++i__) {
|
||||
sx[i__ - 1] = dv[i__ + (np1 + combla_1.incx * 5 << 3) - 49];
|
||||
sx[i__ - 1] = dv[i__ + ((np1 + combla_1.incx * 5) << 3) - 49];
|
||||
/* L20: */
|
||||
}
|
||||
|
||||
@@ -482,7 +484,7 @@ L40:
|
||||
, sx, &combla_1.incx);
|
||||
i__1 = len;
|
||||
for (i__ = 1; i__ <= i__1; ++i__) {
|
||||
strue[i__ - 1] = dtrue5[i__ + (np1 + combla_1.incx * 5 <<
|
||||
strue[i__ - 1] = dtrue5[i__ + ((np1 + combla_1.incx * 5) <<
|
||||
3) - 49];
|
||||
/* L40: */
|
||||
}
|
||||
@@ -592,7 +594,8 @@ L40:
|
||||
static integer lenx, leny;
|
||||
extern real sdottest_(integer*,real*,integer*,real*,integer*);
|
||||
static integer i__, j, ksize;
|
||||
extern /* Subroutine */ int stest_(integer*,real*,real*,real*,real*), scopytest_(integer*,real*,integer*,real*,integer*), sswaptest_(integer*,real*,integer*,real*,integer*),
|
||||
extern /* Subroutine */ int stest_(integer*,real*,real*,real*,real*);
|
||||
extern void scopytest_(integer*,real*,integer*,real*,integer*), sswaptest_(integer*,real*,integer*,real*,integer*),
|
||||
saxpytest_(integer*,real*,real*,integer*,real*,integer*);
|
||||
static integer ki;
|
||||
extern /* Subroutine */ int stest1_(real*,real*,real*,real*);
|
||||
@@ -710,7 +713,8 @@ L40:
|
||||
/* Local variables */
|
||||
extern /* Subroutine */ void srottest_(integer*,real*,integer*,real*,integer*,real*,real*);
|
||||
static integer i__, k, ksize;
|
||||
extern /* Subroutine */ int stest_(integer*,real*,real*,real*,real*), srotmtest_(integer*,real*,integer*,real*,integer*,real*);
|
||||
extern /* Subroutine */ int stest_(integer*,real*,real*,real*,real*);
|
||||
extern void srotmtest_(integer*,real*,integer*,real*,integer*,real*);
|
||||
static integer ki, kn;
|
||||
static real sx[19], sy[19], sparam[5], stx[19], sty[19];
|
||||
|
||||
|
||||
+34
-55
@@ -21,19 +21,6 @@ typedef float real;
|
||||
typedef double doublereal;
|
||||
typedef struct { real r, i; } complex;
|
||||
typedef struct { doublereal r, i; } doublecomplex;
|
||||
#ifdef _MSC_VER
|
||||
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
|
||||
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
|
||||
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
|
||||
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
|
||||
#else
|
||||
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
|
||||
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
|
||||
#endif
|
||||
#define pCf(z) (*_pCf(z))
|
||||
#define pCd(z) (*_pCd(z))
|
||||
typedef int logical;
|
||||
typedef short int shortlogical;
|
||||
typedef char logical1;
|
||||
@@ -319,7 +306,7 @@ extern /* Subroutine */ int schk6_(char* sname, real* eps, real* thresh, integer
|
||||
static logical rorder;
|
||||
static integer layout;
|
||||
static logical ltestt;
|
||||
extern /* Subroutine */ int cs2chke_(char*, ftnlen);
|
||||
extern /* Subroutine */ void cs2chke_(char*);
|
||||
static logical tsterr;
|
||||
static real alf[7];
|
||||
static integer inc[7], nkb;
|
||||
@@ -702,7 +689,7 @@ L100:
|
||||
ftnlen)12);
|
||||
/* Test error exits. */
|
||||
if (tsterr) {
|
||||
cs2chke_(snames[isnum - 1], (ftnlen)12);
|
||||
cs2chke_(snames[isnum - 1]);
|
||||
}
|
||||
/* Test computations. */
|
||||
infoc_1.infot = 0;
|
||||
@@ -880,8 +867,8 @@ L240:
|
||||
static integer ia, ib, ic;
|
||||
static logical banded;
|
||||
static integer nc, nd, im, in, kl, ml, nk, nl, ku, ix, iy, ms, lx, ly, ns;
|
||||
extern /* Subroutine */ void csgbmv_(integer*, char*, integer*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void csgemv_(integer*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void csgbmv_(integer*, char*, integer*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
|
||||
extern /* Subroutine */ void csgemv_(integer*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
|
||||
static char ctrans[14];
|
||||
static real errmax;
|
||||
extern logical lseres_(char* type__, char* uplo, integer* m, integer* n, real* aa, real* as, integer* lda, ftnlen ltype_len, ftnlen uplo_len);
|
||||
@@ -1115,8 +1102,7 @@ L240:
|
||||
}
|
||||
csgemv_(iorder, trans, &m, &n, &alpha,
|
||||
&aa[1], &lda, &xx[1], &incx,
|
||||
&beta, &yy[1], &incy, (ftnlen)
|
||||
1);
|
||||
&beta, &yy[1], &incy);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1132,7 +1118,7 @@ L240:
|
||||
csgbmv_(iorder, trans, &m, &n, &kl, &
|
||||
ku, &alpha, &aa[1], &lda, &xx[
|
||||
1], &incx, &beta, &yy[1], &
|
||||
incy, (ftnlen)1);
|
||||
incy);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -1327,10 +1313,10 @@ L140:
|
||||
static integer nk, ks, ix, iy, ns, lx, ly;
|
||||
static real errmax;
|
||||
extern logical lseres_(char* , char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cssbmv_(integer*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cssbmv_(integer*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
|
||||
static real transl;
|
||||
extern /* Subroutine */ void csspmv_(integer*, char*, integer*, real*, real*, real*, integer*, real*, real*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cssymv_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void csspmv_(integer*, char*, integer*, real*, real*, real*, integer*, real*, real*, integer*);
|
||||
extern /* Subroutine */ void cssymv_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
|
||||
static integer laa, lda;
|
||||
static real als, bls;
|
||||
extern logical lse_(real*, real*, integer*);
|
||||
@@ -1531,7 +1517,7 @@ L140:
|
||||
}
|
||||
cssymv_(iorder, uplo, &n, &alpha, &aa[1],
|
||||
&lda, &xx[1], &incx, &beta, &yy[1]
|
||||
, &incy, (ftnlen)1);
|
||||
, &incy);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1546,7 +1532,7 @@ L140:
|
||||
}
|
||||
cssbmv_(iorder, uplo, &n, &k, &alpha, &aa[
|
||||
1], &lda, &xx[1], &incx, &beta, &
|
||||
yy[1], &incy, (ftnlen)1);
|
||||
yy[1], &incy);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1561,7 +1547,7 @@ L140:
|
||||
}
|
||||
csspmv_(iorder, uplo, &n, &alpha, &aa[1],
|
||||
&xx[1], &incx, &beta, &yy[1], &
|
||||
incy, (ftnlen)1);
|
||||
incy);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -1767,14 +1753,14 @@ L130:
|
||||
static char ctrans[14];
|
||||
static real errmax;
|
||||
extern logical lseres_(char*, char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cstbmv_(integer*, char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cstbmv_(integer*, char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*);
|
||||
static real transl;
|
||||
extern /* Subroutine */ void cstbsv_(integer*, char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cstbsv_(integer*, char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*);
|
||||
static char transs[1];
|
||||
extern /* Subroutine */ void cstpmv_(integer*, char*, char*, char*, integer*, real*, real*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cstrmv_(integer*, char*, char*, char*, integer*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cstpsv_(integer*, char*, char*, char*, integer*, real*, real*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cstrsv_(integer*, char*, char*, char*, integer*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cstpmv_(integer*, char*, char*, char*, integer*, real*, real*, integer*);
|
||||
extern /* Subroutine */ void cstrmv_(integer*, char*, char*, char*, integer*, real*, integer*, real*, integer*);
|
||||
extern /* Subroutine */ void cstpsv_(integer*, char*, char*, char*, integer*, real*, real*, integer*);
|
||||
extern /* Subroutine */ void cstrsv_(integer*, char*, char*, char*, integer*, real*, integer*, real*, integer*);
|
||||
static integer laa, icd, lda, ict, icu;
|
||||
extern logical lse_(real*, real*, integer*);
|
||||
static real err;
|
||||
@@ -1972,8 +1958,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cstrmv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
aa[1], &lda, &xx[1], &incx);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1987,8 +1972,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cstbmv_(iorder, uplo, trans, diag, &n, &k,
|
||||
&aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
&aa[1], &lda, &xx[1], &incx);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2002,8 +1986,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cstpmv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &xx[1], &incx, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
aa[1], &xx[1], &incx);
|
||||
}
|
||||
} else if (s_cmp(sname + 9, "sv", (ftnlen)2, (
|
||||
ftnlen)2) == 0) {
|
||||
@@ -2020,8 +2003,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cstrsv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
aa[1], &lda, &xx[1], &incx);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2035,8 +2017,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cstbsv_(iorder, uplo, trans, diag, &n, &k,
|
||||
&aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
&aa[1], &lda, &xx[1], &incx);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2050,8 +2031,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cstpsv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &xx[1], &incx, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
aa[1], &xx[1], &incx);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2585,10 +2565,10 @@ L150:
|
||||
static logical reset;
|
||||
static char cuplo[14];
|
||||
static integer incxs;
|
||||
extern /* Subroutine */ void csspr_(integer*, char*, integer*, real*, real*, integer*, real*, ftnlen);
|
||||
extern /* Subroutine */ void csspr_(integer*, char*, integer*, real*, real*, integer*, real*);
|
||||
static logical upper;
|
||||
static char uplos[1];
|
||||
extern /* Subroutine */ void cssyr_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cssyr_(integer*, char*, integer*, real*, real*, integer*, real*, integer*);
|
||||
static integer ia, ja, ic, nc, jj, lj, in;
|
||||
static logical packed;
|
||||
static integer ix, ns, lx;
|
||||
@@ -2747,7 +2727,7 @@ L150:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cssyr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]
|
||||
, &lda, (ftnlen)1);
|
||||
, &lda);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2760,8 +2740,7 @@ L150:
|
||||
al__1.aunit = *ntra;
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
csspr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]
|
||||
, (ftnlen)1);
|
||||
csspr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -2945,13 +2924,13 @@ L130:
|
||||
static logical upper;
|
||||
static char uplos[1];
|
||||
static integer ia, ja, ic;
|
||||
extern /* Subroutine */ void csspr2_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*, ftnlen);
|
||||
extern /* Subroutine */ void csspr2_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*);
|
||||
static integer nc, jj, lj, in;
|
||||
static logical packed;
|
||||
extern /* Subroutine */ void cssyr2_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void cssyr2_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*, integer*);
|
||||
static integer ix, iy, ns, lx, ly;
|
||||
static real errmax;
|
||||
extern logical lseres_(char* type__, char* uplo, integer* m, integer* n, real* aa, real* as, integer* lda, ftnlen ltype_len, ftnlen uplo_len);
|
||||
extern logical lseres_(char* type__, char* uplo, integer* m, integer* n, real* aa, real* as, integer* lda, ftnlen, ftnlen);
|
||||
static real transl;
|
||||
static integer laa, lda;
|
||||
static real als;
|
||||
@@ -3131,7 +3110,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cssyr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
|
||||
yy[1], &incy, &aa[1], &lda, (ftnlen)1);
|
||||
yy[1], &incy, &aa[1], &lda);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -3145,7 +3124,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
csspr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
|
||||
yy[1], &incy, &aa[1], (ftnlen)1);
|
||||
yy[1], &incy, &aa[1]);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -3380,7 +3359,7 @@ L170:
|
||||
i__2 = *m;
|
||||
for (i__ = 1; i__ <= i__2; ++i__) {
|
||||
if (gen || (upper && i__ <= j) || (lower && i__ >= j)) {
|
||||
if (i__ <= j && (j - i__ <= *ku || i__ >= j && i__ - j <= *kl))
|
||||
if (((i__ <= j && j - i__ <= *ku) || (i__ >= j && i__ - j <= *kl)))
|
||||
{
|
||||
a[i__ + j * a_dim1] = sbeg_(reset) + *transl;
|
||||
} else {
|
||||
|
||||
+15
-33
@@ -21,19 +21,6 @@ typedef float real;
|
||||
typedef double doublereal;
|
||||
typedef struct { real r, i; } complex;
|
||||
typedef struct { doublereal r, i; } doublecomplex;
|
||||
#ifdef _MSC_VER
|
||||
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
|
||||
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
|
||||
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
|
||||
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
|
||||
#else
|
||||
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
|
||||
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
|
||||
#endif
|
||||
#define pCf(z) (*_pCf(z))
|
||||
#define pCd(z) (*_pCd(z))
|
||||
typedef int logical;
|
||||
typedef short int shortlogical;
|
||||
typedef char logical1;
|
||||
@@ -309,7 +296,7 @@ static logical c_false = FALSE_;
|
||||
static logical rorder;
|
||||
static integer layout;
|
||||
static logical ltestt, tsterr;
|
||||
extern /* Subroutine */ void cs3chke_(char*, ftnlen);
|
||||
extern /* Subroutine */ void cs3chke_(char*);
|
||||
static real alf[7], bet[7];
|
||||
extern logical lse_(real*, real*, integer*);
|
||||
static real eps, err;
|
||||
@@ -522,7 +509,7 @@ L30:
|
||||
if (i__1 < 2) {
|
||||
goto L60;
|
||||
}
|
||||
for (i__ = 1; i__ <= 9; ++i__) {
|
||||
for (i__ = 1; i__ <= 6; ++i__) {
|
||||
if (s_cmp(snamet, snames[i__ - 1] , (ftnlen)12, (ftnlen)12) ==
|
||||
0) {
|
||||
goto L50;
|
||||
@@ -656,7 +643,7 @@ L80:
|
||||
ftnlen)12);
|
||||
/* Test error exits. */
|
||||
if (tsterr) {
|
||||
cs3chke_(snames[isnum - 1], (ftnlen)12);
|
||||
cs3chke_(snames[isnum - 1]);
|
||||
}
|
||||
/* Test computations. */
|
||||
infoc_1.infot = 0;
|
||||
@@ -800,7 +787,7 @@ L230:
|
||||
extern /* Subroutine */ int smake_(char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*, logical*, real*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ int smmch_(char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, real*, real*, real*, integer*, real*, real*, logical*, integer*, logical*, ftnlen, ftnlen);
|
||||
static integer ia, ib, ma, mb, na, nb, nc, ik, im, in, ks, ms, ns;
|
||||
extern /* Subroutine */ void csgemm_(integer*, char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void csgemm_(integer*, char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
|
||||
static char tranas[1], tranbs[1], transa[1], transb[1];
|
||||
static real errmax;
|
||||
extern logical lseres_(char*, char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
|
||||
@@ -1003,8 +990,7 @@ L230:
|
||||
}
|
||||
csgemm_(iorder, transa, transb, &m, &n, &k, &
|
||||
alpha, &aa[1], &lda, &bb[1], &ldb, &
|
||||
beta, &cc[1], &ldc, (ftnlen)1, (
|
||||
ftnlen)1);
|
||||
beta, &cc[1], &ldc);
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
@@ -1197,7 +1183,7 @@ L130:
|
||||
static integer ia, ib, na, nc, im, in, ms, ns;
|
||||
static real errmax;
|
||||
extern logical lseres_(char*, char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cssymm_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cssymm_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
|
||||
extern void sprcn2_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, real*, integer*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ int smake_(char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*, logical*, real*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ int smmch_(char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, real*, real*, real*, integer*, real*, real*, logical*, integer*, logical*, ftnlen, ftnlen);
|
||||
@@ -1378,8 +1364,7 @@ L130:
|
||||
// f_rew(&al__1);
|
||||
}
|
||||
cssymm_(iorder, side, uplo, &m, &n, &alpha, &aa[1]
|
||||
, &lda, &bb[1], &ldb, &beta, &cc[1], &ldc,
|
||||
(ftnlen)1, (ftnlen)1);
|
||||
, &lda, &bb[1], &ldb, &beta, &cc[1], &ldc);
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
@@ -1575,8 +1560,8 @@ L120:
|
||||
extern /* Subroutine */ int smake_(char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*, logical*, real*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ int smmch_(char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, real*, real*, real*, integer*, real*, real*, logical*, integer*, logical*, ftnlen, ftnlen);
|
||||
extern logical lseres_(char*, char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cstrmm_(integer*, char*, char*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cstrsm_(integer*, char*, char*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cstrmm_(integer*, char*, char*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*);
|
||||
extern /* Subroutine */ void cstrsm_(integer*, char*, char*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*);
|
||||
static integer laa, icd, lbb, lda, ldb, ics;
|
||||
static real als;
|
||||
static integer ict, icu;
|
||||
@@ -1752,8 +1737,7 @@ L120:
|
||||
}
|
||||
cstrmm_(iorder, side, uplo, transa, diag,
|
||||
&m, &n, &alpha, &aa[1], &lda, &bb[
|
||||
1], &ldb, (ftnlen)1, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
1], &ldb);
|
||||
} else if (s_cmp(sname + 9, "sm", (ftnlen)2, (
|
||||
ftnlen)2) == 0) {
|
||||
if (*trace) {
|
||||
@@ -1768,8 +1752,7 @@ L120:
|
||||
}
|
||||
cstrsm_(iorder, side, uplo, transa, diag,
|
||||
&m, &n, &alpha, &aa[1], &lda, &bb[
|
||||
1], &ldb, (ftnlen)1, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
1], &ldb);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -2028,7 +2011,7 @@ L160:
|
||||
static real errmax;
|
||||
extern logical lseres_(char*, char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
|
||||
static char transs[1];
|
||||
extern /* Subroutine */ void cssyrk_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, real*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cssyrk_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, real*, integer*);
|
||||
static integer laa, lda, lcc, ldc;
|
||||
static real als;
|
||||
static integer ict, icu;
|
||||
@@ -2186,8 +2169,7 @@ L160:
|
||||
// f_rew(&al__1);
|
||||
}
|
||||
cssyrk_(iorder, uplo, trans, &n, &k, &alpha, &aa[
|
||||
1], &lda, &beta, &cc[1], &ldc, (ftnlen)1,
|
||||
(ftnlen)1);
|
||||
1], &lda, &beta, &cc[1], &ldc);
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
@@ -2409,7 +2391,7 @@ L130:
|
||||
static integer laa, lbb, lda, lcc, ldb, ldc;
|
||||
static real als;
|
||||
static integer ict, icu;
|
||||
extern /* Subroutine */ void cssyr2k_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cssyr2k_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
|
||||
extern logical lse_(real*, real*, integer*);
|
||||
extern /* Subroutine */ int smmch_(char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, real*, real*, real*, integer*, real*, real*, logical*, integer*, logical*, ftnlen, ftnlen);
|
||||
static real err;
|
||||
@@ -2591,7 +2573,7 @@ L130:
|
||||
}
|
||||
cssyr2k_(iorder, uplo, trans, &n, &k, &alpha, &aa[
|
||||
1], &lda, &bb[1], &ldb, &beta, &cc[1], &
|
||||
ldc, (ftnlen)1, (ftnlen)1);
|
||||
ldc);
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
|
||||
+11
-18
@@ -22,18 +22,10 @@ typedef double doublereal;
|
||||
typedef struct { real r, i; } complex;
|
||||
typedef struct { doublereal r, i; } doublecomplex;
|
||||
#ifdef _MSC_VER
|
||||
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
|
||||
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
|
||||
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
|
||||
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
|
||||
#else
|
||||
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
|
||||
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
|
||||
#endif
|
||||
#define pCf(z) (*_pCf(z))
|
||||
#define pCd(z) (*_pCd(z))
|
||||
typedef int logical;
|
||||
typedef short int shortlogical;
|
||||
typedef char logical1;
|
||||
@@ -380,11 +372,12 @@ static doublereal c_b43 = 1.;
|
||||
static integer i__;
|
||||
extern /* Subroutine */ int ctest_(integer*, doublecomplex*, doublecomplex*, doublecomplex*, doublereal*);
|
||||
static doublecomplex mwpcs[5], mwpct[5];
|
||||
extern /* Subroutine */ int zscaltest_(integer*, doublecomplex*, doublecomplex*, integer*), itest1_(integer*, integer*), stest1_(doublereal*, doublereal*, doublereal*, doublereal*);
|
||||
extern /* Subroutine */ void zscaltest_(integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
extern int itest1_(integer*, integer*), stest1_(doublereal*, doublereal*, doublereal*, doublereal*);
|
||||
static doublecomplex cx[8];
|
||||
extern doublereal dznrm2test_(integer*, doublecomplex*, integer*);
|
||||
static integer np1;
|
||||
extern /* Subroutine */ int zdscaltest_(integer*, doublereal*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void zdscaltest_(integer*, doublereal*, doublecomplex*, integer*);
|
||||
extern integer izamaxtest_(integer*, doublecomplex*, integer*);
|
||||
extern doublereal dzasumtest_(integer*, doublecomplex*, integer*);
|
||||
static integer len;
|
||||
@@ -408,7 +401,7 @@ static doublereal c_b43 = 1.;
|
||||
i__1 = len;
|
||||
for (i__ = 1; i__ <= i__1; ++i__) {
|
||||
i__2 = i__ - 1;
|
||||
i__3 = i__ + (np1 + combla_1.incx * 5 << 3) - 49;
|
||||
i__3 = i__ + ((np1 + combla_1.incx * 5) << 3) - 49;
|
||||
cx[i__2].r = cv[i__3].r, cx[i__2].i = cv[i__3].i;
|
||||
/* L20: */
|
||||
}
|
||||
@@ -423,13 +416,13 @@ static doublereal c_b43 = 1.;
|
||||
} else if (combla_1.icase == 8) {
|
||||
/* .. ZSCALTEST .. */
|
||||
zscaltest_(&combla_1.n, &ca, cx, &combla_1.incx);
|
||||
ctest_(&len, cx, &ctrue5[(np1 + combla_1.incx * 5 << 3) - 48],
|
||||
&ctrue5[(np1 + combla_1.incx * 5 << 3) - 48], sfac);
|
||||
ctest_(&len, cx, &ctrue5[((np1 + combla_1.incx * 5) << 3) - 48],
|
||||
&ctrue5[((np1 + combla_1.incx * 5) << 3) - 48], sfac);
|
||||
} else if (combla_1.icase == 9) {
|
||||
/* .. ZDSCALTEST .. */
|
||||
zdscaltest_(&combla_1.n, &sa, cx, &combla_1.incx);
|
||||
ctest_(&len, cx, &ctrue6[(np1 + combla_1.incx * 5 << 3) - 48],
|
||||
&ctrue6[(np1 + combla_1.incx * 5 << 3) - 48], sfac);
|
||||
ctest_(&len, cx, &ctrue6[((np1 + combla_1.incx * 5) << 3) - 48],
|
||||
&ctrue6[((np1 + combla_1.incx * 5) << 3) - 48], sfac);
|
||||
} else if (combla_1.icase == 10) {
|
||||
/* .. IZAMAXTEST .. */
|
||||
i__1 = izamaxtest_(&combla_1.n, cx, &combla_1.incx);
|
||||
@@ -591,11 +584,11 @@ static doublereal c_b43 = 1.;
|
||||
extern /* Subroutine */ int ctest_(integer*, doublecomplex*, doublecomplex*, doublecomplex*, doublereal*);
|
||||
static integer ksize;
|
||||
static doublecomplex ztemp;
|
||||
extern /* Subroutine */ int zdotctest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*), zcopytest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void zdotctest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*), zcopytest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
static integer ki;
|
||||
extern /* Subroutine */ int zdotutest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*), zswaptest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void zdotutest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*), zswaptest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
static integer kn;
|
||||
extern /* Subroutine */ int zaxpytest_(integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void zaxpytest_(integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
static doublecomplex cx[7], cy[7];
|
||||
static integer mx, my;
|
||||
|
||||
|
||||
+32
-48
@@ -22,17 +22,12 @@ typedef double doublereal;
|
||||
typedef struct { real r, i; } complex;
|
||||
typedef struct { doublereal r, i; } doublecomplex;
|
||||
#ifdef _MSC_VER
|
||||
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
|
||||
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
|
||||
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
|
||||
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
|
||||
#else
|
||||
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
|
||||
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
|
||||
#endif
|
||||
#define pCf(z) (*_pCf(z))
|
||||
#define pCd(z) (*_pCd(z))
|
||||
typedef int logical;
|
||||
typedef short int shortlogical;
|
||||
@@ -322,7 +317,7 @@ static logical c_false = FALSE_;
|
||||
static logical rorder;
|
||||
static integer layout;
|
||||
static logical ltestt, tsterr;
|
||||
extern /* Subroutine */ void cz2chke_(char*, ftnlen);
|
||||
extern /* Subroutine */ void cz2chke_(char*);
|
||||
static doublecomplex alf[7];
|
||||
static integer inc[7], nkb;
|
||||
static doublecomplex bet[7];
|
||||
@@ -713,7 +708,7 @@ L100:
|
||||
ftnlen)12);
|
||||
/* Test error exits. */
|
||||
if (tsterr) {
|
||||
cz2chke_(snames[isnum - 1], (ftnlen)12);
|
||||
cz2chke_(snames[isnum - 1]);
|
||||
}
|
||||
/* Test computations. */
|
||||
infoc_1.infot = 0;
|
||||
@@ -893,9 +888,9 @@ L240:
|
||||
static integer ia, ib, ic;
|
||||
static logical banded;
|
||||
static integer nc, nd, im, in, kl, ml, nk, nl, ku, ix, iy, ms, lx, ly, ns;
|
||||
extern /* Subroutine */ void czgbmv_(integer*, char*, integer*, integer*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void czgbmv_(integer*, char*, integer*, integer*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
static char ctrans[14];
|
||||
extern /* Subroutine */ void czgemv_(integer*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void czgemv_(integer*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
static doublereal errmax;
|
||||
static doublecomplex transl;
|
||||
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
@@ -1144,8 +1139,7 @@ L240:
|
||||
}
|
||||
czgemv_(iorder, trans, &m, &n, &alpha,
|
||||
&aa[1], &lda, &xx[1], &incx,
|
||||
&beta, &yy[1], &incy, (ftnlen)
|
||||
1);
|
||||
&beta, &yy[1], &incy);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1160,8 +1154,7 @@ L240:
|
||||
}
|
||||
czgbmv_(iorder, trans, &m, &n, &kl, &
|
||||
ku, &alpha, &aa[1], &lda, &xx[
|
||||
1], &incx, &beta, &yy[1], &
|
||||
incy, (ftnlen)1);
|
||||
1], &incx, &beta, &yy[1], &incy);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -1349,12 +1342,12 @@ L140:
|
||||
static integer nc, ik, in;
|
||||
static logical packed;
|
||||
static integer nk, ks, ix, iy, ns, lx, ly;
|
||||
extern /* Subroutine */ void czhbmv_(integer*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void czhemv_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void czhbmv_(integer*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void czhemv_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
static doublereal errmax;
|
||||
static doublecomplex transl;
|
||||
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void czhpmv_(integer*, char*, integer*, doublecomplex*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void czhpmv_(integer*, char*, integer*, doublecomplex*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
static integer laa, lda;
|
||||
static doublecomplex als, bls;
|
||||
static doublereal err;
|
||||
@@ -1568,7 +1561,7 @@ L140:
|
||||
}
|
||||
czhemv_(iorder, uplo, &n, &alpha, &aa[1],
|
||||
&lda, &xx[1], &incx, &beta, &yy[1]
|
||||
, &incy, (ftnlen)1);
|
||||
, &incy);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1583,7 +1576,7 @@ L140:
|
||||
}
|
||||
czhbmv_(iorder, uplo, &n, &k, &alpha, &aa[
|
||||
1], &lda, &xx[1], &incx, &beta, &
|
||||
yy[1], &incy, (ftnlen)1);
|
||||
yy[1], &incy);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -1597,8 +1590,7 @@ L140:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
czhpmv_(iorder, uplo, &n, &alpha, &aa[1],
|
||||
&xx[1], &incx, &beta, &yy[1], &
|
||||
incy, (ftnlen)1);
|
||||
&xx[1], &incx, &beta, &yy[1], &incy);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -1798,13 +1790,13 @@ L130:
|
||||
static doublereal errmax;
|
||||
static doublecomplex transl;
|
||||
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cztbmv_(integer*, char*, char*, char*, integer*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cztbmv_(integer*, char*, char*, char*, integer*, integer*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
static char transs[1];
|
||||
extern /* Subroutine */ void cztbsv_(integer*, char*, char*, char*, integer*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cztpmv_(integer*, char*, char*, char*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cztpsv_(integer*, char*, char*, char*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cztrmv_(integer*, char*, char*, char*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cztrsv_(integer*, char*, char*, char*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cztbsv_(integer*, char*, char*, char*, integer*, integer*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void cztpmv_(integer*, char*, char*, char*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void cztpsv_(integer*, char*, char*, char*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void cztrmv_(integer*, char*, char*, char*, integer*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void cztrsv_(integer*, char*, char*, char*, integer*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
static integer laa, icd, lda, ict, icu;
|
||||
static doublereal err;
|
||||
extern logical lze_(doublecomplex*, doublecomplex*, integer*);
|
||||
@@ -2014,8 +2006,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cztrmv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
aa[1], &lda, &xx[1], &incx);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2029,8 +2020,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cztbmv_(iorder, uplo, trans, diag, &n, &k,
|
||||
&aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
&aa[1], &lda, &xx[1], &incx);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2044,8 +2034,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cztpmv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &xx[1], &incx, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
aa[1], &xx[1], &incx);
|
||||
}
|
||||
} else if (s_cmp(sname + 9, "sv", (ftnlen)2, (
|
||||
ftnlen)2) == 0) {
|
||||
@@ -2062,8 +2051,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cztrsv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
aa[1], &lda, &xx[1], &incx);
|
||||
} else if (banded) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2077,8 +2065,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cztbsv_(iorder, uplo, trans, diag, &n, &k,
|
||||
&aa[1], &lda, &xx[1], &incx, (
|
||||
ftnlen)1, (ftnlen)1, (ftnlen)1);
|
||||
&aa[1], &lda, &xx[1], &incx);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2092,8 +2079,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
cztpsv_(iorder, uplo, trans, diag, &n, &
|
||||
aa[1], &xx[1], &incx, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
aa[1], &xx[1], &incx);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2644,11 +2630,11 @@ L150:
|
||||
static logical isame[13];
|
||||
extern /* Subroutine */ int zmake_(char*, char*, char*, integer*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, integer*, integer*, logical*, doublecomplex*, ftnlen, ftnlen, ftnlen);
|
||||
static integer nargs;
|
||||
extern /* Subroutine */ void czher_(integer*, char*, integer*, doublereal*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void czher_(integer*, char*, integer*, doublereal*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
static logical reset;
|
||||
static char cuplo[14];
|
||||
static integer incxs;
|
||||
extern /* Subroutine */ void czhpr_(integer*, char*, integer*, doublereal*, doublecomplex*, integer*, doublecomplex*, ftnlen);
|
||||
extern /* Subroutine */ void czhpr_(integer*, char*, integer*, doublereal*, doublecomplex*, integer*, doublecomplex*);
|
||||
extern /* Subroutine */ int zmvch_(char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublereal*, doublecomplex*, doublereal*, doublereal*, logical*, integer*, logical*, ftnlen);
|
||||
static logical upper;
|
||||
static char uplos[1];
|
||||
@@ -2817,8 +2803,7 @@ L150:
|
||||
al__1.aunit = *ntra;
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
czher_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[
|
||||
1], &lda, (ftnlen)1);
|
||||
czher_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[1], &lda);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -2831,8 +2816,7 @@ L150:
|
||||
al__1.aunit = *ntra;
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
czhpr_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[
|
||||
1], (ftnlen)1);
|
||||
czhpr_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[1]);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -3011,8 +2995,8 @@ L130:
|
||||
extern /* Subroutine */ int zmvch_(char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublereal*, doublecomplex*, doublereal*, doublereal*, logical*, integer*, logical*, ftnlen);
|
||||
static logical upper;
|
||||
static char uplos[1];
|
||||
extern /* Subroutine */ void czher2_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen);
|
||||
extern /* Subroutine */ void czhpr2_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, ftnlen);
|
||||
extern /* Subroutine */ void czher2_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void czhpr2_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*);
|
||||
static integer ia, ja, ic, nc, jj, lj, in;
|
||||
static logical packed;
|
||||
static integer ix, iy, ns, lx, ly;
|
||||
@@ -3208,7 +3192,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
czher2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
|
||||
yy[1], &incy, &aa[1], &lda, (ftnlen)1);
|
||||
yy[1], &incy, &aa[1], &lda);
|
||||
} else if (packed) {
|
||||
if (*trace) {
|
||||
/*
|
||||
@@ -3222,7 +3206,7 @@ L130:
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
czhpr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
|
||||
yy[1], &incy, &aa[1], (ftnlen)1);
|
||||
yy[1], &incy, &aa[1]);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
+20
-27
@@ -25,11 +25,9 @@ typedef struct { doublereal r, i; } doublecomplex;
|
||||
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
|
||||
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
|
||||
#else
|
||||
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
|
||||
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
|
||||
#endif
|
||||
#define pCf(z) (*_pCf(z))
|
||||
#define pCd(z) (*_pCd(z))
|
||||
typedef int logical;
|
||||
typedef short int shortlogical;
|
||||
@@ -314,7 +312,7 @@ static logical c_false = FALSE_;
|
||||
static logical rorder;
|
||||
static integer layout;
|
||||
static logical ltestt, tsterr;
|
||||
extern /* Subroutine */ int cz3chke_(char*, ftnlen);
|
||||
extern /* Subroutine */ void cz3chke_(char*);
|
||||
static doublecomplex alf[7], bet[7];
|
||||
static doublereal eps, err;
|
||||
extern logical lze_(doublecomplex*, doublecomplex*, integer*);
|
||||
@@ -679,7 +677,7 @@ L80:
|
||||
ftnlen)12);
|
||||
/* Test error exits. */
|
||||
if (tsterr) {
|
||||
cz3chke_(snames[isnum - 1], (ftnlen)12);
|
||||
cz3chke_(snames[isnum - 1]);
|
||||
}
|
||||
/* Test computations. */
|
||||
infoc_1.infot = 0;
|
||||
@@ -831,7 +829,7 @@ L230:
|
||||
static integer ia, ib;
|
||||
extern /* Subroutine */ int zprcn1_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, integer*, doublecomplex*, integer*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
static integer ma, mb, na, nb, nc, ik, im, in, ks, ms, ns;
|
||||
extern /* Subroutine */ void czgemm_(integer*, char*, char*, integer*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void czgemm_(integer*, char*, char*, integer*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
static char tranas[1], tranbs[1], transa[1], transb[1];
|
||||
static doublereal errmax;
|
||||
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
@@ -1047,8 +1045,7 @@ L230:
|
||||
}
|
||||
czgemm_(iorder, transa, transb, &m, &n, &k, &
|
||||
alpha, &aa[1], &lda, &bb[1], &ldb, &
|
||||
beta, &cc[1], &ldc, (ftnlen)1, (
|
||||
ftnlen)1);
|
||||
beta, &cc[1], &ldc);
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
@@ -1242,10 +1239,10 @@ return 0;
|
||||
static integer ia, ib;
|
||||
extern /* Subroutine */ int zprcn2_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, doublecomplex*, integer*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
static integer na, nc, im, in, ms, ns;
|
||||
extern /* Subroutine */ void czhemm_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void czhemm_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
static doublereal errmax;
|
||||
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void czsymm_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void czsymm_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
static integer laa, lbb, lda, lcc, ldb, ldc, ics;
|
||||
static doublecomplex als, bls;
|
||||
static integer icu;
|
||||
@@ -1438,11 +1435,11 @@ return 0;
|
||||
if (isconj) {
|
||||
czhemm_(iorder, side, uplo, &m, &n, &alpha, &
|
||||
aa[1], &lda, &bb[1], &ldb, &beta, &cc[
|
||||
1], &ldc, (ftnlen)1, (ftnlen)1);
|
||||
1], &ldc);
|
||||
} else {
|
||||
czsymm_(iorder, side, uplo, &m, &n, &alpha, &
|
||||
aa[1], &lda, &bb[1], &ldb, &beta, &cc[
|
||||
1], &ldc, (ftnlen)1, (ftnlen)1);
|
||||
1], &ldc);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -1641,8 +1638,8 @@ return 0;
|
||||
static char tranas[1], transa[1];
|
||||
static doublereal errmax;
|
||||
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cztrmm_(integer*, char*, char*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cztrsm_(integer*, char*, char*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void cztrmm_(integer*, char*, char*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
extern /* Subroutine */ void cztrsm_(integer*, char*, char*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*);
|
||||
static integer laa, icd, lbb, lda, ldb, ics;
|
||||
static doublecomplex als;
|
||||
static integer ict, icu;
|
||||
@@ -1828,8 +1825,7 @@ return 0;
|
||||
}
|
||||
cztrmm_(iorder, side, uplo, transa, diag,
|
||||
&m, &n, &alpha, &aa[1], &lda, &bb[
|
||||
1], &ldb, (ftnlen)1, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
1], &ldb);
|
||||
} else if (s_cmp(sname + 9, "sm", (ftnlen)2, (
|
||||
ftnlen)2) == 0) {
|
||||
if (*trace) {
|
||||
@@ -1846,8 +1842,7 @@ return 0;
|
||||
}
|
||||
cztrsm_(iorder, side, uplo, transa, diag,
|
||||
&m, &n, &alpha, &aa[1], &lda, &bb[
|
||||
1], &ldb, (ftnlen)1, (ftnlen)1, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
1], &ldb);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -2119,11 +2114,11 @@ return 0;
|
||||
extern /* Subroutine */ int zprcn6_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
|
||||
static integer ik, in, jj, lj, ks, ns;
|
||||
static doublereal ralpha;
|
||||
extern /* Subroutine */ int czherk_(integer*, char*, char*, integer*, integer*, doublereal*, doublecomplex*, integer*, doublereal*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void czherk_(integer*, char*, char*, integer*, integer*, doublereal*, doublecomplex*, integer*, doublereal*, doublecomplex*, integer*);
|
||||
static doublereal errmax;
|
||||
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
static char transs[1], transt[1];
|
||||
extern /* Subroutine */ int czsyrk_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void czsyrk_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
static integer laa, lda, lcc, ldc;
|
||||
static doublecomplex als;
|
||||
static integer ict, icu;
|
||||
@@ -2319,8 +2314,7 @@ return 0;
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
czherk_(iorder, uplo, trans, &n, &k, &ralpha,
|
||||
&aa[1], &lda, &rbeta, &cc[1], &ldc, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
&aa[1], &lda, &rbeta, &cc[1], &ldc);
|
||||
} else {
|
||||
if (*trace) {
|
||||
zprcn4_(ntra, &nc, sname, iorder, uplo,
|
||||
@@ -2334,8 +2328,7 @@ return 0;
|
||||
f_rew(&al__1);*/
|
||||
}
|
||||
czsyrk_(iorder, uplo, trans, &n, &k, &alpha, &
|
||||
aa[1], &lda, &beta, &cc[1], &ldc, (
|
||||
ftnlen)1, (ftnlen)1);
|
||||
aa[1], &lda, &beta, &cc[1], &ldc);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
@@ -2615,11 +2608,11 @@ return 0;
|
||||
static doublereal errmax;
|
||||
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
static char transs[1], transt[1];
|
||||
extern /* Subroutine */ int czher2k_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublereal*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void czher2k_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublereal*, doublecomplex*, integer*);
|
||||
static integer laa, lbb, lda, lcc, ldb, ldc;
|
||||
static doublecomplex als;
|
||||
static integer ict, icu;
|
||||
extern /* Subroutine */ int czsyr2k_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
|
||||
extern /* Subroutine */ void czsyr2k_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
|
||||
static doublereal err;
|
||||
extern logical lze_(doublecomplex*, doublecomplex*, integer*);
|
||||
|
||||
@@ -2830,7 +2823,7 @@ return 0;
|
||||
}
|
||||
czher2k_(iorder, uplo, trans, &n, &k, &alpha,
|
||||
&aa[1], &lda, &bb[1], &ldb, &rbeta, &
|
||||
cc[1], &ldc, (ftnlen)1, (ftnlen)1);
|
||||
cc[1], &ldc);
|
||||
} else {
|
||||
if (*trace) {
|
||||
zprcn5_(ntra, &nc, sname, iorder, uplo,
|
||||
@@ -2845,7 +2838,7 @@ return 0;
|
||||
}
|
||||
czsyr2k_(iorder, uplo, trans, &n, &k, &alpha,
|
||||
&aa[1], &lda, &bb[1], &ldb, &beta, &
|
||||
cc[1], &ldc, (ftnlen)1, (ftnlen)1);
|
||||
cc[1], &ldc);
|
||||
}
|
||||
|
||||
/* Check if error-exit was taken incorrectly. */
|
||||
|
||||
+13
-1
@@ -13,7 +13,9 @@ This page documents those non-standard APIs.
|
||||
| ?omatcopy | s,d,c,z | out-of-place transposition/copying |
|
||||
| ?geadd | s,d,c,z | ATLAS-like matrix add `B = α*A+β*B` |
|
||||
| ?gemmt | s,d,c,z | `gemm` but only a triangular part updated |
|
||||
|
||||
| cblas_?gemm_batch | s,d,c,z,b | `gemm` with several groups of input data
|
||||
|
|
||||
| cblas_?gemm_batch_strided | s,d,c,z,b | `gemm` with groups of data stored at fixed offsets in the input arrays
|
||||
|
||||
## bfloat16 functionality
|
||||
|
||||
@@ -26,6 +28,15 @@ BLAS-like and conversion functions for `bfloat16` (available when OpenBLAS was c
|
||||
* `float cblas_sbdot` computes the dot product of two bfloat16 arrays
|
||||
* `void cblas_sbgemv` performs the matrix-vector operations of GEMV with the input matrix and X vector as bfloat16
|
||||
* `void cblas_sbgemm` performs the matrix-matrix operations of GEMM with both input arrays containing bfloat16
|
||||
* `void cblas_bgemv` performs the matrix-vector operations of GEMV with the input matrix, X vector and result as bfloat16
|
||||
* `void cblas_bgemm` performs the matrix-matrix operations of GEMM with both input arrays containing bfloat16 and the output being bfloat16 as well
|
||||
|
||||
## half-precision float or fp16 functionality
|
||||
|
||||
BLAS-like and conversion functions for `hfloat16` (available when OpenBLAS was compiled with `BUILD_HFLOAT16=1`):
|
||||
|
||||
* `void cblas_shgemm` performs the matrix-matrix operations of GEMM with both input arrays containing hfloat16
|
||||
|
||||
|
||||
## Utility functions
|
||||
|
||||
@@ -36,4 +47,5 @@ BLAS-like and conversion functions for `bfloat16` (available when OpenBLAS was c
|
||||
* `char * openblas_get_config()` returns the options OpenBLAS was built with, something like `NO_LAPACKE DYNAMIC_ARCH NO_AFFINITY Haswell`
|
||||
* `int openblas_set_affinity(int thread_index, size_t cpusetsize, cpu_set_t *cpuset)` sets the CPU affinity mask of the given thread
|
||||
to the provided cpuset. Only available on Linux, with semantics identical to `pthread_setaffinity_np`.
|
||||
* `openblas_set_thread_callback_function` overrides the default multithreading backend with the provided argument
|
||||
|
||||
|
||||
+8
-2
@@ -47,7 +47,8 @@ You can find the full list of modifications in Changelog.txt.
|
||||
The detailed explanation is probably in the original publication authored by Kazushige Goto - Goto, Kazushige; van de Geijn, Robert A; Anatomy of high-performance matrix multiplication. ACM Transactions on Mathematical Software (TOMS). Volume 34 Issue 3, May 2008
|
||||
While this article is paywalled and too old for preprints to be available on arxiv.org, more recent
|
||||
publications like https://arxiv.org/pdf/1609.00076 contain at least a brief description of the algorithm.
|
||||
In practice, the values are derived by experimentation to yield the block sizes that give the highest performance. A general rule of thumb for selecting a starting point seems to be that PxQ is about half the size of L2 cache.
|
||||
In practice, the values are derived by experimentation to yield the block sizes that give the highest performance. A general rule of thumb for selecting a starting point seems to be that PxQ is about half the size of L2 cache. R needs to be greater than the bigger of P and Q by
|
||||
at least 64, or bad things will happen with the work splitting in (at least) POTRF.
|
||||
|
||||
### <a name="reportbug"></a>How can I report a bug?
|
||||
|
||||
@@ -344,7 +345,12 @@ Multithreading support in OpenBLAS requires the use of internal buffers for shar
|
||||
If you get a message "error while loading shared libraries: libopenblas.so.0: ELF load command address/offset not properly aligned" when starting a program that is (dynamically) linked to OpenBLAS, this is very likely due to a bug in the GNU linker (ld) that is part of the
|
||||
GNU binutils package. This error was specifically observed on older versions of Ubuntu Linux updated with the (at the time) most recent binutils version 2.38, but an internet search turned up sporadic reports involving various other libraries dating back several years. A bugfix was created by the binutils developers and should be available in later versions of binutils.(See issue 3708 for details)
|
||||
|
||||
#### <a name="OpenMP"></a>Using OpenBLAS with OpenMP
|
||||
### <a name="CallingConvention"></a>The tests work fine, but calling any complex function from my code produces wrong or no results
|
||||
|
||||
This is almost certainly a problem with the calling convention used, in particular with the way the computed result is transported back to the caller. By default, OpenBLAS follows the F2C convention of returning the result on the stack rather than as the first argument to the function. So if your code has a prototype like "void cdotu ( complex *res, int n,...)" change it to "complex cdotu (int n,...)". Better yet,
|
||||
use the CBLAS interface rather than the Fortran one.
|
||||
|
||||
### <a name="OpenMP"></a>Using OpenBLAS with OpenMP
|
||||
|
||||
OpenMP provides its own locking mechanisms, so when your code makes BLAS/LAPACK calls from inside OpenMP parallel regions it is imperative
|
||||
that you use an OpenBLAS that is built with USE_OPENMP=1, as otherwise deadlocks might occur. Furthermore, OpenBLAS will automatically restrict itself to using only a single thread when called from an OpenMP parallel region. When it is certain that calls will only occur
|
||||
|
||||
+19
-10
@@ -217,8 +217,11 @@ in this section, since the process for each is quite different.
|
||||
For Visual Studio, you can use CMake to generate Visual Studio solution files;
|
||||
note that you will need at least CMake 3.11 for linking to work correctly).
|
||||
|
||||
Note that you need a Fortran compiler if you plan to build and use the LAPACK
|
||||
functions included with OpenBLAS. The sections below describe using either
|
||||
Note that you need a Fortran compiler if you plan to build and use the latest version
|
||||
of the LAPACK functions included with OpenBLAS. (If you do not have a Fortran compiler
|
||||
installed, you can build an older version of the LAPACK sources that has been converted
|
||||
to C - but its performance will likely be slower and accuracy may be poorer too.)
|
||||
The sections below describe using either
|
||||
`flang` as an add-on to clang/LLVM or `gfortran` as part of MinGW for this
|
||||
purpose. If you want to use the Intel Fortran compiler (`ifort` or `ifx`) for
|
||||
this, be sure to also use the Intel C compiler (`icc` or `icx`) for building
|
||||
@@ -226,21 +229,22 @@ the C parts, as the ABI imposed by `ifort` is incompatible with MSVC
|
||||
|
||||
A fully-optimized OpenBLAS that can be statically or dynamically linked to your
|
||||
application can currently be built for the 64-bit architecture with the LLVM
|
||||
compiler infrastructure. We're going to use [Miniconda3](https://docs.anaconda.com/miniconda/)
|
||||
compiler infrastructure. We're going to use [Miniforge3] the pre-configured
|
||||
and more versatile alternative to [Miniconda](https://docs.anaconda.com/miniconda/)
|
||||
to grab all of the tools we need, since some of them are in an experimental
|
||||
status. Before you begin, you'll need to have Microsoft Visual Studio 2015 or
|
||||
newer installed.
|
||||
|
||||
1. Install Miniconda3 for 64-bit Windows using `winget install --id Anaconda.Miniconda3`,
|
||||
or easily download from [conda.io](https://docs.conda.io/en/latest/miniconda.html).
|
||||
2. Open the "Anaconda Command Prompt" now available in the Start Menu, or at `%USERPROFILE%\miniconda3\shell\condabin\conda-hook.ps1`.
|
||||
1. Install Miniforge for 64-bit Windows with the latest version of the installer Miniforge3-Windows-x86_64.exe
|
||||
available on [github.com](https://github.com/conda-forge/miniforge/releases/)
|
||||
2. Open the "Miniforge Command Prompt" now available in the Start Menu, or at `%USERPROFILE%\miniforge3\shell\condabin\conda-hook.ps1`.
|
||||
3. In that command prompt window, use `cd` to change to the directory where you want to build OpenBLAS.
|
||||
4. Now install all of the tools we need:
|
||||
```
|
||||
conda update -n base conda
|
||||
conda config --add channels conda-forge
|
||||
conda install -y cmake flang clangdev perl libflang ninja
|
||||
conda install -y cmake flang_win-64 clangdev perl libflang ninja
|
||||
```
|
||||
(if you want to build with OpenMP support, add `llvm-openmp` and `llvm-openmp-fortran`)
|
||||
5. Still in the Anaconda Command Prompt window, activate the 64-bit MSVC environment with `vcvarsall x64`.
|
||||
On Windows 11 with Visual Studio 2022, this would be done by invoking:
|
||||
|
||||
@@ -439,6 +443,10 @@ To then use the built OpenBLAS shared library in Visual Studio:
|
||||
|
||||
### Windows on Arm
|
||||
|
||||
If you want to use a regular x64 Windows build of OpenBLAS with x64 software in the Prism emulator, be sure to use the latest version of Prism, and to check the box
|
||||
to "Disable floating point optimization" in the Emulation settings. (Right-click on the executable to open "Properties", then on the "Compatibility" tab click on
|
||||
"Change emulation settings").
|
||||
|
||||
A fully functional native OpenBLAS for WoA that can be built as both a static and dynamic library using LLVM toolchain and Visual Studio 2022. Before starting to build, make sure that you have installed Visual Studio 2022 on your ARM device, including the "Desktop Development with C++" component (that contains the cmake tool).
|
||||
(Note that you can use the free "Visual Studio 2022 Community Edition" for this task. In principle it would be possible to build with VisualStudio alone, but using
|
||||
the LLVM toolchain enables native compilation of the Fortran sources of LAPACK and of all the optimized assembly files, which VisualStudio cannot handle on its own)
|
||||
@@ -708,9 +716,10 @@ fully working OpenBLAS for this platform.
|
||||
|
||||
Go to the directory where you unpacked OpenBLAS,and enter the following commands:
|
||||
```bash
|
||||
CC=/Applications/Xcode_12.4.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang
|
||||
CC="/Applications/Xcode.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang"
|
||||
|
||||
CFLAGS= -O2 -Wno-macro-redefined -isysroot /Applications/Xcode_12.4.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS14.4.sdk -arch arm64 -miphoneos-version-min=10.0
|
||||
SDKROOT="$(xcrun --sdk iphoneos --show-sdk-path)"
|
||||
CFLAGS="-O2 -Wno-macro-redefined -isysroot $SDKROOT -arch arm64 -miphoneos-version-min=10.0"
|
||||
|
||||
make TARGET=ARMV8 DYNAMIC_ARCH=1 NUM_THREADS=32 HOSTCC=clang NOFORTRAN=1
|
||||
```
|
||||
|
||||
@@ -30,7 +30,7 @@ OpenBLAS checks the following environment variables on startup:
|
||||
cache where it is not reported correctly (in virtual environments)
|
||||
|
||||
|
||||
Deprecated variables still recognized for compatibilty:
|
||||
Deprecated variables still recognized for compatibility:
|
||||
|
||||
* `GOTO_NUM_THREADS`: equivalent to `OPENBLAS_NUM_THREADS`
|
||||
* `GOTOBLAS_MAIN_FREE`: equivalent to `OPENBLAS_MAIN_FREE`
|
||||
|
||||
+12
-4
@@ -59,13 +59,21 @@
|
||||
#define GEMM_Q 128
|
||||
#endif
|
||||
|
||||
#ifdef GEMM_DIVIDE_RATE
|
||||
#ifdef DYNAMIC_ARCH
|
||||
#define DIVIDE_LIMIT gotoblas->divide_limit
|
||||
#define DIVIDE_RATE gotoblas->divide_rate
|
||||
#else
|
||||
#define DIVIDE_LIMIT GEMM_DIVIDE_LIMIT
|
||||
#define DIVIDE_RATE GEMM_DIVIDE_RATE
|
||||
#endif
|
||||
|
||||
#ifdef GEMM_DIVIDE_LIMIT
|
||||
#define DIVIDE_LIMIT GEMM_DIVIDE_LIMIT
|
||||
#endif
|
||||
//#ifdef GEMM_DIVIDE_RATE
|
||||
//#define DIVIDE_RATE GEMM_DIVIDE_RATE
|
||||
//#endif
|
||||
|
||||
//#ifdef GEMM_DIVIDE_LIMIT
|
||||
//#define DIVIDE_LIMIT GEMM_DIVIDE_LIMIT
|
||||
//#endif
|
||||
|
||||
#ifdef THREADED_LEVEL3
|
||||
#include "level3_thread.c"
|
||||
|
||||
@@ -41,6 +41,7 @@
|
||||
#define CACHE_LINE_SIZE 8
|
||||
#endif
|
||||
|
||||
#define DIVIDE_RATE_MAX 2
|
||||
#ifndef DIVIDE_RATE
|
||||
#define DIVIDE_RATE 2
|
||||
#endif
|
||||
@@ -93,7 +94,7 @@ typedef struct {
|
||||
#else
|
||||
volatile
|
||||
#endif
|
||||
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE];
|
||||
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE_MAX];
|
||||
} job_t;
|
||||
|
||||
|
||||
@@ -294,7 +295,7 @@ static int inner_thread(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n,
|
||||
FLOAT *a, *b, *c;
|
||||
job_t *job = (job_t *)args -> common;
|
||||
BLASLONG xxx, bufferside;
|
||||
FLOAT *buffer[DIVIDE_RATE];
|
||||
FLOAT *buffer[DIVIDE_RATE_MAX];
|
||||
|
||||
BLASLONG ls, min_l, jjs, min_jj;
|
||||
BLASLONG is, min_i, div_n;
|
||||
|
||||
@@ -41,6 +41,8 @@
|
||||
#define CACHE_LINE_SIZE 8
|
||||
#endif
|
||||
|
||||
#define DIVIDE_RATE_MAX 2
|
||||
|
||||
#ifndef DIVIDE_RATE
|
||||
#define DIVIDE_RATE 2
|
||||
#endif
|
||||
@@ -69,7 +71,7 @@ _Atomic
|
||||
#else
|
||||
volatile
|
||||
#endif
|
||||
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE];
|
||||
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE_MAX];
|
||||
} job_t;
|
||||
|
||||
|
||||
@@ -133,7 +135,7 @@ _Atomic
|
||||
|
||||
static int inner_thread(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLOAT *sb, BLASLONG mypos){
|
||||
|
||||
FLOAT *buffer[DIVIDE_RATE];
|
||||
FLOAT *buffer[DIVIDE_RATE_MAX];
|
||||
|
||||
BLASLONG k, lda, ldc;
|
||||
BLASLONG m_from, m_to, n_from, n_to;
|
||||
@@ -504,6 +506,33 @@ static int inner_thread(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n,
|
||||
|
||||
int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLOAT *sb, BLASLONG mypos){
|
||||
|
||||
#ifdef USE_OPENMP
|
||||
static omp_lock_t level3_lock, critical_section_lock;
|
||||
static volatile BLASULONG init_lock = 0, omp_lock_initialized = 0,
|
||||
parallel_section_left = MAX_PARALLEL_NUMBER;
|
||||
|
||||
// Lock initialization; Todo : Maybe this part can be moved to blas_init() in blas_server_omp.c
|
||||
while(omp_lock_initialized == 0)
|
||||
{
|
||||
blas_lock(&init_lock);
|
||||
{
|
||||
if(omp_lock_initialized == 0)
|
||||
{
|
||||
omp_init_lock(&level3_lock);
|
||||
omp_init_lock(&critical_section_lock);
|
||||
omp_lock_initialized = 1;
|
||||
WMB;
|
||||
}
|
||||
blas_unlock(&init_lock);
|
||||
}
|
||||
}
|
||||
#elif defined(OS_WINDOWS)
|
||||
CRITICAL_SECTION level3_lock;
|
||||
InitializeCriticalSection((PCRITICAL_SECTION)&level3_lock);
|
||||
#else
|
||||
static pthread_mutex_t level3_lock = PTHREAD_MUTEX_INITIALIZER;
|
||||
#endif
|
||||
|
||||
blas_arg_t newarg;
|
||||
|
||||
#ifndef USE_ALLOC_HEAP
|
||||
@@ -560,6 +589,30 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef USE_OPENMP
|
||||
omp_set_lock(&level3_lock);
|
||||
omp_set_lock(&critical_section_lock);
|
||||
|
||||
parallel_section_left--;
|
||||
|
||||
/*
|
||||
How OpenMP locks works with NUM_PARALLEL
|
||||
1) parallel_section_left = Number of available concurrent executions of OpenBLAS - Number of currently executing OpenBLAS executions
|
||||
2) level3_lock is acting like a master lock or barrier which stops OpenBLAS calls when all the parallel_section are currently busy executing other OpenBLAS calls
|
||||
3) critical_section_lock is used for updating variables shared between threads executing OpenBLAS calls concurrently and for unlocking of master lock whenever required
|
||||
4) Unlock master lock only when we have not already exhausted all the parallel_sections and allow another thread with a OpenBLAS call to enter
|
||||
*/
|
||||
if(parallel_section_left != 0)
|
||||
omp_unset_lock(&level3_lock);
|
||||
|
||||
omp_unset_lock(&critical_section_lock);
|
||||
|
||||
#elif defined(OS_WINDOWS)
|
||||
EnterCriticalSection((PCRITICAL_SECTION)&level3_lock);
|
||||
#else
|
||||
pthread_mutex_lock(&level3_lock);
|
||||
#endif
|
||||
|
||||
newarg.m = args -> m;
|
||||
newarg.n = args -> n;
|
||||
newarg.k = args -> k;
|
||||
@@ -706,5 +759,25 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
|
||||
free(job);
|
||||
#endif
|
||||
|
||||
#ifdef USE_OPENMP
|
||||
omp_set_lock(&critical_section_lock);
|
||||
parallel_section_left++;
|
||||
|
||||
/*
|
||||
Unlock master lock only when all the parallel_sections are already exhausted and one of the thread has completed its OpenBLAS call
|
||||
otherwise just increment the parallel_section_left
|
||||
The master lock is only locked when we have exhausted all the parallel_sections, So only unlock it then and otherwise just increment the count
|
||||
*/
|
||||
if(parallel_section_left == 1)
|
||||
omp_unset_lock(&level3_lock);
|
||||
|
||||
omp_unset_lock(&critical_section_lock);
|
||||
|
||||
#elif defined(OS_WINDOWS)
|
||||
LeaveCriticalSection((PCRITICAL_SECTION)&level3_lock);
|
||||
#else
|
||||
pthread_mutex_unlock(&level3_lock);
|
||||
#endif
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -41,12 +41,17 @@
|
||||
#define CACHE_LINE_SIZE 8
|
||||
#endif
|
||||
|
||||
#define DIVIDE_RATE_MAX 2
|
||||
|
||||
#ifndef DIVIDE_RATE
|
||||
#define DIVIDE_RATE 2
|
||||
#endif
|
||||
|
||||
#ifndef GEMM_PREFERED_SIZE
|
||||
#define GEMM_PREFERED_SIZE 1
|
||||
#ifdef DYNAMIC_ARCH
|
||||
#define GEMM_PREFERRED_SIZE gotoblas->preferred_size
|
||||
#endif
|
||||
#ifndef GEMM_PREFERRED_SIZE
|
||||
#define GEMM_PREFERRED_SIZE 1
|
||||
#endif
|
||||
|
||||
//The array of job_t may overflow the stack.
|
||||
@@ -93,7 +98,7 @@
|
||||
|
||||
typedef struct {
|
||||
volatile
|
||||
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE];
|
||||
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE_MAX];
|
||||
} job_t;
|
||||
|
||||
|
||||
@@ -234,7 +239,7 @@ typedef struct {
|
||||
|
||||
static int inner_thread(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, IFLOAT *sa, IFLOAT *sb, BLASLONG mypos){
|
||||
|
||||
IFLOAT *buffer[DIVIDE_RATE];
|
||||
IFLOAT *buffer[DIVIDE_RATE_MAX];
|
||||
|
||||
BLASLONG k, lda, ldb, ldc;
|
||||
BLASLONG m_from, m_to, n_from, n_to;
|
||||
@@ -707,7 +712,7 @@ static int gemm_driver(blas_arg_t *args, BLASLONG *range_m, BLASLONG
|
||||
while (m > 0){
|
||||
width = blas_quickdivide(m + nthreads_m - num_parts - 1, nthreads_m - num_parts);
|
||||
|
||||
width = round_up(m, width, GEMM_PREFERED_SIZE);
|
||||
width = round_up(m, width, GEMM_PREFERRED_SIZE);
|
||||
|
||||
m -= width;
|
||||
|
||||
@@ -758,7 +763,7 @@ static int gemm_driver(blas_arg_t *args, BLASLONG *range_m, BLASLONG
|
||||
if (width < switch_ratio) {
|
||||
width = switch_ratio;
|
||||
}
|
||||
width = round_up(width_n, width, GEMM_PREFERED_SIZE);
|
||||
width = round_up(width_n, width, GEMM_PREFERRED_SIZE);
|
||||
|
||||
width_n -= width;
|
||||
if (width_n < 0) {
|
||||
|
||||
@@ -27,7 +27,6 @@ if (USE_THREAD)
|
||||
${BLAS_SERVER}
|
||||
divtable.c # TODO: Makefile has -UDOUBLE
|
||||
blas_l1_thread.c
|
||||
blas_server_callback.c
|
||||
)
|
||||
|
||||
if (NOT NO_AFFINITY)
|
||||
@@ -42,6 +41,7 @@ set(COMMON_SOURCES
|
||||
openblas_env.c
|
||||
openblas_get_num_procs.c
|
||||
openblas_get_num_threads.c
|
||||
blas_server_callback.c
|
||||
)
|
||||
|
||||
# these need to have NAME/CNAME set, so use GenerateNamedObjects, but don't use standard name mangling
|
||||
|
||||
@@ -1,12 +1,12 @@
|
||||
TOPDIR = ../..
|
||||
include ../../Makefile.system
|
||||
|
||||
COMMONOBJS = memory.$(SUFFIX) xerbla.$(SUFFIX) c_abs.$(SUFFIX) z_abs.$(SUFFIX) openblas_set_num_threads.$(SUFFIX) openblas_get_num_threads.$(SUFFIX) openblas_get_num_procs.$(SUFFIX) openblas_get_config.$(SUFFIX) openblas_get_parallel.$(SUFFIX) openblas_error_handle.$(SUFFIX) openblas_env.$(SUFFIX)
|
||||
COMMONOBJS = memory.$(SUFFIX) xerbla.$(SUFFIX) c_abs.$(SUFFIX) z_abs.$(SUFFIX) openblas_set_num_threads.$(SUFFIX) openblas_get_num_threads.$(SUFFIX) openblas_get_num_procs.$(SUFFIX) openblas_get_config.$(SUFFIX) openblas_get_parallel.$(SUFFIX) openblas_error_handle.$(SUFFIX) openblas_env.$(SUFFIX) blas_server_callback.$(SUFFIX)
|
||||
|
||||
#COMMONOBJS += slamch.$(SUFFIX) slamc3.$(SUFFIX) dlamch.$(SUFFIX) dlamc3.$(SUFFIX)
|
||||
|
||||
ifdef SMP
|
||||
COMMONOBJS += blas_server.$(SUFFIX) divtable.$(SUFFIX) blasL1thread.$(SUFFIX) blas_server_callback.$(SUFFIX)
|
||||
COMMONOBJS += blas_server.$(SUFFIX) divtable.$(SUFFIX) blasL1thread.$(SUFFIX)
|
||||
ifneq ($(NO_AFFINITY), 1)
|
||||
COMMONOBJS += init.$(SUFFIX)
|
||||
endif
|
||||
|
||||
@@ -38,6 +38,7 @@
|
||||
/*********************************************************************/
|
||||
|
||||
#include "common.h"
|
||||
#include <strings.h>
|
||||
#if (defined OS_LINUX || defined OS_ANDROID)
|
||||
#include <asm/hwcap.h>
|
||||
#include <sys/auxv.h>
|
||||
@@ -128,6 +129,18 @@ extern gotoblas_t gotoblas_ARMV9SME;
|
||||
#else
|
||||
#define gotoblas_ARMV9SME gotoblas_ARMV8
|
||||
#endif
|
||||
#ifdef DYN_VORTEX
|
||||
extern gotoblas_t gotoblas_VORTEX;
|
||||
#elif defined(DYN_NEOVERSEN1)
|
||||
#define gotoblas_VORTEX gotoblas_NEOVERSEN1
|
||||
#else
|
||||
#define gotoblas_VORTEX gotoblas_ARMV8
|
||||
#endif
|
||||
#ifdef DYN_VORTEXM4
|
||||
extern gotoblas_t gotoblas_VORTEXM4;
|
||||
#else
|
||||
#define gotoblas_VORTEXM4 gotoblas_ARMV8
|
||||
#endif
|
||||
#ifdef DYN_CORTEXA55
|
||||
extern gotoblas_t gotoblas_CORTEXA55;
|
||||
#else
|
||||
@@ -138,7 +151,7 @@ extern gotoblas_t gotoblas_A64FX;
|
||||
#else
|
||||
#define gotoblas_A64FX gotoblas_ARMV8
|
||||
#endif
|
||||
#else
|
||||
#else //not a user-specified dynamic_list
|
||||
extern gotoblas_t gotoblas_CORTEXA53;
|
||||
#define gotoblas_CORTEXA55 gotoblas_CORTEXA53
|
||||
extern gotoblas_t gotoblas_CORTEXA57;
|
||||
@@ -150,22 +163,32 @@ extern gotoblas_t gotoblas_THUNDERX2T99;
|
||||
extern gotoblas_t gotoblas_TSV110;
|
||||
extern gotoblas_t gotoblas_EMAG8180;
|
||||
extern gotoblas_t gotoblas_NEOVERSEN1;
|
||||
#define gotoblas_VORTEX gotoblas_NEOVERSEN1
|
||||
#ifndef NO_SVE
|
||||
extern gotoblas_t gotoblas_NEOVERSEV1;
|
||||
extern gotoblas_t gotoblas_NEOVERSEN2;
|
||||
extern gotoblas_t gotoblas_ARMV8SVE;
|
||||
extern gotoblas_t gotoblas_A64FX;
|
||||
#ifndef NO_SME
|
||||
extern gotoblas_t gotoblas_ARMV9SME;
|
||||
#else
|
||||
#define gotoblas_ARMV9SME gotoblas_ARMV8SVE
|
||||
#endif
|
||||
#else
|
||||
#define gotoblas_NEOVERSEV1 gotoblas_ARMV8
|
||||
#define gotoblas_NEOVERSEN2 gotoblas_ARMV8
|
||||
#define gotoblas_ARMV8SVE gotoblas_ARMV8
|
||||
#define gotoblas_A64FX gotoblas_ARMV8
|
||||
#define gotoblas_ARMV9SME gotoblas_ARMV8
|
||||
#endif
|
||||
#ifndef NO_SME
|
||||
extern gotoblas_t gotoblas_ARMV9SME;
|
||||
#if defined (__clang__) && defined(OS_DARWIN)
|
||||
extern gotoblas_t gotoblas_VORTEXM4;
|
||||
#else
|
||||
#define gotoblas_VORTEXM4 gotoblas_NEOVERSEN1
|
||||
#endif
|
||||
#else
|
||||
#ifndef NO_SVE
|
||||
#define gotoblas_ARMV9SME gotoblas_ARMV8SVE
|
||||
#else
|
||||
#define gotoblas_ARMV9SME gotoblas_NEOVERSEN1
|
||||
#endif
|
||||
#define gotoblas_VORTEXM4 gotoblas_NEOVERSEN1
|
||||
#endif
|
||||
|
||||
extern gotoblas_t gotoblas_THUNDERX3T110;
|
||||
@@ -176,7 +199,7 @@ extern void openblas_warning(int verbose, const char * msg);
|
||||
#define FALLBACK_VERBOSE 1
|
||||
#define NEOVERSEN1_FALLBACK "OpenBLAS : Your OS does not support SVE instructions. OpenBLAS is using Neoverse N1 kernels as a fallback, which may give poorer performance.\n"
|
||||
|
||||
#define NUM_CORETYPES 19
|
||||
#define NUM_CORETYPES 21
|
||||
|
||||
/*
|
||||
* In case asm/hwcap.h is outdated on the build system, make sure
|
||||
@@ -216,6 +239,8 @@ static char *corename[] = {
|
||||
"armv8sve",
|
||||
"a64fx",
|
||||
"armv9sme",
|
||||
"vortex",
|
||||
"vortexm4",
|
||||
"unknown"
|
||||
};
|
||||
|
||||
@@ -239,6 +264,8 @@ char *gotoblas_corename(void) {
|
||||
if (gotoblas == &gotoblas_ARMV8SVE) return corename[16];
|
||||
if (gotoblas == &gotoblas_A64FX) return corename[17];
|
||||
if (gotoblas == &gotoblas_ARMV9SME) return corename[18];
|
||||
if (gotoblas == &gotoblas_VORTEX) return corename[19];
|
||||
if (gotoblas == &gotoblas_VORTEXM4) return corename[20];
|
||||
return corename[NUM_CORETYPES];
|
||||
}
|
||||
|
||||
@@ -277,6 +304,8 @@ static gotoblas_t *force_coretype(char *coretype) {
|
||||
case 16: return (&gotoblas_ARMV8SVE);
|
||||
case 17: return (&gotoblas_A64FX);
|
||||
case 18: return (&gotoblas_ARMV9SME);
|
||||
case 19: return (&gotoblas_VORTEX);
|
||||
case 20: return (&gotoblas_VORTEXM4);
|
||||
}
|
||||
snprintf(message, 128, "Core not found: %s\n", coretype);
|
||||
openblas_warning(1, message);
|
||||
@@ -288,12 +317,12 @@ static gotoblas_t *get_coretype(void) {
|
||||
char coremsg[128];
|
||||
|
||||
#if defined (OS_DARWIN)
|
||||
//future #if !defined(NO_SME)
|
||||
// if (support_sme1()) {
|
||||
// return &gotoblas_ARMV9SME;
|
||||
// }
|
||||
// #endif
|
||||
return &gotoblas_NEOVERSEN1;
|
||||
#if !defined(NO_SME)
|
||||
if (support_sme1()) {
|
||||
return &gotoblas_VORTEXM4;
|
||||
}
|
||||
#endif
|
||||
return &gotoblas_VORTEX;
|
||||
#endif
|
||||
|
||||
#if (!defined OS_LINUX && !defined OS_ANDROID)
|
||||
@@ -378,6 +407,8 @@ static gotoblas_t *get_coretype(void) {
|
||||
case 0xd08: // Cortex A72
|
||||
return &gotoblas_CORTEXA72;
|
||||
case 0xd09: // Cortex A73
|
||||
case 0xd0a: // Cortex A75
|
||||
case 0xd0b: // Cortex A76
|
||||
return &gotoblas_CORTEXA73;
|
||||
case 0xd0c: // Neoverse N1
|
||||
return &gotoblas_NEOVERSEN1;
|
||||
@@ -395,6 +426,9 @@ static gotoblas_t *get_coretype(void) {
|
||||
}else
|
||||
return &gotoblas_NEOVERSEV1;
|
||||
case 0xd4f:
|
||||
case 0xd83:
|
||||
case 0xd85:
|
||||
case 0xd87:
|
||||
if (!(getauxval(AT_HWCAP) & HWCAP_SVE)) {
|
||||
openblas_warning(FALLBACK_VERBOSE, NEOVERSEN1_FALLBACK);
|
||||
return &gotoblas_NEOVERSEN1;
|
||||
@@ -463,8 +497,8 @@ static gotoblas_t *get_coretype(void) {
|
||||
}
|
||||
break;
|
||||
case 0x61: // Apple
|
||||
//future if (support_sme1()) return &gotoblas_ARMV9SME;
|
||||
return &gotoblas_NEOVERSEN1;
|
||||
if (support_sme1()) return &gotoblas_VORTEXM4;
|
||||
return &gotoblas_VORTEX;
|
||||
break;
|
||||
default:
|
||||
snprintf(coremsg, 128, "Unknown CPU model - implementer %x part %x\n",implementer,part);
|
||||
|
||||
@@ -99,7 +99,7 @@ struct riscv_hwprobe {
|
||||
#define RISCV_HWPROBE_IMA_V (1 << 2)
|
||||
#define RISCV_HWPROBE_EXT_ZFH (1 << 27)
|
||||
#define RISCV_HWPROBE_EXT_ZVFH (1 << 30)
|
||||
#define RISCV_HWPROBE_EXT_ZVFBFWMA (1 << 54)
|
||||
#define RISCV_HWPROBE_EXT_ZVFBFWMA (1ULL << 54)
|
||||
|
||||
#ifndef NR_riscv_hwprobe
|
||||
#ifndef NR_arch_specific_syscall
|
||||
|
||||
+20
-7
@@ -1317,7 +1317,11 @@ UNLOCK_COMMAND(&alloc_lock);
|
||||
error:
|
||||
printf("OpenBLAS : Program will terminate because you tried to allocate too many TLS memory regions.\n");
|
||||
printf("This library was built to support a maximum of %d threads - either rebuild OpenBLAS\n", NUM_BUFFERS);
|
||||
printf("with a larger NUM_THREADS value or set the environment variable OPENBLAS_NUM_THREADS to\n");
|
||||
#ifdef USE_OPENMP
|
||||
printf("with a larger NUM_THREADS value or set the environment variable OMP_NUM_THREADS to\n");
|
||||
#else
|
||||
printf("with a larger NUM_THREADS value or set the environment variable OPENBLAS_NUM_THREADS to\n");
|
||||
#endif
|
||||
printf("a sufficiently small number. This error typically occurs when the software that relies on\n");
|
||||
printf("OpenBLAS calls BLAS functions from many threads in parallel, or when your computer has more\n");
|
||||
printf("cpu cores than what OpenBLAS was configured to handle.\n");
|
||||
@@ -1601,7 +1605,7 @@ void DESTRUCTOR gotoblas_quit(void) {
|
||||
}
|
||||
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
BOOL APIENTRY DllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved)
|
||||
BOOL APIENTRY OpenBLASDllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved)
|
||||
{
|
||||
switch (ul_reason_for_call)
|
||||
{
|
||||
@@ -1650,10 +1654,10 @@ static int on_process_term(void)
|
||||
#endif
|
||||
|
||||
#ifdef _WIN64
|
||||
static const PIMAGE_TLS_CALLBACK dll_callback(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = DllMain;
|
||||
static const PIMAGE_TLS_CALLBACK dll_callback(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = OpenBLASDllMain;
|
||||
#pragma const_seg()
|
||||
#else
|
||||
static void (APIENTRY *dll_callback)(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = DllMain;
|
||||
static void (APIENTRY *dll_callback)(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = OpenBLASDllMain;
|
||||
#pragma data_seg()
|
||||
#endif
|
||||
|
||||
@@ -3039,8 +3043,13 @@ void *blas_memory_alloc(int procpos){
|
||||
#endif
|
||||
if (memory_overflowed) goto terminate;
|
||||
fprintf(stderr,"OpenBLAS warning: precompiled NUM_THREADS exceeded, adding auxiliary array for thread metadata.\n");
|
||||
fprintf(stderr,"Note that your application may still crash, if it is calling OpenBLAS from multiple threads in parallel\n");
|
||||
fprintf(stderr,"To avoid this warning, please rebuild your copy of OpenBLAS with a larger NUM_THREADS setting\n");
|
||||
#ifdef USE_OPENMP
|
||||
fprintf(stderr,"or set the environment variable OMP_NUM_THREADS to %d or lower\n", MAX_CPU_NUMBER);
|
||||
#else
|
||||
fprintf(stderr,"or set the environment variable OPENBLAS_NUM_THREADS to %d or lower\n", MAX_CPU_NUMBER);
|
||||
#endif
|
||||
memory_overflowed=1;
|
||||
MB;
|
||||
new_release_info = (struct release_t*) malloc(NEW_BUFFERS * sizeof(struct release_t));
|
||||
@@ -3142,7 +3151,11 @@ terminate:
|
||||
#endif
|
||||
printf("OpenBLAS : Program is Terminated. Because you tried to allocate too many memory regions.\n");
|
||||
printf("This library was built to support a maximum of %d threads - either rebuild OpenBLAS\n", NUM_BUFFERS);
|
||||
printf("with a larger NUM_THREADS value or set the environment variable OPENBLAS_NUM_THREADS to\n");
|
||||
#ifdef USE_OPENMP
|
||||
printf("with a larger NUM_THREADS value or set the environment variable OMP_NUM_THREADS to\n");
|
||||
#else
|
||||
printf("with a larger NUM_THREADS value or set the environment variable OPENBLAS_NUM_THREADS to\n");
|
||||
#endif
|
||||
printf("a sufficiently small number. This error typically occurs when the software that relies on\n");
|
||||
printf("OpenBLAS calls BLAS functions from many threads in parallel, or when your computer has more\n");
|
||||
printf("cpu cores than what OpenBLAS was configured to handle.\n");
|
||||
@@ -3473,7 +3486,7 @@ void DESTRUCTOR gotoblas_quit(void) {
|
||||
}
|
||||
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
BOOL APIENTRY DllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved)
|
||||
BOOL APIENTRY OpenBLASDllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved)
|
||||
{
|
||||
switch (ul_reason_for_call)
|
||||
{
|
||||
@@ -3517,7 +3530,7 @@ static int on_process_term(void)
|
||||
#else
|
||||
#pragma data_seg(".CRT$XLB")
|
||||
#endif
|
||||
static void (APIENTRY *dll_callback)(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = DllMain;
|
||||
static void (APIENTRY *dll_callback)(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = OpenBLASDllMain;
|
||||
#ifdef _WIN64
|
||||
#pragma const_seg()
|
||||
#else
|
||||
|
||||
@@ -162,11 +162,15 @@ ifeq ($(F_COMPILER), INTEL)
|
||||
else
|
||||
ifeq ($(F_COMPILER), FLANG)
|
||||
$(FC) $(FFLAGS) $(LDFLAGS) -fno-fortran-main -Mnomain -all_load -headerpad_max_install_names -install_name "$(CURDIR)/../$(INTERNALNAME)" -dynamiclib -o ../$(LIBDYNNAME) $< -Wl,-exported_symbols_list,osx.def $(FEXTRALIB)
|
||||
else
|
||||
ifeq ($(F_COMPILER), FLANGNEW)
|
||||
$(FC) $(FFLAGS) $(LDFLAGS) -Wl,-all_load -Wl,-headerpad_max_install_names -Wl,-install_name,"$(CURDIR)/../$(INTERNALNAME)" -Wl,-dylib -o ../$(LIBDYNNAME) $< -Wl,-exported_symbols_list,osx.def $(FEXTRALIB)
|
||||
else
|
||||
$(FC) $(FFLAGS) $(LDFLAGS) -all_load -headerpad_max_install_names -install_name "$(CURDIR)/../$(INTERNALNAME)" -dynamiclib -o ../$(LIBDYNNAME) $< -Wl,-exported_symbols_list,osx.def $(FEXTRALIB)
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
dllinit.$(SUFFIX) : dllinit.c
|
||||
$(CC) $(CFLAGS) -c -o $(@F) -s $<
|
||||
|
||||
@@ -181,6 +181,7 @@ misc_no_underscore_objs="
|
||||
goto_set_num_threads
|
||||
openblas_get_config
|
||||
openblas_get_corename
|
||||
openblas_set_threads_callback_function
|
||||
"
|
||||
|
||||
misc_underscore_objs=""
|
||||
|
||||
@@ -177,6 +177,7 @@
|
||||
goto_set_num_threads,
|
||||
openblas_get_config,
|
||||
openblas_get_corename,
|
||||
openblas_set_threads_callback_function,
|
||||
);
|
||||
|
||||
@misc_underscore_objs = (
|
||||
|
||||
@@ -92,7 +92,7 @@ else
|
||||
vendor=FLANG
|
||||
openmp='-fopenmp'
|
||||
;;
|
||||
*GNU*|*GCC*)
|
||||
*GCC*)
|
||||
|
||||
v="${data#*GCC: *\) }"
|
||||
v="${v%%\"*}"
|
||||
@@ -343,13 +343,13 @@ linker_a=""
|
||||
|
||||
if [ -n "$link" ]; then
|
||||
|
||||
link=`echo "$link" | sed 's/\-Y[[:space:]]P\,/\-Y/g'`
|
||||
link=`echo " $link" | sed 's/ \-Y[[:space:]]P\,/ \-Y/g'`
|
||||
|
||||
link=`echo "$link" | sed 's/\-R[[:space:]]*/\-rpath\%/g'`
|
||||
link=`echo "$link" | sed 's/ \-R[[:space:]]*/ \-rpath\%/g'`
|
||||
|
||||
link=`echo "$link" | sed 's/\-rpath[[:space:]]+/\-rpath\%/g'`
|
||||
link=`echo "$link" | sed 's/ \-rpath[[:space:]]+/ \-rpath\%/g'`
|
||||
|
||||
link=`echo "$link" | sed 's/\-rpath-link[[:space:]]+/\-rpath-link\%/g'`
|
||||
link=`echo "$link" | sed 's/ \-rpath-link[[:space:]]+/ \-rpath-link\%/g'`
|
||||
|
||||
flags=`echo "$link" | tr "',\n" " "`
|
||||
# remove leading and trailing quotes from each flag.
|
||||
|
||||
@@ -1232,6 +1232,20 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#else
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_WASM128_GENERIC
|
||||
#define FORCE
|
||||
#define ARCHITECTURE "WASM"
|
||||
#define SUBARCHITECTURE "WASM128_GENERIC"
|
||||
#define SUBDIRNAME "wasm"
|
||||
#define ARCHCONFIG "-DWASM128_GENERIC " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=32 " \
|
||||
"-DL2_SIZE=1048576 -DL2_LINESIZE=32 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=128 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=4 "
|
||||
#define LIBNAME "wasm128"
|
||||
#define CORENAME "WASM128_GENERIC"
|
||||
#else
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_CORTEXA15
|
||||
#define FORCE
|
||||
#define ARCHITECTURE "ARM"
|
||||
@@ -1654,6 +1668,28 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define CORENAME "VORTEX"
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_VORTEXM4
|
||||
#define FORCE
|
||||
#define ARCHITECTURE "ARM64"
|
||||
#define SUBARCHITECTURE "VORTEXM4"
|
||||
#define SUBDIRNAME "arm64"
|
||||
#ifdef __clang__
|
||||
#define ARCHCONFIG "-DVORTEXM4 " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=32 " \
|
||||
"-DHAVE_VFPV4 -DHAVE_VFPV3 -DHAVE_VFP -DHAVE_NEON -DHAVE_SME -DARMV8"
|
||||
#else
|
||||
#define ARCHCONFIG "-DVORTEX " \
|
||||
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
|
||||
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
|
||||
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=32 " \
|
||||
"-DHAVE_VFPV4 -DHAVE_VFPV3 -DHAVE_VFP -DHAVE_NEON -DARMV8"
|
||||
#endif
|
||||
#define LIBNAME "vortexm4"
|
||||
#define CORENAME "VORTEXM4"
|
||||
#endif
|
||||
|
||||
#ifdef FORCE_A64FX
|
||||
#define ARMV8
|
||||
#define FORCE
|
||||
@@ -1927,6 +1963,10 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define OPENBLAS_SUPPORTED
|
||||
#endif
|
||||
|
||||
#ifdef __wasm__
|
||||
#define OPENBLAS_SUPPORTED
|
||||
#endif
|
||||
|
||||
#ifndef OPENBLAS_SUPPORTED
|
||||
#error "This arch/CPU is not supported by OpenBLAS."
|
||||
#endif
|
||||
|
||||
+2
-2
@@ -530,8 +530,8 @@ ifneq ($(NO_LAPACK), 1)
|
||||
SBLASOBJS += $(SLAPACKOBJS)
|
||||
DBLASOBJS += $(DLAPACKOBJS)
|
||||
#QBLASOBJS += $(QLAPACKOBJS)
|
||||
CBLASOBJS += $(CLAPACKOBJS)
|
||||
ZBLASOBJS += $(ZLAPACKOBJS)
|
||||
CBLASOBJS += $(CLAPACKOBJS) slaed3.$(SUFFIX)
|
||||
ZBLASOBJS += $(ZLAPACKOBJS) dlaed3.$(SUFFIX)
|
||||
#XBLASOBJS += $(XLAPACKOBJS)
|
||||
|
||||
endif
|
||||
|
||||
+36
-25
@@ -184,11 +184,11 @@ static int init_amxtile_permission() {
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef SMP
|
||||
#ifdef DYNAMIC_ARCH
|
||||
extern char* gotoblas_corename(void);
|
||||
#endif
|
||||
|
||||
#ifdef SMP
|
||||
#if defined(DYNAMIC_ARCH) || defined(NEOVERSEV1)
|
||||
static inline int get_gemm_optimal_nthreads_neoversev1(double MNK, int ncpu) {
|
||||
return
|
||||
@@ -266,6 +266,7 @@ void NAME(char *TRANSA, char *TRANSB,
|
||||
|
||||
int transa, transb, nrowa, nrowb;
|
||||
blasint info;
|
||||
int order = -1;
|
||||
|
||||
char transA, transB;
|
||||
IFLOAT *buffer;
|
||||
@@ -424,30 +425,6 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_TRANSPOSE TransA, enum CBLAS_TRANS
|
||||
|
||||
PRINT_DEBUG_CNAME;
|
||||
|
||||
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
|
||||
#if defined(ARCH_x86) && (defined(USE_SGEMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
|
||||
#if defined(DYNAMIC_ARCH)
|
||||
if (support_avx512() )
|
||||
#endif
|
||||
if (beta == 0 && alpha == 1.0 && order == CblasRowMajor && TransA == CblasNoTrans && TransB == CblasNoTrans && SGEMM_DIRECT_PERFORMANT(m,n,k)) {
|
||||
SGEMM_DIRECT(m, n, k, a, lda, b, ldb, c, ldc);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
#if defined(ARCH_ARM64) && (defined(USE_SGEMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
|
||||
#if defined(DYNAMIC_ARCH)
|
||||
if (support_sme1())
|
||||
#endif
|
||||
if (beta == 0 && alpha == 1.0 && order == CblasRowMajor && TransA == CblasNoTrans && TransB == CblasNoTrans) {
|
||||
SGEMM_DIRECT(m, n, k, a, lda, b, ldb, c, ldc);
|
||||
return;
|
||||
}else if (order == CblasRowMajor && TransA == CblasNoTrans && TransB == CblasNoTrans) {
|
||||
SGEMM_DIRECT_ALPHA_BETA(m, n, k, alpha, a, lda, b, ldb, beta, c, ldc);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifndef COMPLEX
|
||||
args.alpha = (void *)α
|
||||
args.beta = (void *)β
|
||||
@@ -564,6 +541,40 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_TRANSPOSE TransA, enum CBLAS_TRANS
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
if ((args.m == 0) || (args.n == 0)) return;
|
||||
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
|
||||
#if defined(ARCH_x86) && (defined(USE_SGEMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
|
||||
#if defined(DYNAMIC_ARCH)
|
||||
if (support_avx512() )
|
||||
#endif
|
||||
if (order == CblasRowMajor && beta == 0 && alpha == 1.0 && TransA == CblasNoTrans && TransB == CblasNoTrans && SGEMM_DIRECT_PERFORMANT(m,n,k)) {
|
||||
SGEMM_DIRECT(m, n, k, a, lda, b, ldb, c, ldc);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
#if defined(ARCH_ARM64) && (defined(USE_SGEMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
|
||||
#if defined(DYNAMIC_ARCH)
|
||||
if (strcmp(gotoblas_corename(), "armv9sme") == 0
|
||||
#if defined(__clang__)
|
||||
|| strcmp(gotoblas_corename(), "vortexm4") == 0
|
||||
#endif
|
||||
)
|
||||
// if (support_sme1())
|
||||
#endif
|
||||
if (order == CblasRowMajor && m==lda && n ==ldb && k==ldc && beta == 0 && alpha == 1.0 && TransA == CblasNoTrans && TransB == CblasNoTrans&& SGEMM_DIRECT_PERFORMANT(m,n,k)) {
|
||||
SGEMM_DIRECT(m, n, k, a, lda, b, ldb, c, ldc);
|
||||
return;
|
||||
}
|
||||
else
|
||||
if (order == CblasRowMajor && m==lda && n==ldb && k==ldc && TransA == CblasNoTrans && TransB == CblasNoTrans&& SGEMM_DIRECT_PERFORMANT(m,n,k)) {
|
||||
SGEMM_DIRECT_ALPHA_BETA(m, n, k, alpha, a, lda, b, ldb, beta, c, ldc);
|
||||
return;
|
||||
}
|
||||
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#endif
|
||||
|
||||
#if defined(__linux__) && defined(__x86_64__) && defined(BFLOAT16)
|
||||
|
||||
+9
-5
@@ -1,4 +1,5 @@
|
||||
/*********************************************************************/
|
||||
/* Copyright 2025 The OpenBLAS Project */
|
||||
/* Copyright 2009, 2010 The University of Texas at Austin. */
|
||||
/* All rights reserved. */
|
||||
/* */
|
||||
@@ -81,9 +82,12 @@ static inline int get_gemv_optimal_nthreads_neoversev1(BLASLONG MN, int ncpu) {
|
||||
: (MN < 1050625L) ? MIN(ncpu, 40)
|
||||
: ncpu;
|
||||
#else
|
||||
return (MN < 25600L) ? 1
|
||||
return
|
||||
(MN < 25600L) ? 1
|
||||
: (MN < 63001L) ? MIN(ncpu, 4)
|
||||
: (MN < 459684L) ? MIN(ncpu, 16)
|
||||
: (MN < 202500L) ? MIN(ncpu, 8)
|
||||
: (MN < 806404L) ? MIN(ncpu, 16)
|
||||
: (MN < 1638400L) ? MIN(ncpu, 32)
|
||||
: ncpu;
|
||||
#endif
|
||||
}
|
||||
@@ -93,9 +97,9 @@ static inline int get_gemv_optimal_nthreads_neoversev1(BLASLONG MN, int ncpu) {
|
||||
static inline int get_gemv_optimal_nthreads_neoversev2(BLASLONG MN, int ncpu) {
|
||||
return
|
||||
MN < 24964L ? 1
|
||||
: MN < 65536L ? MIN(ncpu, 8)
|
||||
: MN < 262144L ? MIN(ncpu, 32)
|
||||
: MN < 1638400L ? MIN(ncpu, 64)
|
||||
: MN < 145924L ? MIN(ncpu, 8)
|
||||
: MN < 692224L ? MIN(ncpu, 16)
|
||||
: MN < 1638400L ? MIN(ncpu, 32)
|
||||
: ncpu;
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -99,7 +99,7 @@ int NAME(blasint *N, blasint *NRHS, FLOAT *a, blasint *ldA, blasint *ipiv,
|
||||
|
||||
*Info = 0;
|
||||
|
||||
if (args.m == 0 || args.n == 0) return 0;
|
||||
if (args.m == 0) return 0;
|
||||
|
||||
IDEBUG_START;
|
||||
|
||||
@@ -117,20 +117,20 @@ int NAME(blasint *N, blasint *NRHS, FLOAT *a, blasint *ldA, blasint *ipiv,
|
||||
|
||||
#if defined(_WIN64) && defined(_M_ARM64)
|
||||
#ifdef COMPLEX
|
||||
if (args.m * args.n <= 300)
|
||||
if (args.m * args.m <= 300)
|
||||
#else
|
||||
if (args.m * args.n <= 500)
|
||||
if (args.m * args.m <= 500)
|
||||
#endif
|
||||
args.nthreads = 1;
|
||||
else if (args.m * args.n <= 1000)
|
||||
else if (args.m * args.m <= 1000)
|
||||
args.nthreads = 4;
|
||||
else
|
||||
args.nthreads = num_cpu_avail(4);
|
||||
#else
|
||||
#ifndef DOUBLE
|
||||
if (args.m * args.n < 40000)
|
||||
if (args.m * args.m < 40000)
|
||||
#else
|
||||
if (args.m * args.n < 10000)
|
||||
if (args.m * args.m < 10000)
|
||||
#endif
|
||||
args.nthreads = 1;
|
||||
else
|
||||
@@ -143,7 +143,7 @@ int NAME(blasint *N, blasint *NRHS, FLOAT *a, blasint *ldA, blasint *ipiv,
|
||||
args.n = *N;
|
||||
info = GETRF_SINGLE(&args, NULL, NULL, sa, sb, 0);
|
||||
|
||||
if (info == 0){
|
||||
if (info == 0 && *NRHS >0){
|
||||
args.n = *NRHS;
|
||||
GETRS_N_SINGLE(&args, NULL, NULL, sa, sb, 0);
|
||||
}
|
||||
@@ -154,7 +154,7 @@ int NAME(blasint *N, blasint *NRHS, FLOAT *a, blasint *ldA, blasint *ipiv,
|
||||
args.n = *N;
|
||||
info = GETRF_PARALLEL(&args, NULL, NULL, sa, sb, 0);
|
||||
|
||||
if (info == 0){
|
||||
if (info == 0 && *NRHS > 0){
|
||||
args.n = *NRHS;
|
||||
GETRS_N_PARALLEL(&args, NULL, NULL, sa, sb, 0);
|
||||
}
|
||||
|
||||
+1
-1
@@ -73,7 +73,7 @@ void CNAME(blasint n, FLOAT alpha, FLOAT *x, blasint incx){
|
||||
float alpha_float;
|
||||
SBF16TOS_K(1, &alpha, 1, &alpha_float, 1);
|
||||
#else
|
||||
float alpha_float = alpha;
|
||||
FLOAT alpha_float = alpha;
|
||||
#endif
|
||||
|
||||
if (alpha_float == ONE) return;
|
||||
|
||||
+8
-1
@@ -97,6 +97,9 @@
|
||||
#define GEMM_MULTITHREAD_THRESHOLD 4
|
||||
#endif
|
||||
|
||||
#ifdef DYNAMIC_ARCH
|
||||
extern char* gotoblas_corename(void);
|
||||
#endif
|
||||
|
||||
#ifdef SMP
|
||||
#ifndef COMPLEX
|
||||
@@ -374,7 +377,11 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_SIDE Side, enum CBLAS_UPLO Uplo,
|
||||
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
|
||||
#if defined(ARCH_ARM64) && (defined(USE_SSYMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
|
||||
#if defined(DYNAMIC_ARCH)
|
||||
if (support_sme1())
|
||||
if (strcmp(gotoblas_corename(), "armv9sme") == 0
|
||||
#if defined(__clang__)
|
||||
|| strcmp(gotoblas_corename(), "vortexm4") == 0
|
||||
#endif
|
||||
)
|
||||
#endif
|
||||
if (args.m == 0 || args.n == 0) return;
|
||||
if (order == CblasRowMajor && m == lda && n == ldb && n == ldc)
|
||||
|
||||
+38
-1
@@ -345,9 +345,46 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_UPLO Uplo, enum CBLAS_TRANSPOSE Tr
|
||||
return;
|
||||
}
|
||||
|
||||
if (args.n == 0) return;
|
||||
|
||||
#ifdef DYNAMIC_ARCH
|
||||
extern char* gotoblas_corename(void);
|
||||
#endif
|
||||
|
||||
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
|
||||
#if defined(ARCH_ARM64) && (defined(USE_SSYR2K_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
|
||||
#if defined(DYNAMIC_ARCH)
|
||||
if (strcmp(gotoblas_corename(), "armv9sme") == 0
|
||||
#if defined(__clang__)
|
||||
|| strcmp(gotoblas_corename(), "vortexm4") == 0
|
||||
#endif
|
||||
)
|
||||
#endif
|
||||
if (order == CblasRowMajor && n == ldc) {
|
||||
if (Trans == CblasNoTrans && k == lda && k == ldb) {
|
||||
if (Uplo == CblasUpper) {
|
||||
SSYR2K_DIRECT_ALPHA_BETA_UN(n, k, alpha, a, lda, b, ldb, beta, c, ldc);
|
||||
return;
|
||||
}else if (Uplo == CblasLower) {
|
||||
SSYR2K_DIRECT_ALPHA_BETA_LN(n, k, alpha, a, lda, b, ldb, beta, c, ldc);
|
||||
return;
|
||||
}
|
||||
}
|
||||
else if (Trans == CblasTrans && n == lda && n ==ldb) {
|
||||
if (Uplo == CblasUpper) {
|
||||
SSYR2K_DIRECT_ALPHA_BETA_UT(n, k, alpha, a, lda, b, ldb, beta, c, ldc);
|
||||
return;
|
||||
}else if (Uplo == CblasLower) {
|
||||
SSYR2K_DIRECT_ALPHA_BETA_LT(n, k, alpha, a, lda, b, ldb, beta, c, ldc);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#endif
|
||||
|
||||
if (args.n == 0) return;
|
||||
|
||||
IDEBUG_START;
|
||||
|
||||
|
||||
+12
-3
@@ -338,12 +338,22 @@ double NNK;
|
||||
BLASFUNC(xerbla)(ERROR_NAME, &info, sizeof(ERROR_NAME));
|
||||
return;
|
||||
}
|
||||
|
||||
if (args.n == 0) return;
|
||||
|
||||
#ifdef DYNAMIC_ARCH
|
||||
extern char* gotoblas_corename(void);
|
||||
#endif
|
||||
|
||||
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
|
||||
#if defined(ARCH_ARM64) && (defined(USE_SSYRK_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
|
||||
#if defined(DYNAMIC_ARCH)
|
||||
if (support_sme1())
|
||||
if (strcmp(gotoblas_corename(), "armv9sme") == 0
|
||||
#if defined(__clang__)
|
||||
|| strcmp(gotoblas_corename(), "vortexm4") == 0
|
||||
#endif
|
||||
)
|
||||
#endif
|
||||
if (args.n == 0) return;
|
||||
if (order == CblasRowMajor && n == ldc) {
|
||||
if (Trans == CblasNoTrans && k == lda) {
|
||||
(Uplo == CblasUpper ? SSYRK_DIRECT_ALPHA_BETA_UN : SSYRK_DIRECT_ALPHA_BETA_LN)(n, k, alpha, a, lda, beta, c, ldc);
|
||||
@@ -358,7 +368,6 @@ double NNK;
|
||||
|
||||
#endif
|
||||
|
||||
if (args.n == 0) return;
|
||||
|
||||
IDEBUG_START;
|
||||
|
||||
|
||||
+9
-1
@@ -87,6 +87,10 @@
|
||||
#define SMP_FACTOR 128
|
||||
#endif
|
||||
|
||||
#ifdef DYNAMIC_ARCH
|
||||
extern char* gotoblas_corename(void);
|
||||
#endif
|
||||
|
||||
static int (*trsm[])(blas_arg_t *, BLASLONG *, BLASLONG *, FLOAT *, FLOAT *, BLASLONG) = {
|
||||
#ifndef TRMM
|
||||
TRSM_LNUU, TRSM_LNUN, TRSM_LNLU, TRSM_LNLN,
|
||||
@@ -358,7 +362,11 @@ void CNAME(enum CBLAS_ORDER order,
|
||||
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
|
||||
#if defined(ARCH_ARM64) && (defined(USE_STRMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
|
||||
#if defined(DYNAMIC_ARCH)
|
||||
if (support_sme1())
|
||||
if (strcmp(gotoblas_corename(), "armv9sme") == 0
|
||||
#if defined(__clang__)
|
||||
|| strcmp(gotoblas_corename(), "vortexm4") == 0
|
||||
#endif
|
||||
)
|
||||
#endif
|
||||
if (args.m == 0 || args.n == 0) return;
|
||||
if (order == CblasRowMajor && Diag == CblasNonUnit && Side == CblasLeft && m == lda && n == ldb) {
|
||||
|
||||
+21
-5
@@ -48,7 +48,7 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
endif ()
|
||||
|
||||
if (${ADD_COMMONOBJS})
|
||||
if (X86)
|
||||
if (X86 AND NOT EMSCRIPTEN)
|
||||
if (NOT "${CMAKE_C_COMPILER_ID}" STREQUAL "MSVC")
|
||||
GenerateNamedObjects("${KERNELDIR}/cpuid.S" "" "" false "" "" true)
|
||||
else()
|
||||
@@ -235,7 +235,7 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
# Makefile.L3
|
||||
set(USE_TRMM false)
|
||||
string(TOUPPER ${TARGET_CORE} UC_TARGET_CORE)
|
||||
if (ARM OR ARM64 OR RISCV64 OR (UC_TARGET_CORE MATCHES LONGSOON3B) OR (UC_TARGET_CORE MATCHES GENERIC) OR (UC_TARGET_CORE MATCHES HASWELL) OR (UC_TARGET_CORE MATCHES ZEN) OR (UC_TARGET_CORE MATCHES SKYLAKEX) OR (UC_TARGET_CORE MATCHES COOPERLAKE) OR (UC_TARGET_CORE MATCHES SAPPHIRERAPIDS))
|
||||
if (ARM OR ARM64 OR RISCV64 OR WASM OR (UC_TARGET_CORE MATCHES LONGSOON3B) OR (UC_TARGET_CORE MATCHES GENERIC) OR (UC_TARGET_CORE MATCHES HASWELL) OR (UC_TARGET_CORE MATCHES ZEN) OR (UC_TARGET_CORE MATCHES SKYLAKEX) OR (UC_TARGET_CORE MATCHES COOPERLAKE) OR (UC_TARGET_CORE MATCHES SAPPHIRERAPIDS))
|
||||
set(USE_TRMM true)
|
||||
endif ()
|
||||
if (ZARCH OR (UC_TARGET_CORE MATCHES POWER8) OR (UC_TARGET_CORE MATCHES POWER9) OR (UC_TARGET_CORE MATCHES POWER10))
|
||||
@@ -249,6 +249,10 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
if (ARM64)
|
||||
set(USE_DIRECT_SSYRK true)
|
||||
endif()
|
||||
set(USE_DIRECT_SSYR2K false)
|
||||
if (ARM64)
|
||||
set(USE_DIRECT_SSYR2K true)
|
||||
endif()
|
||||
set(USE_DIRECT_SGEMM false)
|
||||
if (X86_64 OR ARM64)
|
||||
set(USE_DIRECT_SGEMM true)
|
||||
@@ -257,7 +261,7 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
if (ARM64)
|
||||
set(USE_DIRECT_SSYMM true)
|
||||
endif()
|
||||
if (UC_TARGET_CORE MATCHES ARMV9SME)
|
||||
if (UC_TARGET_CORE MATCHES ARMV9SME OR UC_TARGET_CORE MATCHES VORTEXM4)
|
||||
set (HAVE_SME true)
|
||||
endif ()
|
||||
|
||||
@@ -270,14 +274,16 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTKERNEL}" "" "gemm_direct" false "" "" false SINGLE)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTPERFORMANT}" "" "gemm_direct_performant" false "" "" false SINGLE)
|
||||
elseif (ARM64)
|
||||
set (SGEMMDIRECTPERFORMANT sgemm_direct_performant.c)
|
||||
set (SGEMMDIRECTKERNEL sgemm_direct_arm64_sme1.c)
|
||||
set (SGEMMDIRECTKERNEL_ALPHA_BETA sgemm_direct_alpha_beta_arm64_sme1.c)
|
||||
set (SGEMMDIRECTSMEKERNEL sgemm_direct_sme1.S)
|
||||
set (SGEMMDIRECTSMEKERNEL sgemm_direct_sme1_2VLx2VL.S)
|
||||
set (SGEMMDIRECTPREKERNEL sgemm_direct_sme1_preprocess.S)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTPERFORMANT}" "" "gemm_direct_performant" false "" "" false SINGLE)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTKERNEL}" "" "gemm_direct" false "" "" false SINGLE)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTKERNEL_ALPHA_BETA}" "" "gemm_direct_alpha_beta" false "" "" false SINGLE)
|
||||
if (HAVE_SME)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTSMEKERNEL}" "" "gemm_direct_sme1" false "" "" false SINGLE)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTSMEKERNEL}" "" "gemm_direct_sme1_2VLx2VL" false "" "" false SINGLE)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTPREKERNEL}" "" "gemm_direct_sme1_preprocess" false "" "" false SINGLE)
|
||||
endif ()
|
||||
endif ()
|
||||
@@ -311,6 +317,16 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
|
||||
endif ()
|
||||
endif()
|
||||
|
||||
if (USE_DIRECT_SSYR2K)
|
||||
if (ARM64)
|
||||
set (SSYR2KDIRECTKERNEL_ALPHA_BETA ssyr2k_direct_alpha_beta_arm64_sme1.c)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SSYR2KDIRECTKERNEL_ALPHA_BETA}" "" "syr2k_direct_alpha_betaUN" false "" "" false SINGLE)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SSYR2KDIRECTKERNEL_ALPHA_BETA}" "" "syr2k_direct_alpha_betaUT" false "" "" false SINGLE)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SSYR2KDIRECTKERNEL_ALPHA_BETA}" "" "syr2k_direct_alpha_betaLN" false "" "" false SINGLE)
|
||||
GenerateNamedObjects("${KERNELDIR}/${SSYR2KDIRECTKERNEL_ALPHA_BETA}" "" "syr2k_direct_alpha_betaLT" false "" "" false SINGLE)
|
||||
endif ()
|
||||
endif()
|
||||
|
||||
foreach (float_type SINGLE DOUBLE)
|
||||
string(SUBSTRING ${float_type} 0 1 float_char)
|
||||
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMMKERNEL}" "" "gemm_kernel" false "" "" false ${float_type})
|
||||
|
||||
+23
-1
@@ -27,7 +27,29 @@ endif
|
||||
|
||||
ifdef TARGET_CORE
|
||||
ifeq ($(TARGET_CORE), ARMV9SME)
|
||||
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE) -DHAVE_SME -march=armv9-a+sve2+sme
|
||||
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE) -march=armv9-a+sve2+sme
|
||||
ifdef OS_WINDOWS
|
||||
ifeq ($(C_COMPILER), CLANG)
|
||||
override CFLAGS += --aarch64-stack-hazard-size=0
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
ifeq ($(TARGET_CORE), VORTEXM4)
|
||||
ifeq ($(C_COMPILER), GCC)
|
||||
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE) -UHAVE_SME -march=armv8.4-a
|
||||
else
|
||||
ifeq ($(APPLECLANG),1)
|
||||
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE) -march=armv8.4-a+sme
|
||||
else
|
||||
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE) -march=armv8.4-a+sme
|
||||
override LDFLAGS += -lclang_rt_builtins-aarch64
|
||||
endif
|
||||
ifdef OS_WINDOWS
|
||||
ifeq ($(C_COMPILER), CLANG)
|
||||
override CFLAGS += --aarch64-stack-hazard-size=0
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
ifeq ($(TARGET_CORE), SAPPHIRERAPIDS)
|
||||
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE)
|
||||
|
||||
+67
-26
@@ -53,14 +53,27 @@ ifeq ($(ARCH), arm64)
|
||||
USE_TRMM = 1
|
||||
USE_DIRECT_SGEMM = 1
|
||||
USE_DIRECT_SSYMM = 1
|
||||
USE_DIRECT_STRMM = 1
|
||||
USE_DIRECT_SSYRK = 1
|
||||
USE_DIRECT_SSYR2K = 1
|
||||
USE_DIRECT_STRMM = 1
|
||||
ifeq ($(CORE), ARMV9SME)
|
||||
USE_SME = 1
|
||||
endif
|
||||
ifeq ($(CORE), VORTEXM4)
|
||||
ifneq ($(C_COMPILER), GCC)
|
||||
USE_SME = 1
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(ARCH), riscv64)
|
||||
USE_TRMM = 1
|
||||
endif
|
||||
|
||||
ifeq ($(ARCH), wasm)
|
||||
USE_TRMM = 1
|
||||
endif
|
||||
|
||||
ifneq ($(DYNAMIC_ARCH), 1)
|
||||
ifeq ($(TARGET), GENERIC)
|
||||
USE_TRMM = 1
|
||||
@@ -131,11 +144,9 @@ SGEMMDIRECTKERNEL = sgemm_direct_skylakex.c
|
||||
SGEMMDIRECTPERFORMANT = sgemm_direct_performant.c
|
||||
endif
|
||||
ifeq ($(ARCH), arm64)
|
||||
ifeq ($(TARGET_CORE), ARMV9SME)
|
||||
HAVE_SME = 1
|
||||
endif
|
||||
SGEMMDIRECTKERNEL = sgemm_direct_arm64_sme1.c
|
||||
SGEMMDIRECTKERNEL_ALPHA_BETA = sgemm_direct_alpha_beta_arm64_sme1.c
|
||||
SGEMMDIRECTPERFORMANT = sgemm_direct_performant.c
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
@@ -143,9 +154,6 @@ endif
|
||||
ifdef USE_DIRECT_SSYMM
|
||||
ifndef SSYMMDIRECTKERNEL_ALPHA_BETA
|
||||
ifeq ($(ARCH), arm64)
|
||||
ifeq ($(TARGET_CORE), ARMV9SME)
|
||||
HAVE_SME = 1
|
||||
endif
|
||||
SSYMMDIRECTKERNEL_ALPHA_BETA = ssymm_direct_alpha_beta_arm64_sme1.c
|
||||
endif
|
||||
endif
|
||||
@@ -154,9 +162,6 @@ endif
|
||||
ifdef USE_DIRECT_STRMM
|
||||
ifndef STRMMDIRECTKERNEL
|
||||
ifeq ($(ARCH), arm64)
|
||||
ifeq ($(TARGET_CORE), ARMV9SME)
|
||||
HAVE_SME = 1
|
||||
endif
|
||||
STRMMDIRECTKERNEL = strmm_direct_arm64_sme1.c
|
||||
endif
|
||||
endif
|
||||
@@ -165,10 +170,18 @@ endif
|
||||
ifdef USE_DIRECT_SSYRK
|
||||
ifndef SSYRKDIRECTKERNEL_ALPHA_BETA
|
||||
ifeq ($(ARCH), arm64)
|
||||
SSYRKDIRECTKERNEL_ALPHA_BETA = ssyrk_direct_alpha_beta_arm64_sme1.c
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
ifdef USE_DIRECT_SSYR2K
|
||||
ifndef SSYR2KDIRECTKERNEL_ALPHA_BETA
|
||||
ifeq ($(ARCH), arm64)
|
||||
ifeq ($(TARGET_CORE), ARMV9SME)
|
||||
HAVE_SME = 1
|
||||
endif
|
||||
SSYRKDIRECTKERNEL_ALPHA_BETA = ssyrk_direct_alpha_beta_arm64_sme1.c
|
||||
SSYR2KDIRECTKERNEL_ALPHA_BETA = ssyr2k_direct_alpha_beta_arm64_sme1.c
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
@@ -245,11 +258,12 @@ SKERNELOBJS += \
|
||||
endif
|
||||
ifeq ($(ARCH), arm64)
|
||||
SKERNELOBJS += \
|
||||
sgemm_direct_performant$(TSUFFIX).$(SUFFIX) \
|
||||
sgemm_direct$(TSUFFIX).$(SUFFIX) \
|
||||
sgemm_direct_alpha_beta$(TSUFFIX).$(SUFFIX)
|
||||
ifdef HAVE_SME
|
||||
ifdef USE_SME
|
||||
SKERNELOBJS += \
|
||||
sgemm_direct_sme1$(TSUFFIX).$(SUFFIX) \
|
||||
sgemm_direct_sme1_2VLx2VL$(TSUFFIX).$(SUFFIX) \
|
||||
sgemm_direct_sme1_preprocess$(TSUFFIX).$(SUFFIX)
|
||||
endif
|
||||
endif
|
||||
@@ -275,8 +289,18 @@ endif
|
||||
ifdef USE_DIRECT_SSYRK
|
||||
ifeq ($(ARCH), arm64)
|
||||
SKERNELOBJS += \
|
||||
ssyrk_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) ssyrk_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) \
|
||||
ssyrk_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) ssyrk_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX)
|
||||
ssyrk_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) ssyrk_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) \
|
||||
ssyrk_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) ssyrk_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX)
|
||||
endif
|
||||
endif
|
||||
|
||||
ifdef USE_DIRECT_SSYR2K
|
||||
ifeq ($(ARCH), arm64)
|
||||
SKERNELOBJS += \
|
||||
ssyr2k_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) ssyr2k_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) \
|
||||
ssyr2k_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) ssyr2k_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) \
|
||||
ssyr2k_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) ssyr2k_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) \
|
||||
ssyr2k_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX) ssyr2k_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX)
|
||||
endif
|
||||
endif
|
||||
|
||||
@@ -1029,13 +1053,15 @@ $(KDIR)sgemm_direct$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMMDIRECTKERNEL)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
|
||||
endif
|
||||
ifeq ($(ARCH), arm64)
|
||||
$(KDIR)sgemm_direct_performant$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMMDIRECTPERFORMANT)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
|
||||
$(KDIR)sgemm_direct$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMMDIRECTKERNEL)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
|
||||
$(KDIR)sgemm_direct_alpha_beta$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMMDIRECTKERNEL_ALPHA_BETA)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
|
||||
ifdef HAVE_SME
|
||||
$(KDIR)sgemm_direct_sme1$(TSUFFIX).$(SUFFIX) :
|
||||
$(CC) $(CFLAGS) -c $(KERNELDIR)/sgemm_direct_sme1.S -UDOUBLE -UCOMPLEX -o $@
|
||||
ifdef USE_SME
|
||||
$(KDIR)sgemm_direct_sme1_2VLx2VL$(TSUFFIX).$(SUFFIX) :
|
||||
$(CC) $(CFLAGS) -c $(KERNELDIR)/sgemm_direct_sme1_2VLx2VL.S -UDOUBLE -UCOMPLEX -o $@
|
||||
$(KDIR)sgemm_direct_sme1_preprocess$(TSUFFIX).$(SUFFIX) :
|
||||
$(CC) $(CFLAGS) -c $(KERNELDIR)/sgemm_direct_sme1_preprocess.S -UDOUBLE -UCOMPLEX -o $@
|
||||
endif
|
||||
@@ -1051,6 +1077,22 @@ $(KDIR)ssymm_direct_alpha_betaLL$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYMMDIREC
|
||||
endif
|
||||
endif
|
||||
|
||||
ifdef USE_DIRECT_SSYRK
|
||||
ifeq ($(ARCH), arm64)
|
||||
$(KDIR)ssyrk_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DUPPER -UTRANSA $< -o $@
|
||||
|
||||
$(KDIR)ssyrk_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DUPPER -DTRANSA $< -o $@
|
||||
|
||||
$(KDIR)ssyrk_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -UUPPER -UTRANSA $< -o $@
|
||||
|
||||
$(KDIR)ssyrk_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -UUPPER -DTRANSA $< -o $@
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(BUILD_BFLOAT16), 1)
|
||||
$(KDIR)bgemm_kernel$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(BGEMMKERNEL)
|
||||
$(CC) $(CFLAGS) -c -DBFLOAT16 -DBGEMM -UDOUBLE -UCOMPLEX $< -o $@
|
||||
@@ -1177,19 +1219,18 @@ $(KDIR)xgemm_kernel_r$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(XGEMMKERNEL) $(XGEMMD
|
||||
$(KDIR)xgemm_kernel_b$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(XGEMMKERNEL) $(XGEMMDEPEND)
|
||||
$(CC) $(CFLAGS) -c -DXDOUBLE -DCOMPLEX -DCC $< -o $@
|
||||
|
||||
ifdef USE_DIRECT_SSYRK
|
||||
|
||||
ifdef USE_DIRECT_SSYR2K
|
||||
ifeq ($(ARCH), arm64)
|
||||
$(KDIR)ssyrk_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
|
||||
$(KDIR)ssyr2k_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYR2KDIRECTKERNEL_ALPHA_BETA)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DUPPER -UTRANSA $< -o $@
|
||||
|
||||
$(KDIR)ssyrk_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
|
||||
$(KDIR)ssyr2k_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYR2KDIRECTKERNEL_ALPHA_BETA)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DUPPER -DTRANSA $< -o $@
|
||||
|
||||
$(KDIR)ssyrk_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
|
||||
$(KDIR)ssyr2k_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYR2KDIRECTKERNEL_ALPHA_BETA)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -UUPPER -UTRANSA $< -o $@
|
||||
|
||||
$(KDIR)ssyrk_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
|
||||
$(KDIR)ssyr2k_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYR2KDIRECTKERNEL_ALPHA_BETA)
|
||||
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -UUPPER -DTRANSA $< -o $@
|
||||
|
||||
endif
|
||||
endif
|
||||
|
||||
|
||||
+6
-3
@@ -1,5 +1,5 @@
|
||||
/***************************************************************************
|
||||
Copyright (c) 2013, The OpenBLAS Project
|
||||
Copyright (c) 2013-2026, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
@@ -50,8 +50,11 @@ FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x, FLOAT *y, BLASLONG inc_y)
|
||||
|
||||
while(i < n)
|
||||
{
|
||||
|
||||
dot += y[iy] * x[ix] ;
|
||||
#if defined(DSDOT)
|
||||
dot += (double)y[iy] * (double)x[ix] ;
|
||||
#else
|
||||
dot += y[iy] * x[ix];
|
||||
#endif
|
||||
ix += inc_x ;
|
||||
iy += inc_y ;
|
||||
i++ ;
|
||||
|
||||
+1
-1
@@ -42,7 +42,7 @@ FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x)
|
||||
n *= inc_x;
|
||||
if (inc_x == 1)
|
||||
{
|
||||
#if V_SIMD && (!defined(DOUBLE) || (defined(DOUBLE) && V_SIMD_F64 && V_SIMD > 128))
|
||||
#if V_SIMD && (!defined(DOUBLE) || (defined(DOUBLE) && V_SIMD_F64 && (V_SIMD > 128 || defined(ARCH_WASM))))
|
||||
#ifdef DOUBLE
|
||||
const int vstep = v_nlanes_f64;
|
||||
const int unrollx4 = n & (-vstep * 4);
|
||||
|
||||
@@ -80,6 +80,11 @@ DASUMKERNEL = dasum_thunderx2t99.c
|
||||
CASUMKERNEL = casum_thunderx2t99.c
|
||||
ZASUMKERNEL = zasum_thunderx2t99.c
|
||||
|
||||
SSUMKERNEL = ssum_thunderx2t99.c
|
||||
DSUMKERNEL = dsum_thunderx2t99.c
|
||||
CSUMKERNEL = csum_thunderx2t99.c
|
||||
ZSUMKERNEL = zsum_thunderx2t99.c
|
||||
|
||||
SCOPYKERNEL = copy_thunderx2t99.c
|
||||
DCOPYKERNEL = copy_thunderx2t99.c
|
||||
CCOPYKERNEL = copy_thunderx2t99.c
|
||||
@@ -102,18 +107,8 @@ ZNRM2KERNEL = znrm2.S
|
||||
|
||||
DDOTKERNEL = dot.c
|
||||
SDOTKERNEL = dot.c
|
||||
ifeq ($(OSNAME), WINNT)
|
||||
ifeq ($(C_COMPILER), CLANG)
|
||||
CDOTKERNEL = zdot.S
|
||||
ZDOTKERNEL = zdot.S
|
||||
else
|
||||
CDOTKERNEL = zdot_thunderx2t99.c
|
||||
ZDOTKERNEL = zdot_thunderx2t99.c
|
||||
endif
|
||||
else
|
||||
CDOTKERNEL = zdot_thunderx2t99.c
|
||||
ZDOTKERNEL = zdot_thunderx2t99.c
|
||||
endif
|
||||
DSDOTKERNEL = dot.S
|
||||
|
||||
DGEMM_BETA = dgemm_beta.S
|
||||
|
||||
@@ -191,25 +191,48 @@ ZGEMMOTCOPYOBJ = zgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
ifeq ($(BUILD_BFLOAT16), 1)
|
||||
BGEMM_BETA = bgemm_beta_neon.c
|
||||
BGEMMKERNEL = sbgemm_kernel_$(BGEMM_UNROLL_M)x$(BGEMM_UNROLL_N)_neoversen2.c
|
||||
ifneq ($(BGEMM_UNROLL_M), $(BGEMM_UNROLL_N))
|
||||
BGEMMINCOPY = sbgemm_ncopy_$(BGEMM_UNROLL_M)_neoversen2.c
|
||||
BGEMMITCOPY = sbgemm_tcopy_$(BGEMM_UNROLL_M)_neoversen2.c
|
||||
BGEMMONCOPY = sbgemm_ncopy_$(BGEMM_UNROLL_N)_neoversen2.c
|
||||
BGEMMOTCOPY = sbgemm_tcopy_$(BGEMM_UNROLL_N)_neoversen2.c
|
||||
BGEMMINCOPYOBJ = bgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
BGEMMITCOPYOBJ = bgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
endif
|
||||
BGEMMONCOPY = sbgemm_ncopy_$(BGEMM_UNROLL_N)_neoversen2.c
|
||||
BGEMMOTCOPY = sbgemm_tcopy_$(BGEMM_UNROLL_N)_neoversen2.c
|
||||
BGEMMONCOPYOBJ = bgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
BGEMMOTCOPYOBJ = bgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
BGEMVTKERNEL = sbgemv_t_bfdot.c
|
||||
BGEMVNKERNEL = bgemv_n_sve_v3x4.c
|
||||
|
||||
ifeq ($(BUILD_HFLOAT16), 1)
|
||||
SHGEMMKERNEL = shgemm_kernel_$(SHGEMM_UNROLL_M)x$(SHGEMM_UNROLL_N)_neoversen2.c
|
||||
SHGEMMINCOPY = shgemm_ncopy_$(SHGEMM_UNROLL_M)_neoversen2.c
|
||||
SHGEMMITCOPY = shgemm_tcopy_$(SHGEMM_UNROLL_M)_neoversen2.c
|
||||
ifneq ($(SHGEMM_UNROLL_M), $(SHGEMM_UNROLL_N))
|
||||
SHGEMMINCOPY = ../generic/gemm_ncopy_$(SHGEMM_UNROLL_M).c
|
||||
SHGEMMITCOPY = ../generic/gemm_tcopy_$(SHGEMM_UNROLL_M).c
|
||||
endif
|
||||
SHGEMMONCOPY = shgemm_ncopy_$(SHGEMM_UNROLL_N)_neoversen2.c
|
||||
SHGEMMOTCOPY = shgemm_tcopy_$(SHGEMM_UNROLL_N)_neoversen2.c
|
||||
SHGEMMINCOPYOBJ = shgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
SHGEMMITCOPYOBJ = shgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
SHGEMMONCOPYOBJ = shgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
SHGEMMOTCOPYOBJ = shgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
ifndef SHGEMM_BETA
|
||||
SHGEMM_BETA = sbgemm_beta_neoversen2.c
|
||||
endif
|
||||
endif
|
||||
|
||||
SBGEMM_BETA = sbgemm_beta_neoversen2.c
|
||||
SBGEMMKERNEL = sbgemm_kernel_$(SBGEMM_UNROLL_M)x$(SBGEMM_UNROLL_N)_neoversen2.c
|
||||
ifneq ($(SBGEMM_UNROLL_M), $(SBGEMM_UNROLL_N))
|
||||
SBGEMMINCOPY = sbgemm_ncopy_$(SBGEMM_UNROLL_M)_neoversen2.c
|
||||
SBGEMMITCOPY = sbgemm_tcopy_$(SBGEMM_UNROLL_M)_neoversen2.c
|
||||
SBGEMMONCOPY = sbgemm_ncopy_$(SBGEMM_UNROLL_N)_neoversen2.c
|
||||
SBGEMMOTCOPY = sbgemm_tcopy_$(SBGEMM_UNROLL_N)_neoversen2.c
|
||||
SBGEMMINCOPYOBJ = sbgemm_incopy$(TSUFFIX).$(SUFFIX)
|
||||
SBGEMMITCOPYOBJ = sbgemm_itcopy$(TSUFFIX).$(SUFFIX)
|
||||
endif
|
||||
SBGEMMONCOPY = sbgemm_ncopy_$(SBGEMM_UNROLL_N)_neoversen2.c
|
||||
SBGEMMOTCOPY = sbgemm_tcopy_$(SBGEMM_UNROLL_N)_neoversen2.c
|
||||
SBGEMMONCOPYOBJ = sbgemm_oncopy$(TSUFFIX).$(SUFFIX)
|
||||
SBGEMMOTCOPYOBJ = sbgemm_otcopy$(TSUFFIX).$(SUFFIX)
|
||||
SBGEMVTKERNEL = sbgemv_t_bfdot.c
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
include $(KERNELDIR)/KERNEL.NEOVERSEN1
|
||||
@@ -262,7 +262,10 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
static RETURN_TYPE dot_kernel_asimd(BLASLONG n, FLOAT *x, BLASLONG inc_x, FLOAT *y, BLASLONG inc_y)
|
||||
{
|
||||
RETURN_TYPE dot = 0.0;
|
||||
#ifndef DOUBLE
|
||||
volatile
|
||||
#endif
|
||||
RETURN_TYPE dot = 0.0;
|
||||
BLASLONG j = 0;
|
||||
|
||||
__asm__ __volatile__ (
|
||||
|
||||
@@ -0,0 +1,244 @@
|
||||
/***************************************************************************
|
||||
Copyright (c) 2017, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written permission.
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
|
||||
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*****************************************************************************/
|
||||
|
||||
#include "common.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
#define N "x0" /* vector length */
|
||||
#define X "x1" /* "X" vector address */
|
||||
#define INC_X "x2" /* "X" stride */
|
||||
#define J "x5" /* loop variable */
|
||||
|
||||
#define REG0 "xzr"
|
||||
#define SUMF "d0"
|
||||
#define TMPF "d1"
|
||||
|
||||
/******************************************************************************/
|
||||
|
||||
#define KERNEL_F1 \
|
||||
"ldr "TMPF", ["X"] \n" \
|
||||
"add "X", "X", #8 \n" \
|
||||
"fadd "SUMF", "SUMF", "TMPF" \n"
|
||||
|
||||
#define KERNEL_F32 \
|
||||
"ldr q16, ["X"] \n" \
|
||||
"ldr q17, ["X", #16] \n" \
|
||||
"ldr q18, ["X", #32] \n" \
|
||||
"ldr q19, ["X", #48] \n" \
|
||||
"ldp q20, q21, ["X", #64] \n" \
|
||||
"ldp q22, q23, ["X", #96] \n" \
|
||||
"ldp q24, q25, ["X", #128] \n" \
|
||||
"ldp q26, q27, ["X", #160] \n" \
|
||||
"fadd v16.2d, v16.2d, v17.2d \n" \
|
||||
"fadd v18.2d, v18.2d, v19.2d \n" \
|
||||
"ldp q28, q29, ["X", #192] \n" \
|
||||
"ldp q30, q31, ["X", #224] \n" \
|
||||
"add "X", "X", #256 \n" \
|
||||
"fadd v20.2d, v20.2d, v21.2d \n" \
|
||||
"fadd v22.2d, v22.2d, v23.2d \n" \
|
||||
"PRFM PLDL1KEEP, ["X", #1024] \n" \
|
||||
"PRFM PLDL1KEEP, ["X", #1024+64] \n" \
|
||||
"fadd v24.2d, v24.2d, v25.2d \n" \
|
||||
"fadd v26.2d, v26.2d, v27.2d \n" \
|
||||
"fadd v28.2d, v28.2d, v29.2d \n" \
|
||||
"fadd v30.2d, v30.2d, v31.2d \n" \
|
||||
"fadd v0.2d, v0.2d, v16.2d \n" \
|
||||
"fadd v1.2d, v1.2d, v18.2d \n" \
|
||||
"fadd v2.2d, v2.2d, v20.2d \n" \
|
||||
"fadd v3.2d, v3.2d, v22.2d \n" \
|
||||
"PRFM PLDL1KEEP, ["X", #1024+128] \n" \
|
||||
"PRFM PLDL1KEEP, ["X", #1024+192] \n" \
|
||||
"fadd v4.2d, v4.2d, v24.2d \n" \
|
||||
"fadd v5.2d, v5.2d, v26.2d \n" \
|
||||
"fadd v6.2d, v6.2d, v28.2d \n" \
|
||||
"fadd v7.2d, v7.2d, v30.2d \n"
|
||||
|
||||
#define KERNEL_F32_FINALIZE \
|
||||
"fadd v0.2d, v0.2d, v1.2d \n" \
|
||||
"fadd v2.2d, v2.2d, v3.2d \n" \
|
||||
"fadd v4.2d, v4.2d, v5.2d \n" \
|
||||
"fadd v6.2d, v6.2d, v7.2d \n" \
|
||||
"fadd v0.2d, v0.2d, v2.2d \n" \
|
||||
"fadd v4.2d, v4.2d, v6.2d \n" \
|
||||
"fadd v0.2d, v0.2d, v4.2d \n" \
|
||||
"faddp "SUMF", v0.2d \n"
|
||||
|
||||
#define INIT_S \
|
||||
"lsl "INC_X", "INC_X", #3 \n"
|
||||
|
||||
#define KERNEL_S1 \
|
||||
"ldr "TMPF", ["X"] \n" \
|
||||
"add "X", "X", "INC_X" \n" \
|
||||
"fadd "SUMF", "SUMF", "TMPF" \n"
|
||||
|
||||
|
||||
#if defined(SMP)
|
||||
extern int blas_level1_thread_with_return_value(int mode, BLASLONG m, BLASLONG n,
|
||||
BLASLONG k, void *alpha, void *a, BLASLONG lda, void *b, BLASLONG ldb,
|
||||
void *c, BLASLONG ldc, int (*function)(), int nthreads);
|
||||
#endif
|
||||
|
||||
|
||||
static FLOAT dsum_compute(BLASLONG n, FLOAT *x, BLASLONG inc_x)
|
||||
{
|
||||
FLOAT dsum = 0.0 ;
|
||||
|
||||
if ( n < 0 ) return(dsum);
|
||||
|
||||
__asm__ __volatile__ (
|
||||
" mov "N", %[N_] \n"
|
||||
" mov "X", %[X_] \n"
|
||||
" mov "INC_X", %[INCX_] \n"
|
||||
" fmov "SUMF", "REG0" \n"
|
||||
" fmov d1, "REG0" \n"
|
||||
" fmov d2, "REG0" \n"
|
||||
" fmov d3, "REG0" \n"
|
||||
" fmov d4, "REG0" \n"
|
||||
" fmov d5, "REG0" \n"
|
||||
" fmov d6, "REG0" \n"
|
||||
" fmov d7, "REG0" \n"
|
||||
" cmp "N", xzr \n"
|
||||
" ble 9f //dsum_kernel_L999 \n"
|
||||
" cmp "INC_X", xzr \n"
|
||||
" ble 9f //dsum_kernel_L999 \n"
|
||||
" cmp "INC_X", #1 \n"
|
||||
" bne 5f //dsum_kernel_S_BEGIN \n"
|
||||
|
||||
"1: //dsum_kernel_F_BEGIN: \n"
|
||||
" asr "J", "N", #5 \n"
|
||||
" cmp "J", xzr \n"
|
||||
" beq 3f //dsum_kernel_F1 \n"
|
||||
|
||||
#if !(defined(__clang__) && defined(OS_WINDOWS))
|
||||
".align 5 \n"
|
||||
#endif
|
||||
"2: //dsum_kernel_F32: \n"
|
||||
" "KERNEL_F32" \n"
|
||||
" subs "J", "J", #1 \n"
|
||||
" bne 2b //dsum_kernel_F32 \n"
|
||||
" "KERNEL_F32_FINALIZE" \n"
|
||||
|
||||
"3: //dsum_kernel_F1: \n"
|
||||
" ands "J", "N", #31 \n"
|
||||
" ble 9f //dsum_kernel_L999 \n"
|
||||
|
||||
"4: //dsum_kernel_F10: \n"
|
||||
" "KERNEL_F1" \n"
|
||||
" subs "J", "J", #1 \n"
|
||||
" bne 4b //dsum_kernel_F10 \n"
|
||||
" b 9f //dsum_kernel_L999 \n"
|
||||
|
||||
"5: //dsum_kernel_S_BEGIN: \n"
|
||||
" "INIT_S" \n"
|
||||
" asr "J", "N", #2 \n"
|
||||
" cmp "J", xzr \n"
|
||||
" ble 7f //dsum_kernel_S1 \n"
|
||||
|
||||
"6: //dsum_kernel_S4: \n"
|
||||
" "KERNEL_S1" \n"
|
||||
" "KERNEL_S1" \n"
|
||||
" "KERNEL_S1" \n"
|
||||
" "KERNEL_S1" \n"
|
||||
" subs "J", "J", #1 \n"
|
||||
" bne 6b //dsum_kernel_S4 \n"
|
||||
|
||||
"7: //dsum_kernel_S1: \n"
|
||||
" ands "J", "N", #3 \n"
|
||||
" ble 9f //dsum_kernel_L999 \n"
|
||||
|
||||
"8: //dsum_kernel_S10: \n"
|
||||
" "KERNEL_S1" \n"
|
||||
" subs "J", "J", #1 \n"
|
||||
" bne 8b //dsum_kernel_S10 \n"
|
||||
|
||||
"9: //dsum_kernel_L999: \n"
|
||||
" fmov %[DSUM_], "SUMF" \n"
|
||||
|
||||
: [DSUM_] "=r" (dsum) //%0
|
||||
: [N_] "r" (n), //%1
|
||||
[X_] "r" (x), //%2
|
||||
[INCX_] "r" (inc_x) //%3
|
||||
: "cc",
|
||||
"memory",
|
||||
"x0", "x1", "x2", "x3", "x4", "x5",
|
||||
"d0", "d1", "d2", "d3", "d4", "d5", "d6", "d7"
|
||||
);
|
||||
|
||||
return dsum;
|
||||
}
|
||||
|
||||
#if defined(SMP)
|
||||
static int dsum_thread_function(BLASLONG n, BLASLONG dummy0,
|
||||
BLASLONG dummy1, FLOAT dummy2, FLOAT *x, BLASLONG inc_x, FLOAT *y,
|
||||
BLASLONG inc_y, FLOAT *result, BLASLONG dummy3)
|
||||
{
|
||||
*result = dsum_compute(n, x, inc_x);
|
||||
|
||||
return 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x)
|
||||
{
|
||||
#if defined(SMP)
|
||||
int nthreads;
|
||||
FLOAT dummy_alpha;
|
||||
#endif
|
||||
FLOAT dsum = 0.0;
|
||||
|
||||
#if defined(SMP)
|
||||
if (inc_x == 0 || n <= 10000)
|
||||
nthreads = 1;
|
||||
else
|
||||
nthreads = num_cpu_avail(1);
|
||||
|
||||
if (nthreads == 1) {
|
||||
dsum = dsum_compute(n, x, inc_x);
|
||||
} else {
|
||||
int mode, i;
|
||||
char result[MAX_CPU_NUMBER * sizeof(double) * 2];
|
||||
FLOAT *ptr;
|
||||
|
||||
mode = BLAS_DOUBLE;
|
||||
|
||||
blas_level1_thread_with_return_value(mode, n, 0, 0, &dummy_alpha,
|
||||
x, inc_x, NULL, 0, result, 0,
|
||||
( void *)dsum_thread_function, nthreads);
|
||||
|
||||
ptr = (FLOAT *)result;
|
||||
for (i = 0; i < nthreads; i++) {
|
||||
dsum = dsum + (*ptr);
|
||||
ptr = (FLOAT *)(((char *)ptr) + sizeof(double) * 2);
|
||||
}
|
||||
}
|
||||
#else
|
||||
dsum = dsum_compute(n, x, inc_x);
|
||||
#endif
|
||||
|
||||
return dsum;
|
||||
}
|
||||
@@ -155,7 +155,10 @@ static double nrm2_compute(BLASLONG n, FLOAT *x, BLASLONG inc_x)
|
||||
" cmp "J", xzr \n"
|
||||
" beq .Lnrm2_kernel_F1 \n"
|
||||
|
||||
/* https://github.com/llvm/llvm-project/issues/149547 */
|
||||
#if !(defined(__clang__) && defined(OS_WINDOWS))
|
||||
" .align 5 \n"
|
||||
#endif
|
||||
".Lnrm2_kernel_F: \n"
|
||||
" "KERNEL_F" \n"
|
||||
" subs "J", "J", #1 \n"
|
||||
|
||||
+10
-37
@@ -35,16 +35,13 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#define I x3
|
||||
|
||||
#if !defined(DOUBLE)
|
||||
#define SSQ s0
|
||||
#define SCALE s1
|
||||
#define REGZERO s5
|
||||
#define REGONE s6
|
||||
#else
|
||||
#define SSQF s0
|
||||
#endif
|
||||
|
||||
#define SSQ d0
|
||||
#define SCALE d1
|
||||
#define REGZERO d5
|
||||
#define REGONE d6
|
||||
#endif
|
||||
|
||||
/*******************************************************************************
|
||||
* Macro definitions
|
||||
@@ -53,22 +50,10 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
.macro KERNEL_F1
|
||||
#if !defined(DOUBLE)
|
||||
ldr s4, [X], #4
|
||||
fcmp s4, REGZERO
|
||||
beq 2f /* KERNEL_F1_NEXT_\@ */
|
||||
fabs s4, s4
|
||||
fcmp SCALE, s4
|
||||
bge 1f /* KERNEL_F1_SCALE_GE_X_\@ */
|
||||
fdiv s2, SCALE, s4
|
||||
fmul s2, s2, s2
|
||||
fmul s3, SSQ, s2
|
||||
fadd SSQ, REGONE, s3
|
||||
fmov SCALE, s4
|
||||
b 2f /* KERNEL_F1_NEXT_\@ */
|
||||
1: /* KERNEL_F1_SCALE_GE_X_\@: */
|
||||
fdiv s2, s4, SCALE
|
||||
fmla SSQ, s2, v2.s[0]
|
||||
fcvt d4, s4
|
||||
#else
|
||||
ldr d4, [X], #8
|
||||
#endif
|
||||
fcmp d4, REGZERO
|
||||
beq 2f /* KERNEL_F1_NEXT_\@ */
|
||||
fabs d4, d4
|
||||
@@ -83,29 +68,16 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
1: /* KERNEL_F1_SCALE_GE_X_\@: */
|
||||
fdiv d2, d4, SCALE
|
||||
fmla SSQ, d2, v2.d[0]
|
||||
#endif
|
||||
2: /* KERNEL_F1_NEXT_\@: */
|
||||
.endm
|
||||
|
||||
.macro KERNEL_S1
|
||||
#if !defined(DOUBLE)
|
||||
ldr s4, [X]
|
||||
fcmp s4, REGZERO
|
||||
beq KERNEL_S1_NEXT
|
||||
fabs s4, s4
|
||||
fcmp SCALE, s4
|
||||
bge KERNEL_S1_SCALE_GE_X
|
||||
fdiv s2, SCALE, s4
|
||||
fmul s2, s2, s2
|
||||
fmul s3, SSQ, s2
|
||||
fadd SSQ, REGONE, s3
|
||||
fmov SCALE, s4
|
||||
b KERNEL_S1_NEXT
|
||||
KERNEL_S1_SCALE_GE_X:
|
||||
fdiv s2, s4, SCALE
|
||||
fmla SSQ, s2, v2.s[0]
|
||||
fcvt d4, s4
|
||||
#else
|
||||
ldr d4, [X]
|
||||
#endif
|
||||
fcmp d4, REGZERO
|
||||
beq KERNEL_S1_NEXT
|
||||
fabs d4, d4
|
||||
@@ -120,7 +92,6 @@ KERNEL_S1_SCALE_GE_X:
|
||||
KERNEL_S1_SCALE_GE_X:
|
||||
fdiv d2, d4, SCALE
|
||||
fmla SSQ, d2, v2.d[0]
|
||||
#endif
|
||||
KERNEL_S1_NEXT:
|
||||
add X, X, INC_X
|
||||
.endm
|
||||
@@ -218,7 +189,9 @@ KERNEL_S1_NEXT:
|
||||
.Lnrm2_kernel_L999:
|
||||
fsqrt SSQ, SSQ
|
||||
fmul SSQ, SCALE, SSQ
|
||||
|
||||
#if !defined(DOUBLE)
|
||||
fcvt SSQF, SSQ
|
||||
#endif
|
||||
ret
|
||||
|
||||
EPILOGUE
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
/***************************************************************************
|
||||
* Copyright (c) 2026 The OpenBLAS Project
|
||||
* All rights reserved.
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are
|
||||
* met:
|
||||
* 1. Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* 2. Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in
|
||||
* the documentation and/or other materials provided with the
|
||||
* distribution.
|
||||
* 3. Neither the name of the OpenBLAS project nor the names of
|
||||
* its contributors may be used to endorse or promote products
|
||||
* derived from this software without specific prior written permission.
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
* ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
* POSSIBILITY OF SUCH DAMAGE.
|
||||
* *****************************************************************************/
|
||||
|
||||
#include <arm_sve.h>
|
||||
#include <arm_neon.h>
|
||||
|
||||
#include "common.h"
|
||||
|
||||
#define ALPHA_ONE
|
||||
#include "sbgemm_kernel_8x8_neoversen2_impl.c"
|
||||
#undef ALPHA_ONE
|
||||
#undef UPDATE_C
|
||||
#include "sbgemm_kernel_8x8_neoversen2_impl.c"
|
||||
|
||||
int CNAME(BLASLONG m, BLASLONG n, BLASLONG k, FLOAT alpha, IFLOAT *A, IFLOAT *B,
|
||||
FLOAT *C, BLASLONG ldc) {
|
||||
#ifdef BGEMM
|
||||
bfloat16_t alpha_bf16;
|
||||
memcpy(&alpha_bf16, &alpha, sizeof(bfloat16_t));
|
||||
float alpha_f32 = vcvtah_f32_bf16(alpha_bf16);
|
||||
#else
|
||||
float alpha_f32 = alpha;
|
||||
#endif
|
||||
|
||||
if (alpha_f32 == 1.0f)
|
||||
return gemm_kernel_neoversen2_alpha_one(m, n, k, alpha, A, B, C, ldc);
|
||||
else
|
||||
return gemm_kernel_neoversen2_alpha(m, n, k, alpha, A, B, C, ldc);
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,763 @@
|
||||
/***************************************************************************
|
||||
* Copyright (c) 2022,2026 The OpenBLAS Project
|
||||
* All rights reserved.
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are
|
||||
* met:
|
||||
* 1. Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* 2. Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in
|
||||
* the documentation and/or other materials provided with the
|
||||
* distribution.
|
||||
* 3. Neither the name of the OpenBLAS project nor the names of
|
||||
* its contributors may be used to endorse or promote products
|
||||
* derived from this software without specific prior written permission.
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
* ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
* POSSIBILITY OF SUCH DAMAGE.
|
||||
* *****************************************************************************/
|
||||
|
||||
#include <arm_sve.h>
|
||||
#include <arm_neon.h>
|
||||
|
||||
#include "common.h"
|
||||
|
||||
#define INIT_C(M, N) mc##M##N = svdup_f32(0);
|
||||
|
||||
#define MATMUL(M, N) mc##M##N = svbfmmla(mc##M##N, ma##M, mb##N);
|
||||
|
||||
#define INIT_C_8x4 \
|
||||
do { \
|
||||
INIT_C(0, 0); \
|
||||
INIT_C(0, 1); \
|
||||
INIT_C(1, 0); \
|
||||
INIT_C(1, 1); \
|
||||
INIT_C(2, 0); \
|
||||
INIT_C(2, 1); \
|
||||
INIT_C(3, 0); \
|
||||
INIT_C(3, 1); \
|
||||
} while (0);
|
||||
|
||||
#define INIT_C_8x8 \
|
||||
do { \
|
||||
INIT_C(0, 0); \
|
||||
INIT_C(0, 1); \
|
||||
INIT_C(0, 2); \
|
||||
INIT_C(0, 3); \
|
||||
INIT_C(1, 0); \
|
||||
INIT_C(1, 1); \
|
||||
INIT_C(1, 2); \
|
||||
INIT_C(1, 3); \
|
||||
INIT_C(2, 0); \
|
||||
INIT_C(2, 1); \
|
||||
INIT_C(2, 2); \
|
||||
INIT_C(2, 3); \
|
||||
INIT_C(3, 0); \
|
||||
INIT_C(3, 1); \
|
||||
INIT_C(3, 2); \
|
||||
INIT_C(3, 3); \
|
||||
} while (0);
|
||||
|
||||
#ifdef BGEMM
|
||||
#ifdef ALPHA_ONE
|
||||
#define UPDATE_C(PG16, PG32, PTR, SRC) \
|
||||
do { \
|
||||
tmp16 = svld1_bf16((PG16), (PTR)); \
|
||||
tmp32 = svreinterpret_f32(svzip1_bf16(zeros, tmp16)); \
|
||||
tmp32 = svadd_z((PG32), SRC, tmp32); \
|
||||
tmp16 = svcvt_bf16_f32_z((PG32), tmp32); \
|
||||
tmp16 = svuzp1_bf16(tmp16, tmp16); \
|
||||
svst1_bf16((PG16), (PTR), tmp16); \
|
||||
} while (0)
|
||||
#else
|
||||
#define UPDATE_C(PG16, PG32, PTR, SRC) \
|
||||
do { \
|
||||
tmp16 = svld1_bf16((PG16), (PTR)); \
|
||||
tmp32 = svreinterpret_f32(svzip1_bf16(zeros, tmp16)); \
|
||||
tmp32 = svmad_z((PG32), svalpha, SRC, tmp32); \
|
||||
tmp16 = svcvt_bf16_f32_z((PG32), tmp32); \
|
||||
tmp16 = svuzp1_bf16(tmp16, tmp16); \
|
||||
svst1_bf16((PG16), (PTR), tmp16); \
|
||||
} while (0)
|
||||
#endif
|
||||
#else
|
||||
#ifdef ALPHA_ONE
|
||||
#define UPDATE_C(PG16, PG32, PTR, SRC) \
|
||||
do { \
|
||||
tmp32 = svld1_f32((PG32), (PTR)); \
|
||||
tmp32 = svadd_z((PG32), SRC, tmp32); \
|
||||
svst1_f32((PG32), (PTR), tmp32); \
|
||||
} while (0);
|
||||
#else
|
||||
#define UPDATE_C(PG16, PG32, PTR, SRC) \
|
||||
do { \
|
||||
tmp32 = svld1_f32((PG32), (PTR)); \
|
||||
tmp32 = svmad_z((PG32), svalpha, SRC, tmp32); \
|
||||
svst1_f32((PG32), (PTR), tmp32); \
|
||||
} while (0);
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef BGEMM
|
||||
#define OUTPUT_FLOAT bfloat16_t
|
||||
#else
|
||||
#define OUTPUT_FLOAT float
|
||||
#endif
|
||||
|
||||
#ifdef ALPHA_ONE
|
||||
static int gemm_kernel_neoversen2_alpha_one(BLASLONG m, BLASLONG n, BLASLONG k, FLOAT alpha, IFLOAT * A, IFLOAT * B, FLOAT * C, BLASLONG ldc)
|
||||
#else
|
||||
static int gemm_kernel_neoversen2_alpha(BLASLONG m, BLASLONG n, BLASLONG k, FLOAT alpha, IFLOAT * A, IFLOAT * B, FLOAT * C, BLASLONG ldc)
|
||||
#endif
|
||||
{
|
||||
BLASLONG pad_k = (k + 3) & ~3;
|
||||
|
||||
svbfloat16_t ma0, ma1, ma2, ma3, mb0, mb1, mb2, mb3;
|
||||
svfloat32_t mc00, mc01, mc02, mc03;
|
||||
svfloat32_t mc10, mc11, mc12, mc13;
|
||||
svfloat32_t mc20, mc21, mc22, mc23;
|
||||
svfloat32_t mc30, mc31, mc32, mc33;
|
||||
svfloat32_t vc0, vc1, vc2, vc3, vc4, vc5, vc6, vc7;
|
||||
svfloat32_t vc8, vc9, vc10, vc11, vc12, vc13, vc14, vc15;
|
||||
|
||||
#ifndef ALPHA_ONE
|
||||
#ifdef BGEMM
|
||||
bfloat16_t alpha_bf16;
|
||||
memcpy(&alpha_bf16, &alpha, sizeof(bfloat16_t));
|
||||
svfloat32_t svalpha = svdup_f32(vcvtah_f32_bf16(alpha_bf16));
|
||||
#else
|
||||
svfloat32_t svalpha = svdup_f32(alpha);
|
||||
#endif
|
||||
#endif
|
||||
|
||||
svbool_t pg32_first_4 = svdupq_b32(1, 1, 1, 1);
|
||||
svbool_t pg32_first_2 = svdupq_b32(1, 1, 0, 0);
|
||||
svbool_t pg32_first_1 = svdupq_b32(1, 0, 0, 0);
|
||||
svbool_t pg16_first_8 = svdupq_b16(1, 1, 1, 1, 1, 1, 1, 1);
|
||||
svbool_t pg16_first_4 = svdupq_b16(1, 1, 1, 1, 0, 0, 0, 0);
|
||||
#ifdef BGEMM
|
||||
svbool_t pg16_first_2 = svdupq_b16(1, 1, 0, 0, 0, 0, 0, 0);
|
||||
svbool_t pg16_first_1 = svdupq_b16(1, 0, 0, 0, 0, 0, 0, 0);
|
||||
svbfloat16_t zeros = svdup_n_bf16(vcvth_bf16_f32(0.0));
|
||||
#endif
|
||||
|
||||
bfloat16_t *ptr_a = (bfloat16_t *)A;
|
||||
bfloat16_t *ptr_b = (bfloat16_t *)B;
|
||||
OUTPUT_FLOAT *ptr_c = (OUTPUT_FLOAT*)C;
|
||||
|
||||
bfloat16_t *ptr_a0;
|
||||
bfloat16_t *ptr_b0;
|
||||
OUTPUT_FLOAT *ptr_c0, *ptr_c1, *ptr_c2, *ptr_c3;
|
||||
OUTPUT_FLOAT *ptr_c4, *ptr_c5, *ptr_c6, *ptr_c7;
|
||||
|
||||
svfloat32_t tmp32;
|
||||
#ifdef BGEMM
|
||||
svbfloat16_t tmp16;
|
||||
#endif
|
||||
|
||||
for (BLASLONG j = 0; j < n / 8; j++) {
|
||||
ptr_c0 = ptr_c;
|
||||
ptr_c1 = ptr_c0 + ldc;
|
||||
ptr_c2 = ptr_c1 + ldc;
|
||||
ptr_c3 = ptr_c2 + ldc;
|
||||
ptr_c4 = ptr_c3 + ldc;
|
||||
ptr_c5 = ptr_c4 + ldc;
|
||||
ptr_c6 = ptr_c5 + ldc;
|
||||
ptr_c7 = ptr_c6 + ldc;
|
||||
ptr_c += 8 * ldc;
|
||||
ptr_a = (bfloat16_t *)A;
|
||||
|
||||
for (BLASLONG i = 0; i < m / 8; i++) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 8 * pad_k;
|
||||
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C_8x8;
|
||||
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
|
||||
ma2 = svld1_bf16(pg16_first_8, ptr_a0 + 16);
|
||||
ma3 = svld1_bf16(pg16_first_8, ptr_a0 + 24);
|
||||
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
|
||||
mb2 = svld1_bf16(pg16_first_8, ptr_b0 + 16);
|
||||
mb3 = svld1_bf16(pg16_first_8, ptr_b0 + 24);
|
||||
|
||||
MATMUL(0, 0); MATMUL(0, 1); MATMUL(0, 2); MATMUL(0, 3);
|
||||
MATMUL(1, 0); MATMUL(1, 1); MATMUL(1, 2); MATMUL(1, 3);
|
||||
MATMUL(2, 0); MATMUL(2, 1); MATMUL(2, 2); MATMUL(2, 3);
|
||||
MATMUL(3, 0); MATMUL(3, 1); MATMUL(3, 2); MATMUL(3, 3);
|
||||
|
||||
ptr_a0 += 32;
|
||||
ptr_b0 += 32;
|
||||
}
|
||||
|
||||
vc0 = svuzp1(mc00, mc10);
|
||||
vc1 = svuzp1(mc20, mc30);
|
||||
vc2 = svuzp2(mc00, mc10);
|
||||
vc3 = svuzp2(mc20, mc30);
|
||||
vc4 = svuzp1(mc01, mc11);
|
||||
vc5 = svuzp1(mc21, mc31);
|
||||
vc6 = svuzp2(mc01, mc11);
|
||||
vc7 = svuzp2(mc21, mc31);
|
||||
vc8 = svuzp1(mc02, mc12);
|
||||
vc9 = svuzp1(mc22, mc32);
|
||||
vc10 = svuzp2(mc02, mc12);
|
||||
vc11 = svuzp2(mc22, mc32);
|
||||
vc12 = svuzp1(mc03, mc13);
|
||||
vc13 = svuzp1(mc23, mc33);
|
||||
vc14 = svuzp2(mc03, mc13);
|
||||
vc15 = svuzp2(mc23, mc33);
|
||||
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0 + 4, vc1);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc2);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1 + 4, vc3);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2, vc4);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2 + 4, vc5);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3, vc6);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3 + 4, vc7);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c4, vc8);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c4 + 4, vc9);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c5, vc10);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c5 + 4, vc11);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c6, vc12);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c6 + 4, vc13);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c7, vc14);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c7 + 4, vc15);
|
||||
|
||||
ptr_c0 += 8;
|
||||
ptr_c1 += 8;
|
||||
ptr_c2 += 8;
|
||||
ptr_c3 += 8;
|
||||
ptr_c4 += 8;
|
||||
ptr_c5 += 8;
|
||||
ptr_c6 += 8;
|
||||
ptr_c7 += 8;
|
||||
}
|
||||
|
||||
if (m & 4) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 4 * pad_k;
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0); INIT_C(0, 1); INIT_C(0, 2); INIT_C(0, 3);
|
||||
INIT_C(1, 0); INIT_C(1, 1); INIT_C(1, 2); INIT_C(1, 3);
|
||||
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
|
||||
mb2 = svld1_bf16(pg16_first_8, ptr_b0 + 16);
|
||||
mb3 = svld1_bf16(pg16_first_8, ptr_b0 + 24);
|
||||
|
||||
MATMUL(0, 0); MATMUL(0, 1); MATMUL(0, 2); MATMUL(0, 3);
|
||||
MATMUL(1, 0); MATMUL(1, 1); MATMUL(1, 2); MATMUL(1, 3);
|
||||
|
||||
ptr_a0 += 16;
|
||||
ptr_b0 += 32;
|
||||
}
|
||||
|
||||
vc0 = svuzp1(mc00, mc10);
|
||||
vc1 = svuzp2(mc00, mc10);
|
||||
vc2 = svuzp1(mc01, mc11);
|
||||
vc3 = svuzp2(mc01, mc11);
|
||||
vc4 = svuzp1(mc02, mc12);
|
||||
vc5 = svuzp2(mc02, mc12);
|
||||
vc6 = svuzp1(mc03, mc13);
|
||||
vc7 = svuzp2(mc03, mc13);
|
||||
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc1);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2, vc2);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3, vc3);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c4, vc4);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c5, vc5);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c6, vc6);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c7, vc7);
|
||||
|
||||
ptr_c0 += 4;
|
||||
ptr_c1 += 4;
|
||||
ptr_c2 += 4;
|
||||
ptr_c3 += 4;
|
||||
ptr_c4 += 4;
|
||||
ptr_c5 += 4;
|
||||
ptr_c6 += 4;
|
||||
ptr_c7 += 4;
|
||||
}
|
||||
|
||||
if (m & 2) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 2 * pad_k;
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0); INIT_C(0, 1); INIT_C(0, 2); INIT_C(0, 3);
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
|
||||
mb2 = svld1_bf16(pg16_first_8, ptr_b0 + 16);
|
||||
mb3 = svld1_bf16(pg16_first_8, ptr_b0 + 24);
|
||||
|
||||
MATMUL(0, 0); MATMUL(0, 1); MATMUL(0, 2); MATMUL(0, 3);
|
||||
|
||||
ptr_a0 += 8;
|
||||
ptr_b0 += 32;
|
||||
}
|
||||
|
||||
vc0 = svuzp1(mc00, mc00);
|
||||
vc1 = svuzp2(mc00, mc00);
|
||||
vc2 = svuzp1(mc01, mc01);
|
||||
vc3 = svuzp2(mc01, mc01);
|
||||
vc4 = svuzp1(mc02, mc02);
|
||||
vc5 = svuzp2(mc02, mc02);
|
||||
vc6 = svuzp1(mc03, mc03);
|
||||
vc7 = svuzp2(mc03, mc03);
|
||||
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c0, vc0);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c1, vc1);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c2, vc2);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c3, vc3);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c4, vc4);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c5, vc5);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c6, vc6);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c7, vc7);
|
||||
|
||||
ptr_c0 += 2;
|
||||
ptr_c1 += 2;
|
||||
ptr_c2 += 2;
|
||||
ptr_c3 += 2;
|
||||
ptr_c4 += 2;
|
||||
ptr_c5 += 2;
|
||||
ptr_c6 += 2;
|
||||
ptr_c7 += 2;
|
||||
}
|
||||
|
||||
if (m & 1) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0); INIT_C(0, 1); INIT_C(0, 2); INIT_C(0, 3);
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_4, ptr_a0);
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
|
||||
mb2 = svld1_bf16(pg16_first_8, ptr_b0 + 16);
|
||||
mb3 = svld1_bf16(pg16_first_8, ptr_b0 + 24);
|
||||
|
||||
MATMUL(0, 0); MATMUL(0, 1); MATMUL(0, 2); MATMUL(0, 3);
|
||||
|
||||
ptr_a0 += 4;
|
||||
ptr_b0 += 32;
|
||||
}
|
||||
|
||||
vc1 = svuzp2(mc00, mc00);
|
||||
vc3 = svuzp2(mc01, mc01);
|
||||
vc5 = svuzp2(mc02, mc02);
|
||||
vc7 = svuzp2(mc03, mc03);
|
||||
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c0, mc00);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c1, vc1);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c2, mc01);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c3, vc3);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c4, mc02);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c5, vc5);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c6, mc03);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c7, vc7);
|
||||
}
|
||||
|
||||
ptr_b += 8 * pad_k;
|
||||
}
|
||||
|
||||
if (n & 4) {
|
||||
ptr_c0 = ptr_c;
|
||||
ptr_c1 = ptr_c0 + ldc;
|
||||
ptr_c2 = ptr_c1 + ldc;
|
||||
ptr_c3 = ptr_c2 + ldc;
|
||||
ptr_c += 4 * ldc;
|
||||
ptr_a = (bfloat16_t *)A;
|
||||
|
||||
for (BLASLONG i = 0; i < m / 8; i++) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 8 * pad_k;
|
||||
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C_8x4;
|
||||
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
|
||||
ma2 = svld1_bf16(pg16_first_8, ptr_a0 + 16);
|
||||
ma3 = svld1_bf16(pg16_first_8, ptr_a0 + 24);
|
||||
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
|
||||
|
||||
MATMUL(0, 0); MATMUL(0, 1);
|
||||
MATMUL(1, 0); MATMUL(1, 1);
|
||||
MATMUL(2, 0); MATMUL(2, 1);
|
||||
MATMUL(3, 0); MATMUL(3, 1);
|
||||
|
||||
ptr_a0 += 32;
|
||||
ptr_b0 += 16;
|
||||
}
|
||||
|
||||
vc0 = svuzp1(mc00, mc10);
|
||||
vc1 = svuzp1(mc20, mc30);
|
||||
vc2 = svuzp2(mc00, mc10);
|
||||
vc3 = svuzp2(mc20, mc30);
|
||||
vc4 = svuzp1(mc01, mc11);
|
||||
vc5 = svuzp1(mc21, mc31);
|
||||
vc6 = svuzp2(mc01, mc11);
|
||||
vc7 = svuzp2(mc21, mc31);
|
||||
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0+4, vc1);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc2);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1+4, vc3);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2, vc4);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2+4, vc5);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3, vc6);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3+4, vc7);
|
||||
|
||||
ptr_c0 += 8;
|
||||
ptr_c1 += 8;
|
||||
ptr_c2 += 8;
|
||||
ptr_c3 += 8;
|
||||
}
|
||||
|
||||
if (m & 4) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 4 * pad_k;
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0); INIT_C(0, 1);
|
||||
INIT_C(1, 0); INIT_C(1, 1);
|
||||
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
|
||||
|
||||
MATMUL(0, 0); MATMUL(0, 1);
|
||||
MATMUL(1, 0); MATMUL(1, 1);
|
||||
|
||||
ptr_a0 += 16;
|
||||
ptr_b0 += 16;
|
||||
}
|
||||
|
||||
vc0 = svuzp1(mc00, mc10);
|
||||
vc1 = svuzp2(mc00, mc10);
|
||||
vc2 = svuzp1(mc01, mc11);
|
||||
vc3 = svuzp2(mc01, mc11);
|
||||
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc1);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2, vc2);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3, vc3);
|
||||
|
||||
ptr_c0 += 4;
|
||||
ptr_c1 += 4;
|
||||
ptr_c2 += 4;
|
||||
ptr_c3 += 4;
|
||||
}
|
||||
|
||||
if (m & 2) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 2 * pad_k;
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0); INIT_C(0, 1);
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
|
||||
|
||||
MATMUL(0, 0); MATMUL(0, 1);
|
||||
|
||||
ptr_a0 += 8;
|
||||
ptr_b0 += 16;
|
||||
}
|
||||
|
||||
vc0 = svuzp1(mc00, mc00);
|
||||
vc1 = svuzp2(mc00, mc00);
|
||||
vc2 = svuzp1(mc01, mc01);
|
||||
vc3 = svuzp2(mc01, mc01);
|
||||
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c0, vc0);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c1, vc1);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c2, vc2);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c3, vc3);
|
||||
|
||||
ptr_c0 += 2;
|
||||
ptr_c1 += 2;
|
||||
ptr_c2 += 2;
|
||||
ptr_c3 += 2;
|
||||
}
|
||||
|
||||
if (m & 1) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0); INIT_C(0, 1);
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_4, ptr_a0);
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
|
||||
|
||||
MATMUL(0, 0); MATMUL(0, 1);
|
||||
|
||||
ptr_a0 += 4;
|
||||
ptr_b0 += 16;
|
||||
}
|
||||
|
||||
vc1 = svuzp2(mc00, mc00);
|
||||
vc3 = svuzp2(mc01, mc01);
|
||||
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c0, mc00);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c1, vc1);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c2, mc01);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c3, vc3);
|
||||
|
||||
}
|
||||
|
||||
ptr_b += 4 * pad_k;
|
||||
}
|
||||
|
||||
if (n & 2) {
|
||||
ptr_c0 = ptr_c;
|
||||
ptr_c1 = ptr_c0 + ldc;
|
||||
ptr_c += 2 * ldc;
|
||||
ptr_a = (bfloat16_t *)A;
|
||||
|
||||
for (BLASLONG i = 0; i < m / 8; i++) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 8 * pad_k;
|
||||
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0);
|
||||
INIT_C(1, 0);
|
||||
INIT_C(2, 0);
|
||||
INIT_C(3, 0);
|
||||
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
|
||||
ma2 = svld1_bf16(pg16_first_8, ptr_a0 + 16);
|
||||
ma3 = svld1_bf16(pg16_first_8, ptr_a0 + 24);
|
||||
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
|
||||
MATMUL(0, 0);
|
||||
MATMUL(1, 0);
|
||||
MATMUL(2, 0);
|
||||
MATMUL(3, 0);
|
||||
|
||||
ptr_a0 += 32;
|
||||
ptr_b0 += 8;
|
||||
}
|
||||
|
||||
vc0 = svuzp1(mc00, mc10);
|
||||
vc1 = svuzp1(mc20, mc30);
|
||||
vc2 = svuzp2(mc00, mc10);
|
||||
vc3 = svuzp2(mc20, mc30);
|
||||
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0 + 4, vc1);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc2);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1 + 4, vc3);
|
||||
|
||||
ptr_c0 += 8;
|
||||
ptr_c1 += 8;
|
||||
}
|
||||
|
||||
if (m & 4) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 4 * pad_k;
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0);
|
||||
INIT_C(1, 0);
|
||||
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
MATMUL(0, 0);
|
||||
MATMUL(1, 0);
|
||||
ptr_a0 += 16;
|
||||
ptr_b0 += 8;
|
||||
}
|
||||
|
||||
vc0 = svuzp1(mc00, mc10);
|
||||
vc1 = svuzp2(mc00, mc10);
|
||||
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc1);
|
||||
|
||||
ptr_c0 += 4;
|
||||
ptr_c1 += 4;
|
||||
}
|
||||
|
||||
if (m & 2) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 2 * pad_k;
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0);
|
||||
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
|
||||
MATMUL(0, 0);
|
||||
|
||||
ptr_a0 += 8;
|
||||
ptr_b0 += 8;
|
||||
}
|
||||
|
||||
vc0 = svuzp1(mc00, mc00);
|
||||
vc1 = svuzp2(mc00, mc00);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c0, vc0);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c1, vc1);
|
||||
|
||||
ptr_c0 += 2;
|
||||
ptr_c1 += 2;
|
||||
|
||||
}
|
||||
|
||||
if (m & 1) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_b0 = ptr_b;
|
||||
INIT_C(0, 0);
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_4, ptr_a0);
|
||||
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
|
||||
MATMUL(0, 0);
|
||||
ptr_a0 += 4;
|
||||
ptr_b0 += 8;
|
||||
}
|
||||
vc1 = svuzp2(mc00, mc00);
|
||||
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c0, mc00);
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c1, vc1);
|
||||
}
|
||||
|
||||
ptr_b += 2 * pad_k;
|
||||
}
|
||||
|
||||
if (n & 1) {
|
||||
ptr_c0 = ptr_c;
|
||||
ptr_a = (bfloat16_t *)A;
|
||||
|
||||
for (BLASLONG i = 0; i < m / 8; i++) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 8 * pad_k;
|
||||
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0);
|
||||
INIT_C(1, 0);
|
||||
INIT_C(2, 0);
|
||||
INIT_C(3, 0);
|
||||
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
|
||||
ma2 = svld1_bf16(pg16_first_8, ptr_a0 + 16);
|
||||
ma3 = svld1_bf16(pg16_first_8, ptr_a0 + 24);
|
||||
|
||||
mb0 = svld1_bf16(pg16_first_4, ptr_b0);
|
||||
|
||||
MATMUL(0, 0);
|
||||
MATMUL(1, 0);
|
||||
MATMUL(2, 0);
|
||||
MATMUL(3, 0);
|
||||
|
||||
ptr_a0 += 32;
|
||||
ptr_b0 += 4;
|
||||
}
|
||||
|
||||
vc0 = svuzp1(mc00, mc10);
|
||||
vc1 = svuzp1(mc20, mc30);
|
||||
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0 + 4, vc1);
|
||||
|
||||
ptr_c0 += 8;
|
||||
}
|
||||
|
||||
if (m & 4) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 4 * pad_k;
|
||||
ptr_b0 = ptr_b;
|
||||
INIT_C(0, 0);
|
||||
INIT_C(1, 0);
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
|
||||
mb0 = svld1_bf16(pg16_first_4, ptr_b0);
|
||||
MATMUL(0, 0);
|
||||
MATMUL(1, 0);
|
||||
ptr_a0 += 16;
|
||||
ptr_b0 += 4;
|
||||
}
|
||||
vc0 = svuzp1(mc00, mc10);
|
||||
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
|
||||
ptr_c0 += 4;
|
||||
}
|
||||
|
||||
if (m & 2) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_a += 2 * pad_k;
|
||||
ptr_b0 = ptr_b;
|
||||
|
||||
INIT_C(0, 0);
|
||||
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
|
||||
mb0 = svld1_bf16(pg16_first_4, ptr_b0);
|
||||
|
||||
MATMUL(0, 0);
|
||||
|
||||
ptr_a0 += 8;
|
||||
ptr_b0 += 4;
|
||||
}
|
||||
vc0 = svuzp1(mc00, mc00);
|
||||
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c0, vc0);
|
||||
ptr_c0 += 2;
|
||||
}
|
||||
|
||||
if (m & 1) {
|
||||
ptr_a0 = ptr_a;
|
||||
ptr_b0 = ptr_b;
|
||||
INIT_C(0, 0);
|
||||
for (BLASLONG p = 0; p < pad_k; p += 4) {
|
||||
ma0 = svld1_bf16(pg16_first_4, ptr_a0);
|
||||
mb0 = svld1_bf16(pg16_first_4, ptr_b0);
|
||||
MATMUL(0, 0);
|
||||
ptr_a0 += 4;
|
||||
ptr_b0 += 4;
|
||||
}
|
||||
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c0, mc00);
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -238,7 +238,10 @@ static double nrm2_compute(BLASLONG n, FLOAT *x, BLASLONG inc_x)
|
||||
" cmp "J", xzr \n"
|
||||
" beq 5f //nrm2_kernel_S_BEGIN \n"
|
||||
|
||||
/* https://github.com/llvm/llvm-project/issues/149547 */
|
||||
#if !(defined(__clang__) && defined(OS_WINDOWS))
|
||||
" .align 5 \n"
|
||||
#endif
|
||||
"2: //nrm2_kernel_F: \n"
|
||||
" "KERNEL_F" \n"
|
||||
" subs "J", "J", #1 \n"
|
||||
|
||||
@@ -7,17 +7,30 @@
|
||||
#include <stdlib.h>
|
||||
#include <inttypes.h>
|
||||
#include <math.h>
|
||||
#include "sme_abi.h"
|
||||
|
||||
#if defined(DYNAMIC_ARCH)
|
||||
#define COMBINE(a,b) a ## b
|
||||
#define COMBINE2(a,b) COMBINE(a,b)
|
||||
#define SME1_PREPROCESS_BASE sgemm_direct_sme1_preprocess
|
||||
#define SME1_PREPROCESS COMBINE2(SME1_PREPROCESS_BASE,TS)
|
||||
#define SME1_KERNEL2X2_BASE sgemm_direct_alpha_beta_sme1_2VLx2VL
|
||||
#define SME1_KERNEL2X2 COMBINE2(SME1_KERNEL2X2_BASE,TS)
|
||||
#else
|
||||
#define SME1_PREPROCESS sgemm_direct_sme1_preprocess
|
||||
#define SME1_KERNEL2X2 sgemm_direct_alpha_beta_sme1_2VLx2VL
|
||||
#endif
|
||||
|
||||
/* Function prototypes */
|
||||
extern void SME1_PREPROCESS(uint64_t nbr, uint64_t nbc,\
|
||||
const float * restrict a, float * a_mod);
|
||||
|
||||
#if defined(HAVE_SME)
|
||||
#include "sme_abi.h"
|
||||
|
||||
#if defined(__ARM_FEATURE_SME) && defined(__clang__) && __clang_major__ >= 16
|
||||
#include <arm_sme.h>
|
||||
#endif
|
||||
|
||||
/* Function prototypes */
|
||||
extern void sgemm_direct_sme1_preprocess(uint64_t nbr, uint64_t nbc,\
|
||||
const float * restrict a, float * a_mod) __asm__("sgemm_direct_sme1_preprocess");
|
||||
|
||||
/* Function Definitions */
|
||||
static uint64_t sve_cntw() {
|
||||
uint64_t cnt;
|
||||
@@ -99,10 +112,11 @@ kernel_2x2(const float *A, const float *B, float *C, size_t shared_dim,
|
||||
svst1_hor_za32(/*tile*/2, /*slice*/i, pg_c_0, &C[i * ldc]);
|
||||
svst1_hor_za32(/*tile*/3, /*slice*/i, pg_c_1, &C[i * ldc + svl]);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
__arm_new("za") __arm_locally_streaming
|
||||
void sgemm_direct_alpha_beta_sme1_2VLx2VL(uint64_t m, uint64_t k, uint64_t n, const float* alpha,\
|
||||
void SME1_KERNEL2X2(uint64_t m, uint64_t k, uint64_t n, const float* alpha,\
|
||||
const float *ba, const float *restrict bb, const float* beta,\
|
||||
float *restrict C) {
|
||||
|
||||
@@ -125,6 +139,7 @@ void sgemm_direct_alpha_beta_sme1_2VLx2VL(uint64_t m, uint64_t k, uint64_t n, co
|
||||
// Block over row dimension of C
|
||||
for (; row_idx < num_rows; row_idx += row_batch) {
|
||||
row_batch = MIN(row_batch, num_rows - row_idx);
|
||||
|
||||
uint64_t col_idx = 0;
|
||||
uint64_t col_batch = 2*svl;
|
||||
|
||||
@@ -141,9 +156,9 @@ void sgemm_direct_alpha_beta_sme1_2VLx2VL(uint64_t m, uint64_t k, uint64_t n, co
|
||||
}
|
||||
|
||||
#else
|
||||
void sgemm_direct_alpha_beta_sme1_2VLx2VL(uint64_t m, uint64_t k, uint64_t n, const float* alpha,\
|
||||
void SME1_KERNEL2X2(uint64_t m, uint64_t k, uint64_t n, const float* alpha,\
|
||||
const float *ba, const float *restrict bb, const float* beta,\
|
||||
float *restrict C){}
|
||||
float *restrict C){fprintf(stderr,"empty sgemm_alpha_beta2x2 should never get called!!!\n");}
|
||||
#endif
|
||||
|
||||
/*void sgemm_kernel_direct (BLASLONG M, BLASLONG N, BLASLONG K,\
|
||||
@@ -166,7 +181,7 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float alpha, float * __restrict
|
||||
* of reading directly from vector (z) registers.
|
||||
* */
|
||||
asm volatile("" : : :"p0", "p1", "p2", "p3", "p4", "p5", "p6", "p7",
|
||||
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15",
|
||||
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15", "d8", "d9", "d10", "d11", "d12", "d13", "d14", "d15",
|
||||
"z0", "z1", "z2", "z3", "z4", "z5", "z6", "z7",
|
||||
"z8", "z9", "z10", "z11", "z12", "z13", "z14", "z15",
|
||||
"z16", "z17", "z18", "z19", "z20", "z21", "z22", "z23",
|
||||
@@ -175,17 +190,19 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float alpha, float * __restrict
|
||||
/* Pre-process the left matrix to make it suitable for
|
||||
matrix sum of outer-product calculation
|
||||
*/
|
||||
sgemm_direct_sme1_preprocess(M, K, A, A_mod);
|
||||
|
||||
SME1_PREPROCESS(M, K, A, A_mod);
|
||||
|
||||
asm volatile("" : : :"p0", "p1", "p2", "p3", "p4", "p5", "p6", "p7",
|
||||
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15",
|
||||
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15","d8", "d9", "d10", "d11", "d12", "d13", "d14", "d15",
|
||||
"z0", "z1", "z2", "z3", "z4", "z5", "z6", "z7",
|
||||
"z8", "z9", "z10", "z11", "z12", "z13", "z14", "z15",
|
||||
"z16", "z17", "z18", "z19", "z20", "z21", "z22", "z23",
|
||||
"z24", "z25", "z26", "z27", "z28", "z29", "z30", "z31");
|
||||
|
||||
/* Calculate C = alpha*A*B + beta*C */
|
||||
sgemm_direct_alpha_beta_sme1_2VLx2VL(M, K, N, &alpha, A_mod, B, &beta, R);
|
||||
|
||||
SME1_KERNEL2X2(M, K, N, &alpha, A_mod, B, &beta, R);
|
||||
|
||||
free(A_mod);
|
||||
}
|
||||
@@ -194,6 +211,7 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float alpha, float * __restrict
|
||||
|
||||
void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float alpha, float * __restrict A,\
|
||||
BLASLONG strideA, float * __restrict B, BLASLONG strideB ,\
|
||||
float beta, float * __restrict R, BLASLONG strideR){}
|
||||
|
||||
float beta, float * __restrict R, BLASLONG strideR){fprintf(stderr,"empty sgemm_direct_alpha_beta should not be called!!!\n");}
|
||||
#endif
|
||||
|
||||
|
||||
|
||||
@@ -7,18 +7,29 @@
|
||||
#include <stdlib.h>
|
||||
#include <inttypes.h>
|
||||
#include <math.h>
|
||||
#if defined(DYNAMIC_ARCH)
|
||||
#define COMBINE(a,b) a ## b
|
||||
#define COMBINE2(a,b) COMBINE(a,b)
|
||||
#define SME1_PREPROCESS_BASE sgemm_direct_sme1_preprocess
|
||||
#define SME1_PREPROCESS COMBINE2(SME1_PREPROCESS_BASE,TS)
|
||||
#define SME1_DIRECT2X2_BASE sgemm_direct_sme1_2VLx2VL
|
||||
#define SME1_DIRECT2X2 COMBINE2(SME1_DIRECT2X2_BASE,TS)
|
||||
#else
|
||||
#define SME1_PREPROCESS sgemm_direct_sme1_preprocess
|
||||
#define SME1_DIRECT2X2 sgemm_direct_sme1_2VLx2VL
|
||||
#endif
|
||||
#if defined(HAVE_SME)
|
||||
|
||||
/* Function prototypes */
|
||||
extern void sgemm_direct_sme1_preprocess(uint64_t nbr, uint64_t nbc,\
|
||||
const float * restrict a, float * a_mod) __asm__("sgemm_direct_sme1_preprocess");
|
||||
extern void sgemm_direct_sme1_2VLx2VL(uint64_t m, uint64_t k, uint64_t n,\
|
||||
extern void SME1_PREPROCESS(uint64_t nbr, uint64_t nbc,\
|
||||
const float * restrict a, float * a_mod) ;
|
||||
|
||||
extern void SME1_DIRECT2X2(uint64_t m, uint64_t k, uint64_t n,\
|
||||
const float * matLeft,\
|
||||
const float * restrict matRight,\
|
||||
const float * restrict matResult) __asm__("sgemm_direct_sme1_2VLx2VL");
|
||||
const float * restrict matResult) ;
|
||||
|
||||
/* Function Definitions */
|
||||
uint64_t sve_cntw() {
|
||||
static uint64_t sve_cntw() {
|
||||
uint64_t cnt;
|
||||
asm volatile(
|
||||
"rdsvl %[res], #1\n"
|
||||
@@ -39,7 +50,6 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float * __restrict A,\
|
||||
uint64_t m_mod, vl_elms;
|
||||
|
||||
vl_elms = sve_cntw();
|
||||
|
||||
m_mod = ceil((double)M/(double)vl_elms) * vl_elms;
|
||||
|
||||
float *A_mod = (float *) malloc(m_mod*K*sizeof(float));
|
||||
@@ -48,7 +58,7 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float * __restrict A,\
|
||||
* of reading directly from vector (z) registers.
|
||||
* */
|
||||
asm volatile("" : : :"p0", "p1", "p2", "p3", "p4", "p5", "p6", "p7",
|
||||
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15",
|
||||
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15", "d8", "d9", "d10", "d11", "d12", "d13", "d14", "d15",
|
||||
"z0", "z1", "z2", "z3", "z4", "z5", "z6", "z7",
|
||||
"z8", "z9", "z10", "z11", "z12", "z13", "z14", "z15",
|
||||
"z16", "z17", "z18", "z19", "z20", "z21", "z22", "z23",
|
||||
@@ -57,13 +67,13 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float * __restrict A,\
|
||||
/* Pre-process the left matrix to make it suitable for
|
||||
matrix sum of outer-product calculation
|
||||
*/
|
||||
sgemm_direct_sme1_preprocess(M, K, A, A_mod);
|
||||
SME1_PREPROCESS(M, K, A, A_mod);
|
||||
|
||||
/* Calculate C = A*B */
|
||||
sgemm_direct_sme1_2VLx2VL(M, K, N, A_mod, B, R);
|
||||
SME1_DIRECT2X2(M, K, N, A_mod, B, R);
|
||||
|
||||
asm volatile("" : : :"p0", "p1", "p2", "p3", "p4", "p5", "p6", "p7",
|
||||
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15",
|
||||
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15", "d8", "d9", "d10", "d11", "d12", "d13", "d14", "d15",
|
||||
"z0", "z1", "z2", "z3", "z4", "z5", "z6", "z7",
|
||||
"z8", "z9", "z10", "z11", "z12", "z13", "z14", "z15",
|
||||
"z16", "z17", "z18", "z19", "z20", "z21", "z22", "z23",
|
||||
@@ -75,6 +85,16 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float * __restrict A,\
|
||||
|
||||
void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float * __restrict A,\
|
||||
BLASLONG strideA, float * __restrict B, BLASLONG strideB ,\
|
||||
float * __restrict R, BLASLONG strideR){}
|
||||
|
||||
float * __restrict R, BLASLONG strideR){
|
||||
fprintf(stderr,"EMPTY sgemm_kernel_direct should never be called \n");
|
||||
}
|
||||
void SME1_DIRECT2X2( uint64_t M , uint64_t K, uint64_t N,\
|
||||
const float * restrict A_base,\
|
||||
const float * restrict B_base,\
|
||||
const float * restrict C_base){};
|
||||
void SME1_PREPROCESS(uint64_t nbr, uint64_t nbc,\
|
||||
const float * restrict a, float * a_mod){};
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
#include "common.h"
|
||||
/* helper for the direct sgemm code adapted from Arjan van der Ven's x86_64 version */
|
||||
|
||||
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K)
|
||||
{
|
||||
if (M<3) return 0;
|
||||
unsigned long long mnk = M * N * K;
|
||||
/* benchmark performance on M4 peaks around 512 and crosses the graph of the NEON SGEMM at about 3100 */
|
||||
if (mnk >= 3100L * 3100L * 3100L)
|
||||
return 0;
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
@@ -35,16 +35,17 @@
|
||||
#define K_exit x15 //Exit condition for K loop
|
||||
#define M_cntr x16 //M loop counter
|
||||
#define C1 x17 //Constant1: N*(SVLs+1);SVLs-No. of 32-bit elements
|
||||
#define C2 x18 //Constant2: N + SVLs
|
||||
#define C3 x19 //Constant3: K*SVLs + SVLs
|
||||
#define C4 x20 //Constant4: SVLs-2
|
||||
#define C5 x21 //Constant5: K*SVLs
|
||||
#define C6 x22 //Constant6: N*SVLs
|
||||
#define C2 x19 //Constant2: N + SVLs
|
||||
#define C3 x20 //Constant3: K*SVLs + SVLs
|
||||
#define C4 x21 //Constant4: SVLs-2
|
||||
#define C5 x22 //Constant5: K*SVLs
|
||||
#define C6 x23 //Constant6: N*SVLs
|
||||
|
||||
.text
|
||||
.global sgemm_direct_sme1_2VLx2VL
|
||||
.global ASMNAME
|
||||
|
||||
sgemm_direct_sme1_2VLx2VL:
|
||||
ASMNAME:
|
||||
//sgemm_direct_sme1_2VLx2VL:
|
||||
|
||||
stp x19, x20, [sp, #-48]!
|
||||
stp x21, x22, [sp, #16]
|
||||
@@ -61,7 +62,7 @@
|
||||
add C2, N, C4 //N + SVLs
|
||||
add C3, C5, C4 //K*SVLs + SVLs
|
||||
whilelt p2.s, M_cntr, M //Tile 0,1 predicate (M dimension)
|
||||
sub w20, w20, #2 //SVLs-2
|
||||
sub w21, w21, #2 //SVLs-2
|
||||
|
||||
.M_Loop:
|
||||
incw M_cntr
|
||||
@@ -198,7 +199,7 @@ process_K_less_than_equal_2:
|
||||
st1w {za1h.s[w13, #0]}, p5, [Cptr1]
|
||||
st1w {za2h.s[w13, #0]}, p6, [Cptr0, C6, lsl #2]
|
||||
st1w {za3h.s[w13, #0]}, p7, [Cptr1, C6, lsl #2]
|
||||
cmp w13, w20
|
||||
cmp w13, w21
|
||||
b.mi .Loop_store_ZA
|
||||
psel p4, p0, p2.s[w13, 1]
|
||||
psel p5, p1, p2.s[w13, 1]
|
||||
@@ -211,12 +212,12 @@ process_K_less_than_equal_2:
|
||||
addvl Cptr, Cptr, #2
|
||||
addvl Bptr, Bptr, #1
|
||||
whilelt p0.b, Bptr, N_exit //1st Tile predicate (N dimension)
|
||||
b.first .N_Loop
|
||||
b.mi .N_Loop
|
||||
add A_base, A_base, C5, lsl #3 //A_base += 2*K*SVLs FP32 elements
|
||||
add C_base, C_base, C6, lsl #3 //C_base += 2*N*SVLs FP32 elements
|
||||
incw M_cntr
|
||||
whilelt p2.s, M_cntr, M //1st Tile predicate (M dimension)
|
||||
b.first .M_Loop
|
||||
b.mi .M_Loop
|
||||
|
||||
smstop
|
||||
|
||||
@@ -37,9 +37,9 @@
|
||||
#define C6 x15 //Constant6: 3*ncol
|
||||
|
||||
.text
|
||||
.global sgemm_direct_sme1_preprocess
|
||||
.global ASMNAME //sgemm_direct_sme1_preprocess
|
||||
|
||||
sgemm_direct_sme1_preprocess:
|
||||
ASMNAME: //sgemm_direct_sme1_preprocess:
|
||||
|
||||
stp x19, x20, [sp, #-48]!
|
||||
stp x21, x22, [sp, #16]
|
||||
@@ -114,14 +114,14 @@
|
||||
|
||||
addvl mat_ptr0, mat_ptr0, #1 //mat_ptr0 += SVLb
|
||||
whilelt p8.b, mat_ptr0, inner_loop_exit
|
||||
b.first .Loop_process
|
||||
b.mi .Loop_process
|
||||
|
||||
add mat_mod, mat_mod, C3, lsl #2 //mat_mod+=SVLs*nbc FP32 elements
|
||||
add mat, mat, C3, lsl #2 //mat+=SVLs*nbc FP32 elements
|
||||
incw outer_loop_cntr
|
||||
|
||||
whilelt p0.s, outer_loop_cntr, nrow
|
||||
b.first .M_Loop
|
||||
b.mi .M_Loop
|
||||
|
||||
smstop
|
||||
|
||||
|
||||
@@ -0,0 +1,887 @@
|
||||
/***************************************************************************
|
||||
* Copyright (c) 2026 The OpenBLAS Project
|
||||
* All rights reserved.
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are
|
||||
* met:
|
||||
* 1. Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* 2. Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in
|
||||
* the documentation and/or other materials provided with the
|
||||
* distribution.
|
||||
* 3. Neither the name of the OpenBLAS project nor the names of
|
||||
* its contributors may be used to endorse or promote products
|
||||
* derived from this software without specific prior written permission.
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
* ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
* POSSIBILITY OF SUCH DAMAGE.
|
||||
* *****************************************************************************/
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
#include "common.h"
|
||||
|
||||
static inline void kernel_8x8(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
float32x4_t c0_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c0_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c1_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c1_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c2_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c2_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c3_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c3_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c4_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c4_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c5_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c5_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c6_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c6_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c7_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c7_high = vdupq_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float16x8_t a_f16 = vld1q_f16(A);
|
||||
float32x4_t a_low = vcvt_f32_f16(vget_low_f16(a_f16));
|
||||
float32x4_t a_high = vcvt_f32_f16(vget_high_f16(a_f16));
|
||||
|
||||
float16x8_t b_f16 = vld1q_f16(B);
|
||||
float32x4_t b_low = vcvt_f32_f16(vget_low_f16(b_f16));
|
||||
float32x4_t b_high = vcvt_f32_f16(vget_high_f16(b_f16));
|
||||
|
||||
float32_t b0_lane0 = vgetq_lane_f32(b_low, 0);
|
||||
c0_low = vfmaq_n_f32(c0_low, a_low, b0_lane0);
|
||||
c0_high = vfmaq_n_f32(c0_high, a_high, b0_lane0);
|
||||
|
||||
float32_t b0_lane1 = vgetq_lane_f32(b_low, 1);
|
||||
c1_low = vfmaq_n_f32(c1_low, a_low, b0_lane1);
|
||||
c1_high = vfmaq_n_f32(c1_high, a_high, b0_lane1);
|
||||
|
||||
float32_t b0_lane2 = vgetq_lane_f32(b_low, 2);
|
||||
c2_low = vfmaq_n_f32(c2_low, a_low, b0_lane2);
|
||||
c2_high = vfmaq_n_f32(c2_high, a_high, b0_lane2);
|
||||
|
||||
float32_t b0_lane3 = vgetq_lane_f32(b_low, 3);
|
||||
c3_low = vfmaq_n_f32(c3_low, a_low, b0_lane3);
|
||||
c3_high = vfmaq_n_f32(c3_high, a_high, b0_lane3);
|
||||
|
||||
float32_t b1_lane0 = vgetq_lane_f32(b_high, 0);
|
||||
c4_low = vfmaq_n_f32(c4_low, a_low, b1_lane0);
|
||||
c4_high = vfmaq_n_f32(c4_high, a_high, b1_lane0);
|
||||
|
||||
float32_t b1_lane1 = vgetq_lane_f32(b_high, 1);
|
||||
c5_low = vfmaq_n_f32(c5_low, a_low, b1_lane1);
|
||||
c5_high = vfmaq_n_f32(c5_high, a_high, b1_lane1);
|
||||
|
||||
float32_t b1_lane2 = vgetq_lane_f32(b_high, 2);
|
||||
c6_low = vfmaq_n_f32(c6_low, a_low, b1_lane2);
|
||||
c6_high = vfmaq_n_f32(c6_high, a_high, b1_lane2);
|
||||
|
||||
float32_t b1_lane3 = vgetq_lane_f32(b_high, 3);
|
||||
c7_low = vfmaq_n_f32(c7_low, a_low, b1_lane3);
|
||||
c7_high = vfmaq_n_f32(c7_high, a_high, b1_lane3);
|
||||
|
||||
A += 8;
|
||||
B += 8;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C + 0 * ldc;
|
||||
FLOAT *col_1 = C + 1 * ldc;
|
||||
FLOAT *col_2 = C + 2 * ldc;
|
||||
FLOAT *col_3 = C + 3 * ldc;
|
||||
FLOAT *col_4 = C + 4 * ldc;
|
||||
FLOAT *col_5 = C + 5 * ldc;
|
||||
FLOAT *col_6 = C + 6 * ldc;
|
||||
FLOAT *col_7 = C + 7 * ldc;
|
||||
|
||||
float32x4_t t0_l = vld1q_f32(col_0);
|
||||
float32x4_t t0_h = vld1q_f32(col_0 + 4);
|
||||
t0_l = vaddq_f32(t0_l, vmulq_n_f32(c0_low, alpha));
|
||||
t0_h = vaddq_f32(t0_h, vmulq_n_f32(c0_high, alpha));
|
||||
vst1q_f32(col_0, t0_l);
|
||||
vst1q_f32(col_0 + 4, t0_h);
|
||||
|
||||
float32x4_t t1_l = vld1q_f32(col_1);
|
||||
float32x4_t t1_h = vld1q_f32(col_1 + 4);
|
||||
t1_l = vaddq_f32(t1_l, vmulq_n_f32(c1_low, alpha));
|
||||
t1_h = vaddq_f32(t1_h, vmulq_n_f32(c1_high, alpha));
|
||||
vst1q_f32(col_1, t1_l);
|
||||
vst1q_f32(col_1 + 4, t1_h);
|
||||
|
||||
float32x4_t t2_l = vld1q_f32(col_2);
|
||||
float32x4_t t2_h = vld1q_f32(col_2 + 4);
|
||||
t2_l = vaddq_f32(t2_l, vmulq_n_f32(c2_low, alpha));
|
||||
t2_h = vaddq_f32(t2_h, vmulq_n_f32(c2_high, alpha));
|
||||
vst1q_f32(col_2, t2_l);
|
||||
vst1q_f32(col_2 + 4, t2_h);
|
||||
|
||||
float32x4_t t3_l = vld1q_f32(col_3);
|
||||
float32x4_t t3_h = vld1q_f32(col_3 + 4);
|
||||
t3_l = vaddq_f32(t3_l, vmulq_n_f32(c3_low, alpha));
|
||||
t3_h = vaddq_f32(t3_h, vmulq_n_f32(c3_high, alpha));
|
||||
vst1q_f32(col_3, t3_l);
|
||||
vst1q_f32(col_3 + 4, t3_h);
|
||||
|
||||
float32x4_t t4_l = vld1q_f32(col_4);
|
||||
float32x4_t t4_h = vld1q_f32(col_4 + 4);
|
||||
t4_l = vaddq_f32(t4_l, vmulq_n_f32(c4_low, alpha));
|
||||
t4_h = vaddq_f32(t4_h, vmulq_n_f32(c4_high, alpha));
|
||||
vst1q_f32(col_4, t4_l);
|
||||
vst1q_f32(col_4 + 4, t4_h);
|
||||
|
||||
float32x4_t t5_l = vld1q_f32(col_5);
|
||||
float32x4_t t5_h = vld1q_f32(col_5 + 4);
|
||||
t5_l = vaddq_f32(t5_l, vmulq_n_f32(c5_low, alpha));
|
||||
t5_h = vaddq_f32(t5_h, vmulq_n_f32(c5_high, alpha));
|
||||
vst1q_f32(col_5, t5_l);
|
||||
vst1q_f32(col_5 + 4, t5_h);
|
||||
|
||||
float32x4_t t6_l = vld1q_f32(col_6);
|
||||
float32x4_t t6_h = vld1q_f32(col_6 + 4);
|
||||
t6_l = vaddq_f32(t6_l, vmulq_n_f32(c6_low, alpha));
|
||||
t6_h = vaddq_f32(t6_h, vmulq_n_f32(c6_high, alpha));
|
||||
vst1q_f32(col_6, t6_l);
|
||||
vst1q_f32(col_6 + 4, t6_h);
|
||||
|
||||
float32x4_t t7_l = vld1q_f32(col_7);
|
||||
float32x4_t t7_h = vld1q_f32(col_7 + 4);
|
||||
t7_l = vaddq_f32(t7_l, vmulq_n_f32(c7_low, alpha));
|
||||
t7_h = vaddq_f32(t7_h, vmulq_n_f32(c7_high, alpha));
|
||||
vst1q_f32(col_7, t7_l);
|
||||
vst1q_f32(col_7 + 4, t7_h);
|
||||
}
|
||||
|
||||
static inline void kernel_4x8(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
float32x4_t c0 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c1 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c2 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c3 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c4 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c5 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c6 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c7 = vdupq_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float32x4_t a_f16 = vcvt_f32_f16(vld1_f16(A));
|
||||
|
||||
float16x8_t b_f16 = vld1q_f16(B);
|
||||
float32x4_t b_low = vcvt_f32_f16(vget_low_f16(b_f16));
|
||||
float32x4_t b_high = vcvt_f32_f16(vget_high_f16(b_f16));
|
||||
|
||||
float32_t b0_lane0 = vgetq_lane_f32(b_low, 0);
|
||||
c0 = vfmaq_n_f32(c0, a_f16, b0_lane0);
|
||||
|
||||
float32_t b0_lane1 = vgetq_lane_f32(b_low, 1);
|
||||
c1 = vfmaq_n_f32(c1, a_f16, b0_lane1);
|
||||
|
||||
float32_t b0_lane2 = vgetq_lane_f32(b_low, 2);
|
||||
c2 = vfmaq_n_f32(c2, a_f16, b0_lane2);
|
||||
|
||||
float32_t b0_lane3 = vgetq_lane_f32(b_low, 3);
|
||||
c3 = vfmaq_n_f32(c3, a_f16, b0_lane3);
|
||||
|
||||
float32_t b1_lane0 = vgetq_lane_f32(b_high, 0);
|
||||
c4 = vfmaq_n_f32(c4, a_f16, b1_lane0);
|
||||
|
||||
float32_t b1_lane1 = vgetq_lane_f32(b_high, 1);
|
||||
c5 = vfmaq_n_f32(c5, a_f16, b1_lane1);
|
||||
|
||||
float32_t b1_lane2 = vgetq_lane_f32(b_high, 2);
|
||||
c6 = vfmaq_n_f32(c6, a_f16, b1_lane2);
|
||||
|
||||
float32_t b1_lane3 = vgetq_lane_f32(b_high, 3);
|
||||
c7 = vfmaq_n_f32(c7, a_f16, b1_lane3);
|
||||
|
||||
A += 4;
|
||||
B += 8;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C + 0 * ldc;
|
||||
FLOAT *col_1 = C + 1 * ldc;
|
||||
FLOAT *col_2 = C + 2 * ldc;
|
||||
FLOAT *col_3 = C + 3 * ldc;
|
||||
FLOAT *col_4 = C + 4 * ldc;
|
||||
FLOAT *col_5 = C + 5 * ldc;
|
||||
FLOAT *col_6 = C + 6 * ldc;
|
||||
FLOAT *col_7 = C + 7 * ldc;
|
||||
|
||||
float32x4_t t0 = vld1q_f32(col_0);
|
||||
t0 = vaddq_f32(t0, vmulq_n_f32(c0, alpha));
|
||||
vst1q_f32(col_0, t0);
|
||||
|
||||
float32x4_t t1 = vld1q_f32(col_1);
|
||||
t1 = vaddq_f32(t1, vmulq_n_f32(c1, alpha));
|
||||
vst1q_f32(col_1, t1);
|
||||
|
||||
float32x4_t t2 = vld1q_f32(col_2);
|
||||
t2 = vaddq_f32(t2, vmulq_n_f32(c2, alpha));
|
||||
vst1q_f32(col_2, t2);
|
||||
|
||||
float32x4_t t3 = vld1q_f32(col_3);
|
||||
t3 = vaddq_f32(t3, vmulq_n_f32(c3, alpha));
|
||||
vst1q_f32(col_3, t3);
|
||||
|
||||
float32x4_t t4 = vld1q_f32(col_4);
|
||||
t4 = vaddq_f32(t4, vmulq_n_f32(c4, alpha));
|
||||
vst1q_f32(col_4, t4);
|
||||
|
||||
float32x4_t t5 = vld1q_f32(col_5);
|
||||
t5 = vaddq_f32(t5, vmulq_n_f32(c5, alpha));
|
||||
vst1q_f32(col_5, t5);
|
||||
|
||||
float32x4_t t6 = vld1q_f32(col_6);
|
||||
t6 = vaddq_f32(t6, vmulq_n_f32(c6, alpha));
|
||||
vst1q_f32(col_6, t6);
|
||||
|
||||
float32x4_t t7 = vld1q_f32(col_7);
|
||||
t7 = vaddq_f32(t7, vmulq_n_f32(c7, alpha));
|
||||
vst1q_f32(col_7, t7);
|
||||
}
|
||||
|
||||
static inline void kernel_2x8(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
float32x2_t c0 = vdup_n_f32(0.0f);
|
||||
float32x2_t c1 = vdup_n_f32(0.0f);
|
||||
float32x2_t c2 = vdup_n_f32(0.0f);
|
||||
float32x2_t c3 = vdup_n_f32(0.0f);
|
||||
float32x2_t c4 = vdup_n_f32(0.0f);
|
||||
float32x2_t c5 = vdup_n_f32(0.0f);
|
||||
float32x2_t c6 = vdup_n_f32(0.0f);
|
||||
float32x2_t c7 = vdup_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
|
||||
float32x2_t a_low = vget_low_f32(a_f32);
|
||||
|
||||
float16x8_t b_f16 = vld1q_f16(B);
|
||||
float32x4_t b_low = vcvt_f32_f16(vget_low_f16(b_f16));
|
||||
float32x4_t b_high = vcvt_f32_f16(vget_high_f16(b_f16));
|
||||
|
||||
float32_t b0_lane0 = vgetq_lane_f32(b_low, 0);
|
||||
c0 = vfma_n_f32(c0, a_low, b0_lane0);
|
||||
|
||||
float32_t b0_lane1 = vgetq_lane_f32(b_low, 1);
|
||||
c1 = vfma_n_f32(c1, a_low, b0_lane1);
|
||||
|
||||
float32_t b0_lane2 = vgetq_lane_f32(b_low, 2);
|
||||
c2 = vfma_n_f32(c2, a_low, b0_lane2);
|
||||
|
||||
float32_t b0_lane3 = vgetq_lane_f32(b_low, 3);
|
||||
c3 = vfma_n_f32(c3, a_low, b0_lane3);
|
||||
|
||||
float32_t b1_lane0 = vgetq_lane_f32(b_high, 0);
|
||||
c4 = vfma_n_f32(c4, a_low, b1_lane0);
|
||||
|
||||
float32_t b1_lane1 = vgetq_lane_f32(b_high, 1);
|
||||
c5 = vfma_n_f32(c5, a_low, b1_lane1);
|
||||
|
||||
float32_t b1_lane2 = vgetq_lane_f32(b_high, 2);
|
||||
c6 = vfma_n_f32(c6, a_low, b1_lane2);
|
||||
|
||||
float32_t b1_lane3 = vgetq_lane_f32(b_high, 3);
|
||||
c7 = vfma_n_f32(c7, a_low, b1_lane3);
|
||||
|
||||
A += 2;
|
||||
B += 8;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C + 0 * ldc;
|
||||
FLOAT *col_1 = C + 1 * ldc;
|
||||
FLOAT *col_2 = C + 2 * ldc;
|
||||
FLOAT *col_3 = C + 3 * ldc;
|
||||
FLOAT *col_4 = C + 4 * ldc;
|
||||
FLOAT *col_5 = C + 5 * ldc;
|
||||
FLOAT *col_6 = C + 6 * ldc;
|
||||
FLOAT *col_7 = C + 7 * ldc;
|
||||
|
||||
float32x2_t t0 = vld1_f32(col_0);
|
||||
t0 = vadd_f32(t0, vmul_n_f32(c0, alpha));
|
||||
vst1_f32(col_0, t0);
|
||||
|
||||
float32x2_t t1 = vld1_f32(col_1);
|
||||
t1 = vadd_f32(t1, vmul_n_f32(c1, alpha));
|
||||
vst1_f32(col_1, t1);
|
||||
|
||||
float32x2_t t2 = vld1_f32(col_2);
|
||||
t2 = vadd_f32(t2, vmul_n_f32(c2, alpha));
|
||||
vst1_f32(col_2, t2);
|
||||
|
||||
float32x2_t t3 = vld1_f32(col_3);
|
||||
t3 = vadd_f32(t3, vmul_n_f32(c3, alpha));
|
||||
vst1_f32(col_3, t3);
|
||||
|
||||
float32x2_t t4 = vld1_f32(col_4);
|
||||
t4 = vadd_f32(t4, vmul_n_f32(c4, alpha));
|
||||
vst1_f32(col_4, t4);
|
||||
|
||||
float32x2_t t5 = vld1_f32(col_5);
|
||||
t5 = vadd_f32(t5, vmul_n_f32(c5, alpha));
|
||||
vst1_f32(col_5, t5);
|
||||
|
||||
float32x2_t t6 = vld1_f32(col_6);
|
||||
t6 = vadd_f32(t6, vmul_n_f32(c6, alpha));
|
||||
vst1_f32(col_6, t6);
|
||||
|
||||
float32x2_t t7 = vld1_f32(col_7);
|
||||
t7 = vadd_f32(t7, vmul_n_f32(c7, alpha));
|
||||
vst1_f32(col_7, t7);
|
||||
}
|
||||
|
||||
static inline void kernel_1x8(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
FLOAT c0 = 0, c1 = 0, c2 = 0, c3 = 0, c4 = 0, c5 = 0, c6 = 0, c7 = 0;
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
FLOAT a = A[0];
|
||||
c0 += a * B[0];
|
||||
c1 += a * B[1];
|
||||
c2 += a * B[2];
|
||||
c3 += a * B[3];
|
||||
c4 += a * B[4];
|
||||
c5 += a * B[5];
|
||||
c6 += a * B[6];
|
||||
c7 += a * B[7];
|
||||
|
||||
A += 1;
|
||||
B += 8;
|
||||
}
|
||||
|
||||
C[0 * ldc] += alpha * c0;
|
||||
C[1 * ldc] += alpha * c1;
|
||||
C[2 * ldc] += alpha * c2;
|
||||
C[3 * ldc] += alpha * c3;
|
||||
C[4 * ldc] += alpha * c4;
|
||||
C[5 * ldc] += alpha * c5;
|
||||
C[6 * ldc] += alpha * c6;
|
||||
C[7 * ldc] += alpha * c7;
|
||||
}
|
||||
|
||||
static inline void kernel_8x4(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
float32x4_t c0_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c0_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c1_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c1_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c2_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c2_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c3_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c3_high = vdupq_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float16x8_t a_f16 = vld1q_f16(A);
|
||||
float32x4_t a_low = vcvt_f32_f16(vget_low_f16(a_f16));
|
||||
float32x4_t a_high = vcvt_f32_f16(vget_high_f16(a_f16));
|
||||
|
||||
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
|
||||
|
||||
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
|
||||
c0_low = vfmaq_n_f32(c0_low, a_low, b0_lane0);
|
||||
c0_high = vfmaq_n_f32(c0_high, a_high, b0_lane0);
|
||||
|
||||
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
|
||||
c1_low = vfmaq_n_f32(c1_low, a_low, b0_lane1);
|
||||
c1_high = vfmaq_n_f32(c1_high, a_high, b0_lane1);
|
||||
|
||||
float32_t b0_lane2 = vgetq_lane_f32(b_f32, 2);
|
||||
c2_low = vfmaq_n_f32(c2_low, a_low, b0_lane2);
|
||||
c2_high = vfmaq_n_f32(c2_high, a_high, b0_lane2);
|
||||
|
||||
float32_t b0_lane3 = vgetq_lane_f32(b_f32, 3);
|
||||
c3_low = vfmaq_n_f32(c3_low, a_low, b0_lane3);
|
||||
c3_high = vfmaq_n_f32(c3_high, a_high, b0_lane3);
|
||||
|
||||
A += 8;
|
||||
B += 4;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C + 0 * ldc;
|
||||
FLOAT *col_1 = C + 1 * ldc;
|
||||
FLOAT *col_2 = C + 2 * ldc;
|
||||
FLOAT *col_3 = C + 3 * ldc;
|
||||
|
||||
float32x4_t t0_l = vld1q_f32(col_0);
|
||||
float32x4_t t0_h = vld1q_f32(col_0 + 4);
|
||||
t0_l = vaddq_f32(t0_l, vmulq_n_f32(c0_low, alpha));
|
||||
t0_h = vaddq_f32(t0_h, vmulq_n_f32(c0_high, alpha));
|
||||
vst1q_f32(col_0, t0_l);
|
||||
vst1q_f32(col_0 + 4, t0_h);
|
||||
|
||||
float32x4_t t1_l = vld1q_f32(col_1);
|
||||
float32x4_t t1_h = vld1q_f32(col_1 + 4);
|
||||
t1_l = vaddq_f32(t1_l, vmulq_n_f32(c1_low, alpha));
|
||||
t1_h = vaddq_f32(t1_h, vmulq_n_f32(c1_high, alpha));
|
||||
vst1q_f32(col_1, t1_l);
|
||||
vst1q_f32(col_1 + 4, t1_h);
|
||||
|
||||
float32x4_t t2_l = vld1q_f32(col_2);
|
||||
float32x4_t t2_h = vld1q_f32(col_2 + 4);
|
||||
t2_l = vaddq_f32(t2_l, vmulq_n_f32(c2_low, alpha));
|
||||
t2_h = vaddq_f32(t2_h, vmulq_n_f32(c2_high, alpha));
|
||||
vst1q_f32(col_2, t2_l);
|
||||
vst1q_f32(col_2 + 4, t2_h);
|
||||
|
||||
float32x4_t t3_l = vld1q_f32(col_3);
|
||||
float32x4_t t3_h = vld1q_f32(col_3 + 4);
|
||||
t3_l = vaddq_f32(t3_l, vmulq_n_f32(c3_low, alpha));
|
||||
t3_h = vaddq_f32(t3_h, vmulq_n_f32(c3_high, alpha));
|
||||
vst1q_f32(col_3, t3_l);
|
||||
vst1q_f32(col_3 + 4, t3_h);
|
||||
}
|
||||
|
||||
static inline void kernel_4x4(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
float32x4_t c0 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c1 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c2 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c3 = vdupq_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
|
||||
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
|
||||
|
||||
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
|
||||
c0 = vfmaq_n_f32(c0, a_f32, b0_lane0);
|
||||
|
||||
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
|
||||
c1 = vfmaq_n_f32(c1, a_f32, b0_lane1);
|
||||
|
||||
float32_t b0_lane2 = vgetq_lane_f32(b_f32, 2);
|
||||
c2 = vfmaq_n_f32(c2, a_f32, b0_lane2);
|
||||
|
||||
float32_t b0_lane3 = vgetq_lane_f32(b_f32, 3);
|
||||
c3 = vfmaq_n_f32(c3, a_f32, b0_lane3);
|
||||
|
||||
A += 4;
|
||||
B += 4;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C + 0 * ldc;
|
||||
FLOAT *col_1 = C + 1 * ldc;
|
||||
FLOAT *col_2 = C + 2 * ldc;
|
||||
FLOAT *col_3 = C + 3 * ldc;
|
||||
|
||||
float32x4_t t0 = vld1q_f32(col_0);
|
||||
t0 = vaddq_f32(t0, vmulq_n_f32(c0, alpha));
|
||||
vst1q_f32(col_0, t0);
|
||||
|
||||
float32x4_t t1 = vld1q_f32(col_1);
|
||||
t1 = vaddq_f32(t1, vmulq_n_f32(c1, alpha));
|
||||
vst1q_f32(col_1, t1);
|
||||
|
||||
float32x4_t t2 = vld1q_f32(col_2);
|
||||
t2 = vaddq_f32(t2, vmulq_n_f32(c2, alpha));
|
||||
vst1q_f32(col_2, t2);
|
||||
|
||||
float32x4_t t3 = vld1q_f32(col_3);
|
||||
t3 = vaddq_f32(t3, vmulq_n_f32(c3, alpha));
|
||||
vst1q_f32(col_3, t3);
|
||||
}
|
||||
|
||||
static inline void kernel_2x4(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
float32x2_t c0 = vdup_n_f32(0.0f);
|
||||
float32x2_t c1 = vdup_n_f32(0.0f);
|
||||
float32x2_t c2 = vdup_n_f32(0.0f);
|
||||
float32x2_t c3 = vdup_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
|
||||
float32x2_t a_low = vget_low_f32(a_f32);
|
||||
|
||||
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
|
||||
|
||||
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
|
||||
c0 = vfma_n_f32(c0, a_low, b0_lane0);
|
||||
|
||||
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
|
||||
c1 = vfma_n_f32(c1, a_low, b0_lane1);
|
||||
|
||||
float32_t b0_lane2 = vgetq_lane_f32(b_f32, 2);
|
||||
c2 = vfma_n_f32(c2, a_low, b0_lane2);
|
||||
|
||||
float32_t b0_lane3 = vgetq_lane_f32(b_f32, 3);
|
||||
c3 = vfma_n_f32(c3, a_low, b0_lane3);
|
||||
A += 2;
|
||||
B += 4;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C + 0 * ldc;
|
||||
FLOAT *col_1 = C + 1 * ldc;
|
||||
FLOAT *col_2 = C + 2 * ldc;
|
||||
FLOAT *col_3 = C + 3 * ldc;
|
||||
|
||||
float32x2_t t0 = vld1_f32(col_0);
|
||||
t0 = vadd_f32(t0, vmul_n_f32(c0, alpha));
|
||||
vst1_f32(col_0, t0);
|
||||
|
||||
float32x2_t t1 = vld1_f32(col_1);
|
||||
t1 = vadd_f32(t1, vmul_n_f32(c1, alpha));
|
||||
vst1_f32(col_1, t1);
|
||||
|
||||
float32x2_t t2 = vld1_f32(col_2);
|
||||
t2 = vadd_f32(t2, vmul_n_f32(c2, alpha));
|
||||
vst1_f32(col_2, t2);
|
||||
|
||||
float32x2_t t3 = vld1_f32(col_3);
|
||||
t3 = vadd_f32(t3, vmul_n_f32(c3, alpha));
|
||||
vst1_f32(col_3, t3);
|
||||
}
|
||||
|
||||
static inline void kernel_1x4(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
FLOAT c0 = 0, c1 = 0, c2 = 0, c3 = 0;
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
FLOAT a = A[0];
|
||||
c0 += a * B[0];
|
||||
c1 += a * B[1];
|
||||
c2 += a * B[2];
|
||||
c3 += a * B[3];
|
||||
|
||||
A += 1;
|
||||
B += 4;
|
||||
}
|
||||
|
||||
C[0 * ldc] += alpha * c0;
|
||||
C[1 * ldc] += alpha * c1;
|
||||
C[2 * ldc] += alpha * c2;
|
||||
C[3 * ldc] += alpha * c3;
|
||||
}
|
||||
|
||||
static inline void kernel_8x2(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
float32x4_t c0_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c0_high = vdupq_n_f32(0.0f);
|
||||
float32x4_t c1_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c1_high = vdupq_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float16x8_t a_f16 = vld1q_f16(A);
|
||||
float32x4_t a_low = vcvt_f32_f16(vget_low_f16(a_f16));
|
||||
float32x4_t a_high = vcvt_f32_f16(vget_high_f16(a_f16));
|
||||
|
||||
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
|
||||
|
||||
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
|
||||
c0_low = vfmaq_n_f32(c0_low, a_low, b0_lane0);
|
||||
c0_high = vfmaq_n_f32(c0_high, a_high, b0_lane0);
|
||||
|
||||
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
|
||||
c1_low = vfmaq_n_f32(c1_low, a_low, b0_lane1);
|
||||
c1_high = vfmaq_n_f32(c1_high, a_high, b0_lane1);
|
||||
|
||||
A += 8;
|
||||
B += 2;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C + 0 * ldc;
|
||||
FLOAT *col_1 = C + 1 * ldc;
|
||||
|
||||
float32x4_t t0_l = vld1q_f32(col_0);
|
||||
float32x4_t t0_h = vld1q_f32(col_0 + 4);
|
||||
t0_l = vaddq_f32(t0_l, vmulq_n_f32(c0_low, alpha));
|
||||
t0_h = vaddq_f32(t0_h, vmulq_n_f32(c0_high, alpha));
|
||||
vst1q_f32(col_0, t0_l);
|
||||
vst1q_f32(col_0 + 4, t0_h);
|
||||
|
||||
float32x4_t t1_l = vld1q_f32(col_1);
|
||||
float32x4_t t1_h = vld1q_f32(col_1 + 4);
|
||||
t1_l = vaddq_f32(t1_l, vmulq_n_f32(c1_low, alpha));
|
||||
t1_h = vaddq_f32(t1_h, vmulq_n_f32(c1_high, alpha));
|
||||
vst1q_f32(col_1, t1_l);
|
||||
vst1q_f32(col_1 + 4, t1_h);
|
||||
}
|
||||
|
||||
static inline void kernel_4x2(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
float32x4_t c0 = vdupq_n_f32(0.0f);
|
||||
float32x4_t c1 = vdupq_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
|
||||
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
|
||||
|
||||
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
|
||||
c0 = vfmaq_n_f32(c0, a_f32, b0_lane0);
|
||||
|
||||
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
|
||||
c1 = vfmaq_n_f32(c1, a_f32, b0_lane1);
|
||||
|
||||
A += 4;
|
||||
B += 2;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C + 0 * ldc;
|
||||
FLOAT *col_1 = C + 1 * ldc;
|
||||
|
||||
float32x4_t t0 = vld1q_f32(col_0);
|
||||
t0 = vaddq_f32(t0, vmulq_n_f32(c0, alpha));
|
||||
vst1q_f32(col_0, t0);
|
||||
|
||||
float32x4_t t1 = vld1q_f32(col_1);
|
||||
t1 = vaddq_f32(t1, vmulq_n_f32(c1, alpha));
|
||||
vst1q_f32(col_1, t1);
|
||||
}
|
||||
|
||||
static inline void kernel_2x2(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
float32x2_t c0 = vdup_n_f32(0.0f);
|
||||
float32x2_t c1 = vdup_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
|
||||
float32x2_t a_low = vget_low_f32(a_f32);
|
||||
|
||||
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
|
||||
|
||||
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
|
||||
c0 = vfma_n_f32(c0, a_low, b0_lane0);
|
||||
|
||||
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
|
||||
c1 = vfma_n_f32(c1, a_low, b0_lane1);
|
||||
;
|
||||
|
||||
A += 2;
|
||||
B += 2;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C + 0 * ldc;
|
||||
FLOAT *col_1 = C + 1 * ldc;
|
||||
|
||||
float32x2_t t0 = vld1_f32(col_0);
|
||||
t0 = vadd_f32(t0, vmul_n_f32(c0, alpha));
|
||||
vst1_f32(col_0, t0);
|
||||
|
||||
float32x2_t t1 = vld1_f32(col_1);
|
||||
t1 = vadd_f32(t1, vmul_n_f32(c1, alpha));
|
||||
vst1_f32(col_1, t1);
|
||||
}
|
||||
|
||||
static inline void kernel_1x2(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
|
||||
FLOAT c0 = 0, c1 = 0;
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
FLOAT a = A[0];
|
||||
c0 += a * B[0];
|
||||
c1 += a * B[1];
|
||||
|
||||
A += 1;
|
||||
B += 2;
|
||||
}
|
||||
|
||||
C[0 * ldc] += alpha * c0;
|
||||
C[1 * ldc] += alpha * c1;
|
||||
}
|
||||
|
||||
static inline void kernel_8x1(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, FLOAT alpha) {
|
||||
float32x4_t c0_low = vdupq_n_f32(0.0f);
|
||||
float32x4_t c0_high = vdupq_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float16x8_t a_f16 = vld1q_f16(A);
|
||||
float32x4_t a_low = vcvt_f32_f16(vget_low_f16(a_f16));
|
||||
float32x4_t a_high = vcvt_f32_f16(vget_high_f16(a_f16));
|
||||
|
||||
float b_scalar = (float)B[0];
|
||||
|
||||
c0_low = vfmaq_n_f32(c0_low, a_low, b_scalar);
|
||||
c0_high = vfmaq_n_f32(c0_high, a_high, b_scalar);
|
||||
|
||||
A += 8;
|
||||
B += 1;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C;
|
||||
|
||||
float32x4_t t0_l = vld1q_f32(col_0);
|
||||
float32x4_t t0_h = vld1q_f32(col_0 + 4);
|
||||
t0_l = vaddq_f32(t0_l, vmulq_n_f32(c0_low, alpha));
|
||||
t0_h = vaddq_f32(t0_h, vmulq_n_f32(c0_high, alpha));
|
||||
vst1q_f32(col_0, t0_l);
|
||||
vst1q_f32(col_0 + 4, t0_h);
|
||||
}
|
||||
|
||||
static inline void kernel_4x1(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, FLOAT alpha) {
|
||||
float32x4_t c0 = vdupq_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
|
||||
float b_scalar = (float)B[0];
|
||||
c0 = vfmaq_n_f32(c0, a_f32, b_scalar);
|
||||
|
||||
A += 4;
|
||||
B += 1;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C;
|
||||
float32x4_t t0 = vld1q_f32(col_0);
|
||||
t0 = vaddq_f32(t0, vmulq_n_f32(c0, alpha));
|
||||
vst1q_f32(col_0, t0);
|
||||
}
|
||||
|
||||
static inline void kernel_2x1(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, FLOAT alpha) {
|
||||
float32x2_t c0 = vdup_n_f32(0.0f);
|
||||
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
|
||||
float32x2_t a_low = vget_low_f32(a_f32);
|
||||
|
||||
float b_scalar = (float)B[0];
|
||||
c0 = vfma_n_f32(c0, a_low, b_scalar);
|
||||
|
||||
A += 2;
|
||||
B += 1;
|
||||
}
|
||||
|
||||
FLOAT *col_0 = C;
|
||||
float32x2_t t0 = vld1_f32(col_0);
|
||||
t0 = vadd_f32(t0, vmul_n_f32(c0, alpha));
|
||||
vst1_f32(col_0, t0);
|
||||
}
|
||||
|
||||
static inline void kernel_1x1(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, FLOAT alpha) {
|
||||
FLOAT sum = 0.0f;
|
||||
for (BLASLONG k = 0; k < K; ++k) {
|
||||
sum += A[0] * B[0];
|
||||
A += 1;
|
||||
B += 1;
|
||||
}
|
||||
|
||||
C[0] += alpha * sum;
|
||||
}
|
||||
|
||||
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K, FLOAT alpha, IFLOAT *A, IFLOAT *B, FLOAT *C, BLASLONG ldc) {
|
||||
float16_t *A_base = (float16_t *)A;
|
||||
float16_t *B_base = (float16_t *)B;
|
||||
|
||||
FLOAT *Ccol = C;
|
||||
BLASLONG m_rem1, m_rem2, m_rem3, m_rem4;
|
||||
|
||||
while (N >= 8) {
|
||||
const float16_t *Aptr = A_base;
|
||||
const float16_t *Bptr = B_base;
|
||||
FLOAT *Crow = Ccol;
|
||||
|
||||
m_rem1 = M;
|
||||
|
||||
while (m_rem1 >= 8) {
|
||||
kernel_8x8(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
Aptr += K * 8;
|
||||
Crow += 8;
|
||||
m_rem1 -= 8;
|
||||
}
|
||||
if (m_rem1 >= 4) {
|
||||
kernel_4x8(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
Aptr += K * 4;
|
||||
Crow += 4;
|
||||
m_rem1 -= 4;
|
||||
}
|
||||
if (m_rem1 >= 2) {
|
||||
kernel_2x8(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
Aptr += K * 2;
|
||||
Crow += 2;
|
||||
m_rem1 -= 2;
|
||||
}
|
||||
if (m_rem1 >= 1) {
|
||||
kernel_1x8(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
}
|
||||
|
||||
B_base += K * 8;
|
||||
Ccol += ldc * 8;
|
||||
N -= 8;
|
||||
}
|
||||
|
||||
if (N >= 4) {
|
||||
const float16_t *Aptr = A_base;
|
||||
const float16_t *Bptr = B_base;
|
||||
FLOAT *Crow = Ccol;
|
||||
|
||||
m_rem2 = M;
|
||||
while (m_rem2 >= 8) {
|
||||
kernel_8x4(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
Aptr += K * 8;
|
||||
Crow += 8;
|
||||
m_rem2 -= 8;
|
||||
}
|
||||
if (m_rem2 >= 4) {
|
||||
kernel_4x4(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
Aptr += K * 4;
|
||||
Crow += 4;
|
||||
m_rem2 -= 4;
|
||||
}
|
||||
if (m_rem2 >= 2) {
|
||||
kernel_2x4(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
Aptr += K * 2;
|
||||
Crow += 2;
|
||||
m_rem2 -= 2;
|
||||
}
|
||||
if (m_rem2 >= 1) {
|
||||
kernel_1x4(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
}
|
||||
|
||||
B_base += K * 4;
|
||||
Ccol += ldc * 4;
|
||||
N -= 4;
|
||||
}
|
||||
|
||||
if (N >= 2) {
|
||||
const float16_t *Aptr = A_base;
|
||||
const float16_t *Bptr = B_base;
|
||||
FLOAT *Crow = Ccol;
|
||||
|
||||
m_rem3 = M;
|
||||
while (m_rem3 >= 8) {
|
||||
kernel_8x2(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
Aptr += K * 8;
|
||||
Crow += 8;
|
||||
m_rem3 -= 8;
|
||||
}
|
||||
if (m_rem3 >= 4) {
|
||||
kernel_4x2(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
Aptr += K * 4;
|
||||
Crow += 4;
|
||||
m_rem3 -= 4;
|
||||
}
|
||||
if (m_rem3 >= 2) {
|
||||
kernel_2x2(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
Aptr += K * 2;
|
||||
Crow += 2;
|
||||
m_rem3 -= 2;
|
||||
}
|
||||
if (m_rem3 >= 1) {
|
||||
kernel_1x2(K, Aptr, Bptr, Crow, ldc, alpha);
|
||||
}
|
||||
|
||||
B_base += K * 2;
|
||||
Ccol += ldc * 2;
|
||||
N -= 2;
|
||||
}
|
||||
|
||||
if (N >= 1) {
|
||||
const float16_t *Aptr = A_base;
|
||||
const float16_t *Bptr = B_base;
|
||||
FLOAT *Crow = Ccol;
|
||||
|
||||
m_rem4 = M;
|
||||
while (m_rem4 >= 8) {
|
||||
kernel_8x1(K, Aptr, Bptr, Crow, alpha);
|
||||
Aptr += K * 8;
|
||||
Crow += 8;
|
||||
m_rem4 -= 8;
|
||||
}
|
||||
if (m_rem4 >= 4) {
|
||||
kernel_4x1(K, Aptr, Bptr, Crow, alpha);
|
||||
Aptr += K * 4;
|
||||
Crow += 4;
|
||||
m_rem4 -= 4;
|
||||
}
|
||||
if (m_rem4 >= 2) {
|
||||
kernel_2x1(K, Aptr, Bptr, Crow, alpha);
|
||||
Aptr += K * 2;
|
||||
Crow += 2;
|
||||
m_rem4 -= 2;
|
||||
}
|
||||
if (m_rem4 >= 1) {
|
||||
kernel_1x1(K, Aptr, Bptr, Crow, alpha);
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,258 @@
|
||||
/***************************************************************************
|
||||
* Copyright (c) 2026, The OpenBLAS Project
|
||||
* All rights reserved.
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are
|
||||
* met:
|
||||
* 1. Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* 2. Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in
|
||||
* the documentation and/or other materials provided with the
|
||||
* distribution.
|
||||
* 3. Neither the name of the OpenBLAS project nor the names of
|
||||
* its contributors may be used to endorse or promote products
|
||||
* derived from this software without specific prior written permission.
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
* ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
* POSSIBILITY OF SUCH DAMAGE.
|
||||
* *****************************************************************************/
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
#include "common.h"
|
||||
|
||||
static inline void transpose8x8(float16x8_t *rows, float16x8_t *cols) {
|
||||
float64x2_t b0 = vtrn1q_f64(vreinterpretq_f64_f16(rows[0]), vreinterpretq_f64_f16(rows[4]));
|
||||
float64x2_t b1 = vtrn1q_f64(vreinterpretq_f64_f16(rows[1]), vreinterpretq_f64_f16(rows[5]));
|
||||
float64x2_t b2 = vtrn1q_f64(vreinterpretq_f64_f16(rows[2]), vreinterpretq_f64_f16(rows[6]));
|
||||
float64x2_t b3 = vtrn1q_f64(vreinterpretq_f64_f16(rows[3]), vreinterpretq_f64_f16(rows[7]));
|
||||
float64x2_t b4 = vtrn2q_f64(vreinterpretq_f64_f16(rows[0]), vreinterpretq_f64_f16(rows[4]));
|
||||
float64x2_t b5 = vtrn2q_f64(vreinterpretq_f64_f16(rows[1]), vreinterpretq_f64_f16(rows[5]));
|
||||
float64x2_t b6 = vtrn2q_f64(vreinterpretq_f64_f16(rows[2]), vreinterpretq_f64_f16(rows[6]));
|
||||
float64x2_t b7 = vtrn2q_f64(vreinterpretq_f64_f16(rows[3]), vreinterpretq_f64_f16(rows[7]));
|
||||
|
||||
float32x4_t c0 = vtrn1q_f32(vreinterpretq_f32_f64(b0), vreinterpretq_f32_f64(b2));
|
||||
float32x4_t c1 = vtrn1q_f32(vreinterpretq_f32_f64(b1), vreinterpretq_f32_f64(b3));
|
||||
float32x4_t c2 = vtrn2q_f32(vreinterpretq_f32_f64(b0), vreinterpretq_f32_f64(b2));
|
||||
float32x4_t c3 = vtrn2q_f32(vreinterpretq_f32_f64(b1), vreinterpretq_f32_f64(b3));
|
||||
float32x4_t c4 = vtrn1q_f32(vreinterpretq_f32_f64(b4), vreinterpretq_f32_f64(b6));
|
||||
float32x4_t c5 = vtrn1q_f32(vreinterpretq_f32_f64(b5), vreinterpretq_f32_f64(b7));
|
||||
float32x4_t c6 = vtrn2q_f32(vreinterpretq_f32_f64(b4), vreinterpretq_f32_f64(b6));
|
||||
float32x4_t c7 = vtrn2q_f32(vreinterpretq_f32_f64(b5), vreinterpretq_f32_f64(b7));
|
||||
|
||||
float16x8_t d0 = vtrn1q_f16(vreinterpretq_f16_f32(c0), vreinterpretq_f16_f32(c1));
|
||||
float16x8_t d1 = vtrn2q_f16(vreinterpretq_f16_f32(c0), vreinterpretq_f16_f32(c1));
|
||||
float16x8_t d2 = vtrn1q_f16(vreinterpretq_f16_f32(c2), vreinterpretq_f16_f32(c3));
|
||||
float16x8_t d3 = vtrn2q_f16(vreinterpretq_f16_f32(c2), vreinterpretq_f16_f32(c3));
|
||||
float16x8_t d4 = vtrn1q_f16(vreinterpretq_f16_f32(c4), vreinterpretq_f16_f32(c5));
|
||||
float16x8_t d5 = vtrn2q_f16(vreinterpretq_f16_f32(c4), vreinterpretq_f16_f32(c5));
|
||||
float16x8_t d6 = vtrn1q_f16(vreinterpretq_f16_f32(c6), vreinterpretq_f16_f32(c7));
|
||||
float16x8_t d7 = vtrn2q_f16(vreinterpretq_f16_f32(c6), vreinterpretq_f16_f32(c7));
|
||||
|
||||
cols[0] = d0;
|
||||
cols[1] = d1;
|
||||
cols[2] = d2;
|
||||
cols[3] = d3;
|
||||
cols[4] = d4;
|
||||
cols[5] = d5;
|
||||
cols[6] = d6;
|
||||
cols[7] = d7;
|
||||
}
|
||||
|
||||
static inline void transpose_4x4(float16x4_t *rows, float16x4_t *cols) {
|
||||
float16x8_t t0 = vcombine_f16(rows[0], vdup_n_f16(0.0f));
|
||||
float16x8_t t1 = vcombine_f16(rows[1], vdup_n_f16(0.0f));
|
||||
float16x8_t t2 = vcombine_f16(rows[2], vdup_n_f16(0.0f));
|
||||
float16x8_t t3 = vcombine_f16(rows[3], vdup_n_f16(0.0f));
|
||||
|
||||
float16x8_t t02 = vzip1q_f16(t0, t2);
|
||||
float16x8_t t13 = vzip1q_f16(t1, t3);
|
||||
|
||||
float16x8x2_t t0123 = vzipq_f16(t02, t13);
|
||||
|
||||
cols[0] = vget_low_f16(t0123.val[0]);
|
||||
cols[1] = vget_high_f16(t0123.val[0]);
|
||||
cols[2] = vget_low_f16(t0123.val[1]);
|
||||
cols[3] = vget_high_f16(t0123.val[1]);
|
||||
}
|
||||
|
||||
int CNAME(BLASLONG m, BLASLONG n, IFLOAT *a, BLASLONG lda, IFLOAT *b) {
|
||||
BLASLONG i, j;
|
||||
IFLOAT *a_offset = a;
|
||||
IFLOAT *b_offset = b;
|
||||
|
||||
float16x8_t v0, v1, v2, v3, v4, v5, v6, v7;
|
||||
float16x4_t v8, v9, v10, v11;
|
||||
|
||||
BLASLONG n8 = n >> 3;
|
||||
|
||||
for (j = 0; j < n8; j++) {
|
||||
IFLOAT *a0 = a_offset;
|
||||
IFLOAT *a1 = a0 + lda;
|
||||
IFLOAT *a2 = a1 + lda;
|
||||
IFLOAT *a3 = a2 + lda;
|
||||
IFLOAT *a4 = a3 + lda;
|
||||
IFLOAT *a5 = a4 + lda;
|
||||
IFLOAT *a6 = a5 + lda;
|
||||
IFLOAT *a7 = a6 + lda;
|
||||
a_offset += 8 * lda;
|
||||
|
||||
BLASLONG m8 = m >> 3;
|
||||
for (i = 0; i < m8; i++) {
|
||||
v0 = vld1q_f16((float16_t *)a0);
|
||||
v1 = vld1q_f16((float16_t *)a1);
|
||||
v2 = vld1q_f16((float16_t *)a2);
|
||||
v3 = vld1q_f16((float16_t *)a3);
|
||||
v4 = vld1q_f16((float16_t *)a4);
|
||||
v5 = vld1q_f16((float16_t *)a5);
|
||||
v6 = vld1q_f16((float16_t *)a6);
|
||||
v7 = vld1q_f16((float16_t *)a7);
|
||||
|
||||
float16x8_t rows[8] = {v0, v1, v2, v3, v4, v5, v6, v7};
|
||||
float16x8_t cols[8];
|
||||
transpose8x8(rows, cols);
|
||||
|
||||
vst1q_f16((float16_t *)b_offset, cols[0]);
|
||||
vst1q_f16((float16_t *)b_offset + 8, cols[1]);
|
||||
vst1q_f16((float16_t *)b_offset + 16, cols[2]);
|
||||
vst1q_f16((float16_t *)b_offset + 24, cols[3]);
|
||||
vst1q_f16((float16_t *)b_offset + 32, cols[4]);
|
||||
vst1q_f16((float16_t *)b_offset + 40, cols[5]);
|
||||
vst1q_f16((float16_t *)b_offset + 48, cols[6]);
|
||||
vst1q_f16((float16_t *)b_offset + 56, cols[7]);
|
||||
|
||||
a0 += 8;
|
||||
a1 += 8;
|
||||
a2 += 8;
|
||||
a3 += 8;
|
||||
a4 += 8;
|
||||
a5 += 8;
|
||||
a6 += 8;
|
||||
a7 += 8;
|
||||
b_offset += 64;
|
||||
}
|
||||
|
||||
BLASLONG i = (m & 7);
|
||||
if (i > 0) {
|
||||
for (BLASLONG k = 0; k < i; k++) {
|
||||
*(b_offset + 0) = *a0;
|
||||
*(b_offset + 1) = *a1;
|
||||
*(b_offset + 2) = *a2;
|
||||
*(b_offset + 3) = *a3;
|
||||
*(b_offset + 4) = *a4;
|
||||
*(b_offset + 5) = *a5;
|
||||
*(b_offset + 6) = *a6;
|
||||
*(b_offset + 7) = *a7;
|
||||
|
||||
a0++;
|
||||
a1++;
|
||||
a2++;
|
||||
a3++;
|
||||
a4++;
|
||||
a5++;
|
||||
a6++;
|
||||
a7++;
|
||||
|
||||
b_offset += 8;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (n & 4) {
|
||||
IFLOAT *a0 = a_offset;
|
||||
IFLOAT *a1 = a0 + lda;
|
||||
IFLOAT *a2 = a1 + lda;
|
||||
IFLOAT *a3 = a2 + lda;
|
||||
a_offset += 4 * lda;
|
||||
|
||||
BLASLONG m4 = m >> 2;
|
||||
for (i = 0; i < m4; i++) {
|
||||
v8 = vld1_f16((float16_t *)a0);
|
||||
v9 = vld1_f16((float16_t *)a1);
|
||||
v10 = vld1_f16((float16_t *)a2);
|
||||
v11 = vld1_f16((float16_t *)a3);
|
||||
|
||||
float16x4_t rows[4] = {v8, v9, v10, v11};
|
||||
float16x4_t cols[4];
|
||||
transpose_4x4(rows, cols);
|
||||
|
||||
vst1_f16((float16_t *)b_offset, cols[0]);
|
||||
vst1_f16((float16_t *)b_offset + 4, cols[1]);
|
||||
vst1_f16((float16_t *)b_offset + 8, cols[2]);
|
||||
vst1_f16((float16_t *)b_offset + 12, cols[3]);
|
||||
|
||||
a0 += 4;
|
||||
a1 += 4;
|
||||
a2 += 4;
|
||||
a3 += 4;
|
||||
b_offset += 16;
|
||||
}
|
||||
|
||||
BLASLONG i = (m & 3);
|
||||
if (i > 0) {
|
||||
for (BLASLONG k = 0; k < i; k++) {
|
||||
*(b_offset + 0) = *a0;
|
||||
*(b_offset + 1) = *a1;
|
||||
*(b_offset + 2) = *a2;
|
||||
*(b_offset + 3) = *a3;
|
||||
|
||||
a0++;
|
||||
a1++;
|
||||
a2++;
|
||||
a3++;
|
||||
|
||||
b_offset += 4;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (n & 2) {
|
||||
IFLOAT *a0 = a_offset;
|
||||
IFLOAT *a1 = a0 + lda;
|
||||
a_offset += 2 * lda;
|
||||
|
||||
BLASLONG m2 = m >> 1;
|
||||
for (i = 0; i < m2; i++) {
|
||||
|
||||
v8 = vld1_f16((float16_t *)a0);
|
||||
v9 = vld1_f16((float16_t *)a1);
|
||||
|
||||
float16_t col0[2] = {vget_lane_f16(v8, 0), vget_lane_f16(v9, 0)};
|
||||
float16_t col1[2] = {vget_lane_f16(v8, 1), vget_lane_f16(v9, 1)};
|
||||
|
||||
b_offset[0] = col0[0];
|
||||
b_offset[1] = col0[1];
|
||||
b_offset[2] = col1[0];
|
||||
b_offset[3] = col1[1];
|
||||
|
||||
a0 += 2;
|
||||
a1 += 2;
|
||||
b_offset += 4;
|
||||
}
|
||||
|
||||
if (m & 1) {
|
||||
b_offset[0] = *a0;
|
||||
b_offset[1] = *a1;
|
||||
b_offset += 2;
|
||||
}
|
||||
}
|
||||
|
||||
if (n & 1) {
|
||||
IFLOAT *a0 = a_offset;
|
||||
for (i = 0; i < m; i++) {
|
||||
*b_offset++ = *a0;
|
||||
a0++;
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
/***************************************************************************
|
||||
* Copyright (c) 2026, The OpenBLAS Project
|
||||
* All rights reserved.
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are
|
||||
* met:
|
||||
* 1. Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* 2. Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in
|
||||
* the documentation and/or other materials provided with the
|
||||
* distribution.
|
||||
* 3. Neither the name of the OpenBLAS project nor the names of
|
||||
* its contributors may be used to endorse or promote products
|
||||
* derived from this software without specific prior written permission.
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
* ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
* POSSIBILITY OF SUCH DAMAGE.
|
||||
* *****************************************************************************/
|
||||
|
||||
#include <arm_sve.h>
|
||||
|
||||
#include "common.h"
|
||||
|
||||
int CNAME(BLASLONG m, BLASLONG n, IFLOAT *a, BLASLONG lda, IFLOAT *b) {
|
||||
BLASLONG i, j;
|
||||
IFLOAT *aoffset, *aoffset1;
|
||||
IFLOAT *boffset, *boffset1;
|
||||
IFLOAT *boffset2, *boffset3, *boffset4;
|
||||
|
||||
aoffset = a;
|
||||
boffset = b;
|
||||
|
||||
boffset2 = b + m * (n & ~7);
|
||||
boffset3 = b + m * (n & ~3);
|
||||
boffset4 = b + m * (n & ~1);
|
||||
|
||||
svbool_t pg8 = svwhilelt_b16(0, 8);
|
||||
svbool_t pg4 = svwhilelt_b16(0, 4);
|
||||
|
||||
for (j = 0; j < m; j++) {
|
||||
aoffset1 = aoffset;
|
||||
boffset1 = boffset;
|
||||
|
||||
aoffset += lda;
|
||||
boffset += 8;
|
||||
|
||||
for (i = 0; i < (n >> 3); i++) {
|
||||
svfloat16_t v0 = svld1_f16(pg8, (float16_t *)aoffset1);
|
||||
svst1_f16(pg8, (float16_t *)boffset1, v0);
|
||||
|
||||
aoffset1 += 8;
|
||||
boffset1 += 8 * m;
|
||||
}
|
||||
|
||||
if (n & 4) {
|
||||
svfloat16_t v0 = svld1_f16(pg4, (float16_t *)aoffset1);
|
||||
svst1_f16(pg4, (float16_t *)boffset2, v0);
|
||||
|
||||
aoffset1 += 4;
|
||||
boffset2 += 4;
|
||||
}
|
||||
|
||||
if (n & 2) {
|
||||
boffset3[0] = aoffset1[0];
|
||||
boffset3[1] = aoffset1[1];
|
||||
aoffset1 += 2;
|
||||
boffset3 += 2;
|
||||
}
|
||||
|
||||
if (n & 1) {
|
||||
boffset4[0] = aoffset1[0];
|
||||
aoffset1 += 1;
|
||||
boffset4 += 1;
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,245 @@
|
||||
/***************************************************************************
|
||||
Copyright (c) 2017, The OpenBLAS Project
|
||||
All rights reserved.
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are
|
||||
met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in
|
||||
the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
3. Neither the name of the OpenBLAS project nor the names of
|
||||
its contributors may be used to endorse or promote products
|
||||
derived from this software without specific prior written permission.
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
|
||||
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*****************************************************************************/
|
||||
|
||||
#include "common.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
#define N "x0" /* vector length */
|
||||
#define X "x1" /* "X" vector address */
|
||||
#define INC_X "x2" /* "X" stride */
|
||||
#define J "x5" /* loop variable */
|
||||
|
||||
#define REG0 "wzr"
|
||||
#define SUMF "s0"
|
||||
#define SUMFD "d0"
|
||||
|
||||
/******************************************************************************/
|
||||
|
||||
#define KERNEL_F1 \
|
||||
"ldr s1, ["X"] \n" \
|
||||
"add "X", "X", #4 \n" \
|
||||
"fadd "SUMF", "SUMF", s1 \n"
|
||||
|
||||
#define KERNEL_F64 \
|
||||
"ldr q16, ["X"] \n" \
|
||||
"ldr q17, ["X", #16] \n" \
|
||||
"ldr q18, ["X", #32] \n" \
|
||||
"ldr q19, ["X", #48] \n" \
|
||||
"ldp q20, q21, ["X", #64] \n" \
|
||||
"ldp q22, q23, ["X", #96] \n" \
|
||||
"ldp q24, q25, ["X", #128] \n" \
|
||||
"ldp q26, q27, ["X", #160] \n" \
|
||||
"fadd v16.4s, v16.4s, v17.4s \n" \
|
||||
"fadd v18.4s, v18.4s, v19.4s \n" \
|
||||
"ldp q28, q29, ["X", #192] \n" \
|
||||
"ldp q30, q31, ["X", #224] \n" \
|
||||
"add "X", "X", #256 \n" \
|
||||
"fadd v20.4s, v20.4s, v21.4s \n" \
|
||||
"fadd v22.4s, v22.4s, v23.4s \n" \
|
||||
"PRFM PLDL1KEEP, ["X", #1024] \n" \
|
||||
"PRFM PLDL1KEEP, ["X", #1024+64] \n" \
|
||||
"fadd v24.4s, v24.4s, v25.4s \n" \
|
||||
"fadd v26.4s, v26.4s, v27.4s \n" \
|
||||
"fadd v0.4s, v0.4s, v16.4s \n" \
|
||||
"fadd v1.4s, v1.4s, v18.4s \n" \
|
||||
"fadd v2.4s, v2.4s, v20.4s \n" \
|
||||
"fadd v3.4s, v3.4s, v22.4s \n" \
|
||||
"PRFM PLDL1KEEP, ["X", #1024+128] \n" \
|
||||
"PRFM PLDL1KEEP, ["X", #1024+192] \n" \
|
||||
"fadd v28.4s, v28.4s, v29.4s \n" \
|
||||
"fadd v30.4s, v30.4s, v31.4s \n" \
|
||||
"fadd v4.4s, v4.4s, v24.4s \n" \
|
||||
"fadd v5.4s, v5.4s, v26.4s \n" \
|
||||
"fadd v6.4s, v6.4s, v28.4s \n" \
|
||||
"fadd v7.4s, v7.4s, v30.4s \n"
|
||||
|
||||
#define KERNEL_F64_FINALIZE \
|
||||
"fadd v0.4s, v0.4s, v1.4s \n" \
|
||||
"fadd v2.4s, v2.4s, v3.4s \n" \
|
||||
"fadd v4.4s, v4.4s, v5.4s \n" \
|
||||
"fadd v6.4s, v6.4s, v7.4s \n" \
|
||||
"fadd v0.4s, v0.4s, v2.4s \n" \
|
||||
"fadd v4.4s, v4.4s, v6.4s \n" \
|
||||
"fadd v0.4s, v0.4s, v4.4s \n" \
|
||||
"ext v1.16b, v0.16b, v0.16b, #8 \n" \
|
||||
"fadd v0.2s, v0.2s, v1.2s \n" \
|
||||
"faddp "SUMF", v0.2s \n"
|
||||
|
||||
#define INIT_S \
|
||||
"lsl "INC_X", "INC_X", #2 \n"
|
||||
|
||||
#define KERNEL_S1 \
|
||||
"ldr s1, ["X"] \n" \
|
||||
"add "X", "X", "INC_X" \n" \
|
||||
"fadd "SUMF", "SUMF", s1 \n"
|
||||
|
||||
|
||||
#if defined(SMP)
|
||||
extern int blas_level1_thread_with_return_value(int mode, BLASLONG m, BLASLONG n,
|
||||
BLASLONG k, void *alpha, void *a, BLASLONG lda, void *b, BLASLONG ldb,
|
||||
void *c, BLASLONG ldc, int (*function)(), int nthreads);
|
||||
#endif
|
||||
|
||||
|
||||
static FLOAT ssum_compute(BLASLONG n, FLOAT *x, BLASLONG inc_x)
|
||||
{
|
||||
FLOAT ssum = 0.0 ;
|
||||
|
||||
if ( n < 0 ) return(ssum);
|
||||
|
||||
__asm__ __volatile__ (
|
||||
" mov "N", %[N_] \n"
|
||||
" mov "X", %[X_] \n"
|
||||
" mov "INC_X", %[INCX_] \n"
|
||||
" fmov "SUMF", "REG0" \n"
|
||||
" fmov s1, "REG0" \n"
|
||||
" fmov s2, "REG0" \n"
|
||||
" fmov s3, "REG0" \n"
|
||||
" fmov s4, "REG0" \n"
|
||||
" fmov s5, "REG0" \n"
|
||||
" fmov s6, "REG0" \n"
|
||||
" fmov s7, "REG0" \n"
|
||||
" cmp "N", xzr \n"
|
||||
" ble 9f //ssum_kernel_L999 \n"
|
||||
" cmp "INC_X", xzr \n"
|
||||
" ble 9f //ssum_kernel_L999 \n"
|
||||
" cmp "INC_X", #1 \n"
|
||||
" bne 5f //ssum_kernel_S_BEGIN \n"
|
||||
|
||||
"1: //sum_kernel_F_BEGIN: \n"
|
||||
" asr "J", "N", #6 \n"
|
||||
" cmp "J", xzr \n"
|
||||
" beq 3f //ssum_kernel_F1 \n"
|
||||
#if !(defined(__clang__) && defined(OS_WINDOWS))
|
||||
".align 5 \n"
|
||||
#endif
|
||||
"2: //ssum_kernel_F64: \n"
|
||||
" "KERNEL_F64" \n"
|
||||
" subs "J", "J", #1 \n"
|
||||
" bne 2b //ssum_kernel_F64 \n"
|
||||
" "KERNEL_F64_FINALIZE" \n"
|
||||
|
||||
"3: //ssum_kernel_F1: \n"
|
||||
" ands "J", "N", #63 \n"
|
||||
" ble 9f //ssum_kernel_L999 \n"
|
||||
|
||||
"4: //ssum_kernel_F10: \n"
|
||||
" "KERNEL_F1" \n"
|
||||
" subs "J", "J", #1 \n"
|
||||
" bne 4b //ssum_kernel_F10 \n"
|
||||
" b 9f //ssum_kernel_L999 \n"
|
||||
|
||||
"5: //ssum_kernel_S_BEGIN: \n"
|
||||
" "INIT_S" \n"
|
||||
" asr "J", "N", #2 \n"
|
||||
" cmp "J", xzr \n"
|
||||
" ble 7f //ssum_kernel_S1 \n"
|
||||
|
||||
"6: //ssum_kernel_S4: \n"
|
||||
" "KERNEL_S1" \n"
|
||||
" "KERNEL_S1" \n"
|
||||
" "KERNEL_S1" \n"
|
||||
" "KERNEL_S1" \n"
|
||||
" subs "J", "J", #1 \n"
|
||||
" bne 6b //ssum_kernel_S4 \n"
|
||||
|
||||
"7: //ssum_kernel_S1: \n"
|
||||
" ands "J", "N", #3 \n"
|
||||
" ble 9f //ssum_kernel_L999 \n"
|
||||
|
||||
"8: //ssum_kernel_S10: \n"
|
||||
" "KERNEL_S1" \n"
|
||||
" subs "J", "J", #1 \n"
|
||||
" bne 8b //ssum_kernel_S10 \n"
|
||||
|
||||
"9: //ssum_kernel_L999: \n"
|
||||
" fmov %[SSUM_], "SUMFD" \n"
|
||||
|
||||
: [SSUM_] "=r" (ssum) //%0
|
||||
: [N_] "r" (n), //%1
|
||||
[X_] "r" (x), //%2
|
||||
[INCX_] "r" (inc_x) //%3
|
||||
: "cc",
|
||||
"memory",
|
||||
"x0", "x1", "x2", "x3", "x4", "x5",
|
||||
"d0", "d1", "d2", "d3", "d4", "d5", "d6", "d7"
|
||||
);
|
||||
|
||||
return ssum;
|
||||
}
|
||||
|
||||
#if defined(SMP)
|
||||
static int ssum_thread_function(BLASLONG n, BLASLONG dummy0,
|
||||
BLASLONG dummy1, FLOAT dummy2, FLOAT *x, BLASLONG inc_x, FLOAT *y,
|
||||
BLASLONG inc_y, FLOAT *result, BLASLONG dummy3)
|
||||
{
|
||||
*result = ssum_compute(n, x, inc_x);
|
||||
|
||||
return 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x)
|
||||
{
|
||||
#if defined(SMP)
|
||||
int nthreads;
|
||||
FLOAT dummy_alpha;
|
||||
#endif
|
||||
FLOAT ssum = 0.0;
|
||||
|
||||
#if defined(SMP)
|
||||
if (inc_x == 0 || n <= 10000)
|
||||
nthreads = 1;
|
||||
else
|
||||
nthreads = num_cpu_avail(1);
|
||||
|
||||
if (nthreads == 1) {
|
||||
ssum = ssum_compute(n, x, inc_x);
|
||||
} else {
|
||||
int mode, i;
|
||||
char result[MAX_CPU_NUMBER * sizeof(double) * 2];
|
||||
FLOAT *ptr;
|
||||
|
||||
mode = BLAS_SINGLE;
|
||||
|
||||
blas_level1_thread_with_return_value(mode, n, 0, 0, &dummy_alpha,
|
||||
x, inc_x, NULL, 0, result, 0,
|
||||
( void *)ssum_thread_function, nthreads);
|
||||
|
||||
ptr = (FLOAT *)result;
|
||||
for (i = 0; i < nthreads; i++) {
|
||||
ssum = ssum + (*ptr);
|
||||
ptr = (FLOAT *)(((char *)ptr) + sizeof(double) * 2);
|
||||
}
|
||||
}
|
||||
#else
|
||||
ssum = ssum_compute(n, x, inc_x);
|
||||
#endif
|
||||
|
||||
return ssum;
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user