Compare commits

...
Author SHA1 Message Date
Martin Kroeker c59578f314 fix conditionals
apple m / build (cmake, gfortran, 0, 0) (push) Waiting to run
apple m / build (cmake, gfortran, 0, 1) (push) Waiting to run
apple m / build (cmake, gfortran, 1, 0) (push) Waiting to run
apple m / build (cmake, gfortran, 1, 1) (push) Waiting to run
apple m / build (make, gfortran, 0, 0) (push) Waiting to run
apple m / build (make, gfortran, 0, 1) (push) Waiting to run
apple m / build (make, gfortran, 1, 0) (push) Waiting to run
apple m / build (make, gfortran, 1, 1) (push) Waiting to run
c910v qemu test / TEST (riscv64-linux-gnu, NO_SHARED=1 TARGET=C910V, C910V, riscv64-unknown-linux-gnu) (push) Waiting to run
c910v qemu test / TEST (riscv64-linux-gnu, NO_SHARED=1 TARGET=RISCV64_GENERIC, RISCV64_GENERIC, riscv64-linux-gnu) (push) Waiting to run
Run codspeed benchmarks / benchmarks (make, gfortran, ubuntu-22.04, 3.12) (push) Waiting to run
continuous build / build (cmake, clang, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, macos-latest) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang-21, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang-21, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, clang-21, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, gcc, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, gcc, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, gcc, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang, gfortran, macos-latest) (push) Waiting to run
continuous build / build (make, clang, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, clang, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang-21, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang-21, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, clang-21, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, gcc, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, gcc, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, gcc, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / msys2 (None, fc, int32, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, CLANG64, mingw-w64-clang-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, MINGW32, mingw-w64-i686) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int64, -DBINARY=64 -DINTERFACE64=1, CLANG64, mingw-w64-clang-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int64, -DBINARY=64 -DINTERFACE64=1, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / cross_build (DYNAMIC_ARCH=1 TARGET=GENERIC, mips64el, mips64el-linux-gnuabi64) (push) Waiting to run
continuous build / cross_build (TARGET=EV4, alpha, alpha-linux-gnu) (push) Waiting to run
continuous build / cross_build (TARGET=MIPS1004K, mipsel, mipsel-linux-gnu) (push) Waiting to run
continuous build / cross_build (TARGET=RISCV64_GENERIC, riscv64, riscv64-linux-gnu) (push) Waiting to run
continuous build / neoverse_build (push) Waiting to run
harmonyos / build (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=GENERIC, DYNAMIC_ARCH, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA264, LA264, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA464, LA464, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA64_GENERIC, LA64_GENERIC, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON2K1000, LOONGSON2K1000, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON3R5, LOONGSON3R5, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSONGENERIC, LOONGSONGENERIC, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=GENERIC, DYNAMIC_ARCH) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA264, LA264) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA464, LA464) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA64_GENERIC, LA64_GENERIC) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON2K1000, LOONGSON2K1000) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON3R5, LOONGSON3R5) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSONGENERIC, LOONGSONGENERIC) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=I6400, I6400, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=I6500, I6500, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=MIPS64_GENERIC, MIPS64_GENERIC, mips64el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=P6600, P6600, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=SICORTEX, SICORTEX, mips64el-linux-gnuabi64) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_GENERIC BINARY=64 ARCH=riscv64 DYNAMIC_ARCH=1, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=256,elen=64, DYNAMIC_ARCH=1) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_ZVL128B BINARY=64 ARCH=riscv64, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=128,elen=64, RISCV64_ZVL128B) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_ZVL256B BINARY=64 ARCH=riscv64 BUILD_BFLOAT16=1 BUILD_HFLOAT16=1, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=256,elen=64,zfh=true,zvfh=true,zvfbfwma=true, RISCV64_ZVL256B) (push) Waiting to run
2026-04-14 21:32:36 +02:00
Martin Kroeker d9786d387c fix missing eol 2026-04-14 10:56:04 +02:00
Martin Kroeker b9da7dbd24 Quote the respective SUMM file on failure in BLAS2/3 tests 2026-04-12 23:17:36 +02:00
Martin Kroeker 94e053ac10 Work around miscompilation of the AVX512 ?GEMM kernels by Windows LLVM 2026-04-11 19:27:31 +02:00
Martin Kroeker 646d0c9fee Merge pull request #5751 from chenx97/mips-dedup-c-impl
Remove redundant C implemetations from MIPS directories
2026-04-09 23:51:37 +02:00
Martin Kroeker 2c80f8c974 Merge pull request #5755 from martin-frbg/fixup5702
Fix partial merge of changes from PR #5702 (applying Reference-LAPACK PR 1203)
2026-04-09 16:35:13 +02:00
Martin Kroeker 0ea23484c6 Use ROUNDUP_LWORK and remove redundant conversions (Reference-LAPACK PR1203) 2026-04-09 14:54:55 +02:00
Martin Kroeker 1aea1d6237 Merge pull request #5753 from foxtran/fix/1203
Fix DROUNDUP_LWORK: patch was not fully copied
2026-04-09 14:08:23 +02:00
Henry Chen 6a5d2142f4 Fix dsdot precision for arm/dot.c 2026-04-09 18:11:47 +08:00
Igor S. Gerasimov f9f8e94a14 Fix DROUNDUP_LWORK: patch was not fully copied 2026-04-09 02:23:58 +02:00
Martin Kroeker 9a0f76a0a1 Merge pull request #5746 from chenx97/mips-fix-implicit-declaration
MIPS: fix implicit declarations found in the cpuinfo detector
2026-04-08 15:32:20 +02:00
Martin Kroeker 75a99605af Merge pull request #5744 from nakatamaho/fix/slamc3
lapack/laed3: fix MinGW build by matching LAMC3 prototype
2026-04-08 12:18:22 +02:00
Martin Kroeker 9d9fcc1881 Merge pull request #5752 from martin-frbg/fixup5748
Fix missing endif in openblas_config.h
2026-04-08 12:17:35 +02:00
Martin Kroeker e926bb0523 fix missing endif 2026-04-08 10:38:56 +02:00
Henry Chen e875a9cdd0 Remove redundant C implemetations from MIPS directories 2026-04-08 13:59:10 +08:00
Martin Kroeker fb45e7da89 CirrusCI: Fix ranlib confusion between xcode and AndroidNDK (#5749)
* Use ar and ranlib from Android NDK rather than xcode
2026-04-08 00:34:18 +02:00
Martin Kroeker e41cb1ad7a Merge pull request #5748 from martin-frbg/issue5747
Check that _Float16 is available before using it in openblas_config.h
2026-04-07 22:52:20 +02:00
Martin Kroeker dc32a8a90f Try to find out if _Float16 is available on the target before using it 2026-04-07 18:31:40 +02:00
Henry Chen a04ea2b2c4 MIPS: fix implicit declarations found in the cpuinfo detector 2026-04-07 15:59:50 +08:00
NAKATA Maho f272216ae3 lapack/laed3: fix MinGW build for slaed3
common_interface.h declares slamc3 as returning FLOATRET when
NEED_F2CCONV is enabled, but laed3_single.c and laed3_parallel.c
redeclared LAMC3 as returning FLOAT. This causes conflicting-type
errors in MinGW builds.

Use FLOATRET for the local LAMC3 prototype so it matches the shared
declaration. Also undefine the Windows max macro before the local
max definition in laed3_parallel.c to avoid macro redefinition
warnings.
2026-04-06 13:13:21 +09:00
Martin Kroeker 9b3cc7835b Merge pull request #5741 from martin-frbg/issue5696
Add note on using an x86 OpenBLAS in Windows on Arm via Prism
2026-04-02 11:38:16 +02:00
Martin Kroeker bef5f1c6e2 Merge pull request #5740 from martin-frbg/issue5739
Fix out-of-bounds access in the converted C version of the CBLAS tests
2026-04-02 11:37:57 +02:00
Martin Kroeker 3bbd755ba7 Add note on using an x86 OpenBLAS in Windows on Arm via Prism 2026-04-02 10:36:33 +02:00
Martin Kroeker 47be0d8a52 Fix access beyond array size 2026-04-02 10:14:09 +02:00
Martin Kroeker 93515c2f7a Merge pull request #5736 from martin-frbg/lapack1221
Follow-up on ?GESVDQ updates from PR1146 (Reference-LAPACK PR 1221)
2026-04-01 22:47:39 +02:00
Martin Kroeker 7dde52d5d2 Follow-up on ?GESVDQ updates from PR 1146 (Reference-LAPACK PR 1221) 2026-04-01 16:27:56 +02:00
Martin Kroeker c6e4d17819 Follow-up on ?GESVDQ updates from PR 1146 (Reference-LAPACK PR 1221) 2026-04-01 15:34:02 +02:00
Martin Kroeker b9ba9be508 Follow-up on ?GESVDQ updates from PR1146 (Reference-LAPACK PR 1221) 2026-04-01 15:19:41 +02:00
Martin Kroeker d27e98c97a Merge pull request #5734 from martin-frbg/lapack774
Fix workspace size in ?TGSEN (Reference-LAPACK PR 774)
2026-04-01 08:52:37 +02:00
Martin Kroeker 429d23f420 Merge pull request #5730 from martin-frbg/lapack1206
Fix overwriting of LDSWORK in ?TRSYL3 (Reference-LAPACK PR 1206)
2026-03-31 23:58:04 +02:00
Martin Kroeker 3f2338ba85 Merge pull request #5732 from martin-frbg/lapack1209
Remove unused parameter in  DORBDB3/ZUNBDB3 (Reference-LAPACK PR 1209)
2026-03-31 23:57:36 +02:00
Martin Kroeker 62dcdca823 Merge pull request #5733 from martin-frbg/lapack1211
Re-enable testing of the ?BB/?GG driver routines (Reference-LAPACK PR 1211)
2026-03-31 23:57:08 +02:00
Martin Kroeker eaeaf95e23 Merge pull request #5731 from martin-frbg/lapack1207
Fix crossover of INFO variables in some EIG tests (Reference-LAPACK PR 1207)
2026-03-31 23:56:47 +02:00
Martin Kroeker f1f36c02b9 Merge pull request #5729 from martin-frbg/lapack1195
Fix truncation of large workspace values in ZHE routines (Reference-LAPACK PR 1195)
2026-03-31 21:36:54 +02:00
Martin Kroeker 9816062aaf Merge pull request #5727 from martin-frbg/lapack1187
Fix DGGES test seed to avoid bad matrix (Reference-LAPACK PR 1187)
2026-03-31 21:36:35 +02:00
Martin Kroeker 664f17655c Merge pull request #5726 from martin-frbg/lapack1149
Fix display of version number in LAPACK tests (Reference-LAPACK PR 1149)
2026-03-31 21:36:17 +02:00
Martin Kroeker aec6170a8b Merge pull request #5725 from martin-frbg/lapack1146
Fix support for jobu/v in LAPACKE_?GESVDQ_WORK (Reference-LAPACK PR 1146)
2026-03-31 19:21:10 +02:00
Martin Kroeker 66cc9f043d Merge pull request #5724 from martin-frbg/lapack1136
Add NaN checks for input matrix A in ?GEEV (Reference-LAPACK PR 1136)
2026-03-31 16:16:25 +02:00
Martin Kroeker cc74393520 Fix workspace size (Reference-LAPACK PR 774) 2026-03-31 14:06:00 +02:00
Martin Kroeker 4bbb9fefc0 Fix workspace size (Reference-LAPACK PR 774) 2026-03-31 14:02:02 +02:00
Martin Kroeker e48625414f Merge pull request #5723 from martin-frbg/lapack1094
Change WORK dimension in deprecated ?GELQS/?GEQRS (Reference-LAPACK PR 1094)
2026-03-31 12:16:09 +02:00
Martin Kroeker 844939a9fb Enable testing of the driver routines (Reference-LAPACK PR 1211) 2026-03-31 11:53:42 +02:00
Martin Kroeker f085c70784 Remove unused parameter (Reference-LAPACK PR 1209) 2026-03-31 11:47:24 +02:00
Martin Kroeker 391cbf8584 Pass IINFO instead of INFO to ??PGVX (Reference-LAPACK PR 1207) 2026-03-31 11:39:45 +02:00
Martin Kroeker 6e89813300 Fix spurious overwriting of caller variable LDSWORK (Reference-LAPACK PR 1206) 2026-03-31 11:31:38 +02:00
Martin Kroeker 37e189c85d Fix truncation of large workspace values (Reference-LAPACK PR 1195) 2026-03-31 10:59:07 +02:00
Martin Kroeker 6dad37ff8d Merge pull request #5722 from martin-frbg/lapack1023
Change loop order in ?GETC2 (Reference-LAPACK PR 1023)
2026-03-31 09:43:49 +02:00
Martin Kroeker 004cf0d3d0 Fix seed to avoid FMA-sensitive ill-conditioned matrix (Reference-LAPACK PR 1187) 2026-03-31 00:02:26 +02:00
Martin Kroeker 1243314201 Fix display of minor version number (Reference-LAPACK PR 1149) 2026-03-30 23:48:43 +02:00
Martin Kroeker edad2a8b2f Fix display of minor version number (Reference-LAPACK PR 1149) 2026-03-30 23:47:31 +02:00
Martin Kroeker 55d7dd89ae Fix support for jobu and jobv (Reference-LAPACK PR 1146) 2026-03-30 23:36:42 +02:00
Martin Kroeker e19e140619 Add NaN checks for input matrix A (Reference-LAPACK PR 1136) 2026-03-30 23:06:12 +02:00
Martin Kroeker a03cd30185 Change WORK(LWORK) to WORK(*) (Reference-LAPACK PR 1094) 2026-03-30 21:36:05 +02:00
Martin Kroeker 904f9d60b0 Merge pull request #5721 from martin-frbg/lapack1020
Implement ?LARF1F and ?ORM2R (Reference-LAPACK PRs 1019/1020/1196)
2026-03-30 21:25:56 +02:00
Martin Kroeker ff5dc3ebc1 Change loop ordering to improve performance (Reference-LAPACK PR 1023) 2026-03-30 20:24:31 +02:00
Martin Kroeker a5d0f89ea4 Add C replacements for ?LARF1F/?LARF1L 2026-03-30 19:41:54 +02:00
Martin Kroeker af63f2a1aa Add C replacements for ?LARF1F/?LARF1L 2026-03-30 19:40:14 +02:00
Martin Kroeker 4342764c23 Implement ?LARF1F and ?ORM2R (Reference-LAPACK PRs 1019/1020/1196) 2026-03-30 19:15:29 +02:00
Martin Kroeker d9bb8f369f Implement ?LARF1F and ?ORM2R (Reference-LAPACK PRs 1019/1020/1196) 2026-03-30 18:45:36 +02:00
Martin Kroeker f5f789fc52 Implement ?LARF1F and ?ORM2R (Reference-LAPACK PRs 1019/1020/1196) 2026-03-30 18:41:59 +02:00
Martin Kroeker 605b1287e3 Add ?LARF1F and ?LARF1L (Reference-LAPACK PRs 1019/1020) 2026-03-30 18:34:28 +02:00
Martin Kroeker d26960a21e Merge pull request #5719 from martin-frbg/issue5713
ARM64 DYNAMIC_ARCH: add CortexA75/76  and restore VORTEX for DYNAMIC_LIST
2026-03-30 07:19:53 +02:00
Martin Kroeker 16211b7170 Add CortexA75/76 via CortexA73 and restore VORTEX for use with DYNAMIC_LIST 2026-03-29 22:10:44 +02:00
Martin Kroeker 0f9f6e4be5 Merge pull request #5710 from martin-frbg/issue5708
Work around miscompilation of the ARM64 non-SVE DDOT kernel
2026-03-27 22:09:08 +01:00
Martin Kroeker 3ebfc0ef65 Merge pull request #5718 from martin-frbg/issue5625
Fix CMake DYNAMIC_ARCH builds under Windows on Arm
2026-03-27 16:51:58 +01:00
Martin Kroeker 0315003d1f Do not build SME targets in DYNAMIC_ARCH under Windows 2026-03-27 13:42:41 +01:00
Martin Kroeker 75511cb67c POSIX strncasecmp is strnicmp in Windows on Arm 2026-03-27 13:40:39 +01:00
Martin Kroeker b8dbc4a1fc Merge pull request #5716 from yuanjia111/develop
[ARM64] Add optimized fp16 shgemm kernels for Neoverse N2
2026-03-27 13:36:25 +01:00
yuanjia e6eba9fa21 Add optimized FP16 shgemm for for NEOVERSEN2 target 2026-03-27 17:55:06 +08:00
Martin Kroeker 2671786e61 Merge pull request #5715 from martin-frbg/issue5714
typedef the unsupported fp16 as bfloat16 on Loongarch64 too
2026-03-27 10:19:44 +01:00
Martin Kroeker 3c188e4c12 Merge pull request #5712 from murste01/develop
Fix incorrect cast from BF16 to FP32 in SBGEMM
2026-03-27 07:50:00 +01:00
Martin Kroeker 7086a1b075 typedef the unsupported fp16 as bfloat16 on Loongarch64 too 2026-03-26 23:00:04 +01:00
Murray Steele f6d4fe703b Fix incorrect cast from BF16 to FP32 in SBGEMM
This change fixes a regression in SBGEMM where C is assumed to be BF16,
and so unconditionally casts the output to FP32 resulting in incorrect
outputs when beta=1.
2026-03-26 12:10:52 +00:00
Martin Kroeker 1f1fcd4927 Merge pull request #5709 from iv-m/loongarch64-fix-typo
c_check: loongarch64: Fix typo
2026-03-24 23:10:52 +01:00
Martin Kroeker e3ce4623c2 Use volatile attribute for SDOT only, to avoid creating new miscompilations 2026-03-24 23:08:02 +01:00
Ivan A. Melnikov 86971646ed c_check: loongarch64: Fix typo
Fixes: 42c7a27e6b
2026-03-24 21:25:46 +04:00
Martin Kroeker b8697b3448 Update version to 0.3.32.dev 2026-03-24 00:02:13 +01:00
Martin Kroeker d511552e64 Update version to 0.3.32.dev 2026-03-24 00:01:33 +01:00
Martin Kroeker 821242ed9d Merge pull request #5706 from OpenMathLib/release-0.3.0
Merge back from release branch to copy 0.3.32 tag
2026-03-24 00:00:52 +01:00
Martin Kroeker 8cecf899e2 Update version to 0.3.32
apple m / build (cmake, gfortran, 0, 0) (push) Waiting to run
apple m / build (cmake, gfortran, 0, 1) (push) Waiting to run
apple m / build (cmake, gfortran, 1, 0) (push) Waiting to run
apple m / build (cmake, gfortran, 1, 1) (push) Waiting to run
apple m / build (make, gfortran, 0, 0) (push) Waiting to run
apple m / build (make, gfortran, 0, 1) (push) Waiting to run
apple m / build (make, gfortran, 1, 0) (push) Waiting to run
apple m / build (make, gfortran, 1, 1) (push) Waiting to run
c910v qemu test / TEST (riscv64-linux-gnu, NO_SHARED=1 TARGET=C910V, C910V, riscv64-unknown-linux-gnu) (push) Waiting to run
c910v qemu test / TEST (riscv64-linux-gnu, NO_SHARED=1 TARGET=RISCV64_GENERIC, RISCV64_GENERIC, riscv64-linux-gnu) (push) Waiting to run
Run codspeed benchmarks / benchmarks (make, gfortran, ubuntu-22.04, 3.12) (push) Waiting to run
continuous build / build (cmake, clang, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, macos-latest) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang-21, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang-21, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, clang-21, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, gcc, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, gcc, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, gcc, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang, gfortran, macos-latest) (push) Waiting to run
continuous build / build (make, clang, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, clang, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang-21, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang-21, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, clang-21, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, gcc, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, gcc, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, gcc, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / msys2 (None, fc, int32, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, CLANG64, mingw-w64-clang-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, MINGW32, mingw-w64-i686) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int64, -DBINARY=64 -DINTERFACE64=1, CLANG64, mingw-w64-clang-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int64, -DBINARY=64 -DINTERFACE64=1, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / cross_build (DYNAMIC_ARCH=1 TARGET=GENERIC, mips64el, mips64el-linux-gnuabi64) (push) Waiting to run
continuous build / cross_build (TARGET=EV4, alpha, alpha-linux-gnu) (push) Waiting to run
continuous build / cross_build (TARGET=MIPS1004K, mipsel, mipsel-linux-gnu) (push) Waiting to run
continuous build / cross_build (TARGET=RISCV64_GENERIC, riscv64, riscv64-linux-gnu) (push) Waiting to run
continuous build / neoverse_build (push) Waiting to run
harmonyos / build (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=GENERIC, DYNAMIC_ARCH, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA264, LA264, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA464, LA464, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA64_GENERIC, LA64_GENERIC, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON2K1000, LOONGSON2K1000, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON3R5, LOONGSON3R5, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSONGENERIC, LOONGSONGENERIC, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=GENERIC, DYNAMIC_ARCH) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA264, LA264) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA464, LA464) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA64_GENERIC, LA64_GENERIC) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON2K1000, LOONGSON2K1000) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON3R5, LOONGSON3R5) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSONGENERIC, LOONGSONGENERIC) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=I6400, I6400, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=I6500, I6500, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=MIPS64_GENERIC, MIPS64_GENERIC, mips64el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=P6600, P6600, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=SICORTEX, SICORTEX, mips64el-linux-gnuabi64) (push) Waiting to run
Nightly-Homebrew-Build / build-OpenBLAS-with-Homebrew (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_GENERIC BINARY=64 ARCH=riscv64 DYNAMIC_ARCH=1, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=256,elen=64, DYNAMIC_ARCH=1) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_ZVL128B BINARY=64 ARCH=riscv64, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=128,elen=64, RISCV64_ZVL128B) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_ZVL256B BINARY=64 ARCH=riscv64 BUILD_BFLOAT16=1 BUILD_HFLOAT16=1, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=256,elen=64,zfh=true,zvfh=true,zvfbfwma=true, RISCV64_ZVL256B) (push) Waiting to run
2026-03-23 23:53:57 +01:00
Martin Kroeker 3f1eac4ba0 Update version to 0.3.32 2026-03-23 23:53:05 +01:00
Martin Kroeker fd1c5ca01a Merge pull request #5705 from OpenMathLib/develop
Merge from develop for 0.3.32 release
2026-03-23 23:51:55 +01:00
Martin Kroeker 52178f70c7 Merge pull request #5703 from martin-frbg/changelog0332
Update Changelog for 0.3.32
2026-03-23 23:48:23 +01:00
Martin Kroeker f88aa7def7 Merge pull request #5702 from martin-frbg/lapack1203
Roundup LWORK and remove conversions in ?GVD (Reference-LAPACK PR 1203)
2026-03-23 20:05:17 +01:00
Martin Kroeker a24cca9576 Merge pull request #5704 from OpenMathLib/revert-5565-jn/makefile-rule-dynamic
Revert "build: fix rule for building dynamic files"
2026-03-23 20:04:08 +01:00
Martin Kroeker 7eab365219 Revert "build: fix rule for building dynamic files"
apple m / build (cmake, gfortran, 0, 0) (push) Waiting to run
apple m / build (cmake, gfortran, 0, 1) (push) Waiting to run
apple m / build (cmake, gfortran, 1, 0) (push) Waiting to run
apple m / build (cmake, gfortran, 1, 1) (push) Waiting to run
apple m / build (make, gfortran, 0, 0) (push) Waiting to run
apple m / build (make, gfortran, 0, 1) (push) Waiting to run
apple m / build (make, gfortran, 1, 0) (push) Waiting to run
apple m / build (make, gfortran, 1, 1) (push) Waiting to run
c910v qemu test / TEST (riscv64-linux-gnu, NO_SHARED=1 TARGET=C910V, C910V, riscv64-unknown-linux-gnu) (push) Waiting to run
c910v qemu test / TEST (riscv64-linux-gnu, NO_SHARED=1 TARGET=RISCV64_GENERIC, RISCV64_GENERIC, riscv64-linux-gnu) (push) Waiting to run
Run codspeed benchmarks / benchmarks (make, gfortran, ubuntu-22.04, 3.12) (push) Waiting to run
continuous build / build (cmake, clang, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, macos-latest) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang-21, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang-21, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, clang-21, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, gcc, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, gcc, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, gcc, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang, gfortran, macos-latest) (push) Waiting to run
continuous build / build (make, clang, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, clang, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang-21, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang-21, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, clang-21, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, gcc, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, gcc, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, gcc, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / msys2 (None, fc, int32, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, CLANG64, mingw-w64-clang-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, MINGW32, mingw-w64-i686) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int64, -DBINARY=64 -DINTERFACE64=1, CLANG64, mingw-w64-clang-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int64, -DBINARY=64 -DINTERFACE64=1, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / cross_build (DYNAMIC_ARCH=1 TARGET=GENERIC, mips64el, mips64el-linux-gnuabi64) (push) Waiting to run
continuous build / cross_build (TARGET=EV4, alpha, alpha-linux-gnu) (push) Waiting to run
continuous build / cross_build (TARGET=MIPS1004K, mipsel, mipsel-linux-gnu) (push) Waiting to run
continuous build / cross_build (TARGET=RISCV64_GENERIC, riscv64, riscv64-linux-gnu) (push) Waiting to run
continuous build / neoverse_build (push) Waiting to run
harmonyos / build (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=GENERIC, DYNAMIC_ARCH, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA264, LA264, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA464, LA464, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA64_GENERIC, LA64_GENERIC, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON2K1000, LOONGSON2K1000, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON3R5, LOONGSON3R5, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSONGENERIC, LOONGSONGENERIC, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=GENERIC, DYNAMIC_ARCH) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA264, LA264) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA464, LA464) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA64_GENERIC, LA64_GENERIC) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON2K1000, LOONGSON2K1000) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON3R5, LOONGSON3R5) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSONGENERIC, LOONGSONGENERIC) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=I6400, I6400, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=I6500, I6500, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=MIPS64_GENERIC, MIPS64_GENERIC, mips64el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=P6600, P6600, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=SICORTEX, SICORTEX, mips64el-linux-gnuabi64) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_GENERIC BINARY=64 ARCH=riscv64 DYNAMIC_ARCH=1, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=256,elen=64, DYNAMIC_ARCH=1) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_ZVL128B BINARY=64 ARCH=riscv64, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=128,elen=64, RISCV64_ZVL128B) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_ZVL256B BINARY=64 ARCH=riscv64 BUILD_BFLOAT16=1 BUILD_HFLOAT16=1, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=256,elen=64,zfh=true,zvfh=true,zvfbfwma=true, RISCV64_ZVL256B) (push) Waiting to run
2026-03-23 20:03:52 +01:00
Martin Kroeker 6137054da3 Update with 0.3.32 changes 2026-03-23 18:35:22 +01:00
Martin Kroeker b227de9429 Merge pull request #5701 from martin-frbg/lapack1204
Fix internal errors getting ignored in calculation of selected eigenvalues (Reference-LAPACK PR 1204)
2026-03-23 18:32:25 +01:00
Martin Kroeker 99c120916a Use ROUNDUP_LWORK and remove redundant conversions (Reference-LAPACK PR1203) 2026-03-23 16:28:25 +01:00
Martin Kroeker 1b6fc34f0c Fix error codes from ?STEBZ getting ignored, resulting in wrong output (Reference-LAPACK PR1204) 2026-03-23 15:56:50 +01:00
Martin Kroeker 51e904df27 Merge pull request #5699 from martin-frbg/issue5325
Add Q&A on calling convention to the FAQ, from issue 5325
2026-03-23 13:28:27 +01:00
Martin Kroeker 8b9b3f0f5e Merge pull request #5698 from martin-frbg/issue5638
Improve error message displayed when running out of buffers
2026-03-23 08:37:45 +01:00
Martin Kroeker 500e32818a Merge pull request #5697 from martin-frbg/ext_doc
Update documentation of BLAS extensions
2026-03-23 07:21:34 +01:00
Martin Kroeker f6d5eb7af9 Merge pull request #5565 from vtjnash/jn/makefile-rule-dynamic
build: fix rule for building dynamic files
2026-03-22 23:59:41 +01:00
Martin Kroeker 9d3ae22b28 Add section on calling convention, from issue 5325 2026-03-22 23:44:32 +01:00
Martin Kroeker 494a3f735f Improve error message displayed when running out of buffers 2026-03-22 22:46:17 +01:00
Martin Kroeker 496af0d8bb add gemm_batch, gemm_batch_strided, bgemm/bgemv and fp16 extensions 2026-03-22 22:34:27 +01:00
Martin Kroeker 1e48eca408 Merge pull request #5695 from martin-frbg/update_readme_wasm
README.md: Update cpu section and add WASM support
2026-03-22 20:14:53 +01:00
Martin Kroeker faa06bd759 Update cpu section and add WASM support 2026-03-22 00:12:15 +01:00
Martin Kroeker 81d1029950 Merge pull request #5694 from martin-frbg/lapack1191
Update step length selection in ?LAED4  fallback (Reference-LAPACK PR 1191)
2026-03-22 00:06:11 +01:00
Martin Kroeker aa6a59a32e Update step length selection in LAED4 overshoot fallback (Reference-LAPACK PR 1191) 2026-03-21 18:40:54 +01:00
Martin Kroeker 4956446ca2 Merge pull request #5692 from teddygood/wasm-sum-followup
Enable DSUM SIMD path for WASM128_GENERIC
2026-03-21 12:39:30 +01:00
Martin Kroeker a89142fd5d Merge pull request #5688 from martin-frbg/divlimit_dyn
Make PREFERRED_SIZE, GEMM_DIVIDE_LIMIT and _RATE available to DYNAMIC_ARCH builds
2026-03-20 22:23:15 +01:00
Martin Kroeker afcf70dad9 Merge pull request #5691 from martin-frbg/neov2_dotbug
Avoid potential miscompilation of the ARM64 (NeoverseV2) dot kernel
2026-03-20 16:20:43 +01:00
Martin Kroeker c9185e91ad Make GEMM_DIVIDE_RATE and GEMM_PREFERRED_SIZE available in DYNAMIC_ARCH builds 2026-03-20 15:34:04 +01:00
Martin Kroeker 0dd501d794 Add GEMM_DIVIDE_RATE and GEMM_PREFERRED_SIZE to parameters 2026-03-20 15:32:06 +01:00
Martin Kroeker 6bf687b2ef Make divide_rate and preferred_size available to DYNAMIC_ARCH too 2026-03-20 15:30:53 +01:00
Martin Kroeker 3f6e928d34 Declare result as volatile to keep compilers from optimizing it out 2026-03-20 11:32:23 +01:00
Martin Kroeker 7d4a479a29 Merge pull request #5690 from OpenMathLib/revert-5643-neov2_param
Revert "Fix SGEMM returning wrong results in multithreading on NeoverseV2"
2026-03-20 11:28:29 +01:00
Martin Kroeker 57cdef594b Revert "Fix SGEMM returning wrong results in multithreading on NeoverseV2"
apple m / build (cmake, gfortran, 0, 0) (push) Waiting to run
apple m / build (cmake, gfortran, 0, 1) (push) Waiting to run
apple m / build (cmake, gfortran, 1, 0) (push) Waiting to run
apple m / build (cmake, gfortran, 1, 1) (push) Waiting to run
apple m / build (make, gfortran, 0, 0) (push) Waiting to run
apple m / build (make, gfortran, 0, 1) (push) Waiting to run
apple m / build (make, gfortran, 1, 0) (push) Waiting to run
apple m / build (make, gfortran, 1, 1) (push) Waiting to run
c910v qemu test / TEST (riscv64-linux-gnu, NO_SHARED=1 TARGET=C910V, C910V, riscv64-unknown-linux-gnu) (push) Waiting to run
c910v qemu test / TEST (riscv64-linux-gnu, NO_SHARED=1 TARGET=RISCV64_GENERIC, RISCV64_GENERIC, riscv64-linux-gnu) (push) Waiting to run
Run codspeed benchmarks / benchmarks (make, gfortran, ubuntu-22.04, 3.12) (push) Waiting to run
continuous build / build (cmake, clang, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, macos-latest) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang-21, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang-21, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, clang-21, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, gcc, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, gcc, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, gcc, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang, gfortran, macos-latest) (push) Waiting to run
continuous build / build (make, clang, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, clang, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang-21, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang-21, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, clang-21, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, gcc, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, gcc, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, gcc, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / msys2 (None, fc, int32, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, CLANG64, mingw-w64-clang-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, MINGW32, mingw-w64-i686) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int64, -DBINARY=64 -DINTERFACE64=1, CLANG64, mingw-w64-clang-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int64, -DBINARY=64 -DINTERFACE64=1, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / cross_build (DYNAMIC_ARCH=1 TARGET=GENERIC, mips64el, mips64el-linux-gnuabi64) (push) Waiting to run
continuous build / cross_build (TARGET=EV4, alpha, alpha-linux-gnu) (push) Waiting to run
continuous build / cross_build (TARGET=MIPS1004K, mipsel, mipsel-linux-gnu) (push) Waiting to run
continuous build / cross_build (TARGET=RISCV64_GENERIC, riscv64, riscv64-linux-gnu) (push) Waiting to run
continuous build / neoverse_build (push) Waiting to run
harmonyos / build (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=GENERIC, DYNAMIC_ARCH, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA264, LA264, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA464, LA464, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA64_GENERIC, LA64_GENERIC, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON2K1000, LOONGSON2K1000, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON3R5, LOONGSON3R5, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSONGENERIC, LOONGSONGENERIC, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=GENERIC, DYNAMIC_ARCH) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA264, LA264) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA464, LA464) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA64_GENERIC, LA64_GENERIC) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON2K1000, LOONGSON2K1000) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON3R5, LOONGSON3R5) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSONGENERIC, LOONGSONGENERIC) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=I6400, I6400, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=I6500, I6500, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=MIPS64_GENERIC, MIPS64_GENERIC, mips64el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=P6600, P6600, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=SICORTEX, SICORTEX, mips64el-linux-gnuabi64) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_GENERIC BINARY=64 ARCH=riscv64 DYNAMIC_ARCH=1, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=256,elen=64, DYNAMIC_ARCH=1) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_ZVL128B BINARY=64 ARCH=riscv64, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=128,elen=64, RISCV64_ZVL128B) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_ZVL256B BINARY=64 ARCH=riscv64 BUILD_BFLOAT16=1 BUILD_HFLOAT16=1, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=256,elen=64,zfh=true,zvfh=true,zvfbfwma=true, RISCV64_ZVL256B) (push) Waiting to run
2026-03-20 11:27:28 +01:00
teddygood f0d142c4dd Enable DSUM SIMD path for WASM128_GENERIC 2026-03-20 18:39:53 +09:00
Martin Kroeker e9aab19bbc Merge pull request #5689 from teddygood/wasm-sdot-followup
Use generic dot kernels for WASM128_GENERIC
2026-03-19 17:08:58 +01:00
Martin Kroeker b7601ea92f Retrieve cpu-specific GEMM_DIVIDE_LIMIT if DYNAMIC_ARCH 2026-03-19 08:29:15 +01:00
Martin Kroeker 8f5e49556f Add GEMM_DIVIDE_LIMIT to parameters 2026-03-19 08:26:33 +01:00
Martin Kroeker d7b13fec90 Provide a default GEMM_DIVIDE_LIMIT and add it to DYNAMIC_ARCH 2026-03-19 08:25:53 +01:00
teddygood 8c3717f69a Add WASM SIMD widening path for DSDOT 2026-03-19 14:16:18 +09:00
teddygood 6f672df537 Use generic DDOT kernel for WASM128_GENERIC 2026-03-19 14:15:32 +09:00
teddygood 6bb0dbfd3c Use generic SDOT kernel for WASM128_GENERIC 2026-03-19 13:54:58 +09:00
Martin Kroeker 8d6238f52e Merge pull request #5687 from martin-frbg/issue5686-1
Improve tests for SHGEMM and BGEMM
2026-03-19 00:01:31 +01:00
Martin Kroeker adba2c3c02 Merge pull request #5685 from teddygood/wasm-intrin-backend-exp
Add a WebAssembly SIMD backend for reusable intrinsics kernels
2026-03-18 21:49:53 +01:00
Martin Kroeker e5793d8406 Split failure count between comparison to SGEMM and naive loop 2026-03-18 21:46:48 +01:00
Martin Kroeker afbd7c2b0d Reduce expected accuracy compared to naive code and silence matrix element printout 2026-03-18 21:38:39 +01:00
Martin Kroeker c9dae4c1e0 Merge pull request #5684 from martin-frbg/asan_utest
Fix utest issues flagged by Address Sanitizer
2026-03-18 18:21:48 +01:00
teddygood 99d05575d0 Enable SAXPY for WebAssembly SIMD backend 2026-03-18 21:27:45 +09:00
teddygood 7ff3588833 Refine WebAssembly SIMD backend scope 2026-03-18 17:24:02 +09:00
Martin Kroeker b3de37c96b Merge pull request #5683 from martin-frbg/fix_skx_smallgemm
Fix potential over-optimization of the AVX512 small SGEMM kernel by gcc15
2026-03-18 09:08:05 +01:00
Martin Kroeker 2bdfe31986 Free arrays after test 2026-03-17 23:22:20 +01:00
Martin Kroeker 5741aab90b Avoid resolving wild pointers in automatic search for tests 2026-03-17 23:20:51 +01:00
Martin Kroeker 79a50d80d3 Fix potential over-optimization by gcc15 2026-03-17 23:13:58 +01:00
teddygood 53d0be88f8 Add WebAssembly SIMD backend for universal intrinsics 2026-03-18 03:23:31 +09:00
Martin Kroeker 7a95460bb1 Merge pull request #5680 from teddygood/wasm128-generic-target-exp
Add WebAssembly SIMD SGEMM and DGEMM kernels
2026-03-17 14:25:39 +01:00
Martin Kroeker a1fd7a4658 Merge pull request #5677 from CheryDan/riscv/zdrot
Optimize ZROT_RVV for the non-unit-stride case
2026-03-17 11:10:32 +01:00
Martin Kroeker 66063d123a Merge pull request #5679 from martin-frbg/issue5678
Add Jasper Lake Celeron N5105 and allow default fallback to Nehalem
2026-03-16 23:43:22 +01:00
teddygood 86d1451cbe Add WebAssembly SIMD GEMM kernels 2026-03-17 05:51:54 +09:00
Martin Kroeker 99c6a74e7b Add Jasper Lake Celeron N5105 and allow default fallback to Nehalem 2026-03-16 18:33:31 +01:00
Martin Kroeker ddfbc6499b Merge pull request #5676 from martin-frbg/wasm_arch
Move the WebAssembly/Emscripten support to its own architecture and target
2026-03-16 13:52:23 +01:00
Martin Kroeker f2a89889a4 make some arrays static to fix memory requirement issues 2026-03-16 09:32:56 +01:00
daichengrong aa967ef6ba Optimize ZROT_RVV for the non-unit-stride case
Optimize the RVV implementation of ZROT when inc_x and inc_y are
non-unit strides (inc_x != 1, inc_y != 1).

Reorder several operations to reduce vector register pressure and
avoid unnecessary vector register spill to the stack. This helps GCC
keep vector values in registers and reduces redundant spill/reload
instructions, improving runtime performance.

No functional change.

Signed-off-by: daichengrong <daichengrong@iscas.ac.cn>
2026-03-16 14:22:54 +08:00
Martin Kroeker 4a888bcb73 set USE_TRMM for WASM 2026-03-15 23:07:16 +01:00
Martin Kroeker 9a00d4859c Add Makefile.wasm 2026-03-15 19:51:42 +01:00
Martin Kroeker 460f5e8c0b Add the new WebAssembly target WASM128_GENERIC 2026-03-15 19:43:48 +01:00
Martin Kroeker 705a5f2523 Reuse parameters of RISCV64_GENERIC for WASM128_GENERIC 2026-03-15 19:41:33 +01:00
Martin Kroeker 6ed4cc9c86 Add WebAssembly/Emscripten as a dedicated architecute and target 2026-03-15 19:40:14 +01:00
Martin Kroeker 319343a5fd Report WebAssembly/Emscripten as a dedicated architecture 2026-03-15 19:38:49 +01:00
Martin Kroeker ea7d134aec Add wasm settings 2026-03-15 19:37:39 +01:00
Martin Kroeker 62944c9db0 Turn WebAssembly/Emscripten support into a dedicated architecture 2026-03-15 19:37:09 +01:00
Martin Kroeker 01270a94e8 Add WebAssembly as a separate architecture 2026-03-15 19:35:49 +01:00
Martin Kroeker f590468b69 Copy generic Makefile for wasm laswp 2026-03-15 19:34:25 +01:00
Martin Kroeker cd47770092 Add support for WebAssembly architecture "wasm" 2026-03-15 19:31:51 +01:00
Martin Kroeker ef3315527f Don't include the CPUID code in WebAssembly builds 2026-03-15 19:30:13 +01:00
Martin Kroeker 48f0a0f0ec Generate WASM kernel including existing intrinsics-based kernels 2026-03-15 19:28:08 +01:00
Martin Kroeker cc64ce68c3 Create generic C KERNEL as baseline for WASM 2026-03-15 19:26:42 +01:00
Martin Kroeker 450af57a68 Merge pull request #5675 from martin-frbg/fixctestc
Fix function signatures and minor compiler warnings in the CBLAS tests
2026-03-15 18:54:04 +01:00
Martin Kroeker 86ccbe8fea Fix function signatures and minor compiler warnings 2026-03-15 02:09:22 +01:00
Martin Kroeker b95729f5b0 Merge pull request #5672 from martin-frbg/nvidia_nv3
Add support for NeoverseV3 derivatives as NEOVERSEV2
2026-03-12 02:31:10 +01:00
Martin Kroeker fdc04c0e34 Merge pull request #5673 from martin-frbg/fixup-5671
remove inadvertently committed printf from PR 5671
2026-03-11 22:13:21 +01:00
Martin Kroeker f881af5bdf remove inadvertently committed printf 2026-03-11 22:10:18 +01:00
Martin Kroeker bc69f86dba Merge pull request #5671 from martin-frbg/cpuid_x86_cache
Update x86 cache size decoding table from current sandpile.org data
2026-03-11 11:34:44 +01:00
Martin Kroeker 1ff3a1a13d Merge pull request #5670 from amritahs-ibm/handle_fp16_power
powerpc: Bypass FP16 as BF16 on Power
2026-03-11 09:32:31 +01:00
Martin Kroeker 5b1729eb6d Support NeoverseV3 derivatives from NVIDIA Jetson boards as NEOVERSEV2 2026-03-10 22:34:45 +01:00
Martin Kroeker ee70631c4d Add Neoverse V3AE CPUID from NVIDIA Jetson AGX Thor 2026-03-10 22:32:44 +01:00
Martin Kroeker 02f5f620de Update cache size decoding table with sandpile.org data 2026-03-10 22:25:48 +01:00
Amrita H S 1a708bac8a powerpc: Bypass FP16 as BF16 on Power
typedef FP16 as BF16 on Power as FP16 is
not yet supported
2026-03-08 23:32:42 -05:00
Martin Kroeker 55b16e5923 Merge pull request #5643 from martin-frbg/neov2_param
Fix SGEMM returning wrong results in multithreading on NeoverseV2
2026-03-06 11:51:08 +01:00
Martin Kroeker 37262654d9 Merge pull request #5667 from fadara01/accelerate_sve128_sbgemm
Accelerate SVE128 SBGEMM/BGEMM
2026-03-06 09:14:44 +01:00
Martin Kroeker 75e2f12dae Merge pull request #5668 from martin-frbg/issue5665
Fix CMake DYNAMIC_ARCH compilation with old compilers on ARM64
2026-03-05 20:40:48 +01:00
Martin Kroeker d073702cdf Merge pull request #5661 from martin-frbg/update_readme_java
Fix leftover wiki links in the README and add java insights from issue #5109
2026-03-05 18:30:33 +01:00
Martin Kroeker 78fd789da0 Add compiler test for SVE support 2026-03-05 17:35:15 +01:00
Fadi Arafeh f30202b705 Accelerate SVE128 SBGEMM/BGEMM
This accelerates SBGEMM/BGEMM by extending the existing 8x4 kernel to 8x8 (unrolling N by 8)

Not sure if it's a good idea to delete the previous 8x4 kernel?

Here are the speedups on single core Neoverse-V2 (SVE128) compared to prev state:

Per-shape speedup
  M=N=K=64: SBGEMM 1.164x (16.42%), BGEMM 1.133x (13.30%)
  M=N=K=128: SBGEMM 1.220x (22.02%), BGEMM 1.186x (18.56%)
  M=N=K=256: SBGEMM 1.241x (24.08%), BGEMM 1.235x (23.54%)
  M=N=K=512: SBGEMM 1.240x (23.95%), BGEMM 1.227x (22.75%)
  M=N=K=1024: SBGEMM 1.251x (25.11%), BGEMM 1.232x (23.23%)
  M=N=K=2048: SBGEMM 1.235x (23.47%), BGEMM 1.246x (24.64%)

Signed-off-by: Fadi Arafeh <fadi.arafeh@arm.com>
2026-03-05 13:50:07 +00:00
Martin Kroeker 22fc689fa7 Merge pull request #5666 from martin-frbg/issue5664
Improve processing of linker arguments in f_check
2026-03-05 14:19:43 +01:00
Martin Kroeker 91eb0a638c Avoid splitting linker args on dashes not preceded by a space 2026-03-05 10:09:29 +01:00
Martin Kroeker 1590d8baf0 fix install.md link for cortex-m 2026-03-04 23:12:33 +01:00
Martin Kroeker 98864c7c6f fix reintroduced typo again 2026-03-04 20:40:45 +01:00
Martin Kroeker db6bbc7150 Merge pull request #5660 from martin-frbg/issue5658
Rewrite the Haswell SROT/DROT kernel tail loop with AVX2 to get consistent FMA rounding
2026-03-04 18:15:34 +01:00
Martin Kroeker ecdabf9d74 Merge pull request #5663 from martin-frbg/issue5662
Move the early exit in ?GESV for NRHS=0 after the GETRF call
2026-03-04 16:14:06 +01:00
Martin Kroeker dc8b16c57c Move the early exit for NRHS=0 after the GETRF call 2026-03-04 12:21:27 +01:00
Martin Kroeker 754ad2ad4f Fix leftover wiki links and add java insights from issue 5109 2026-03-04 11:44:36 +01:00
Martin Kroeker 3166fffcec Merge pull request #5659 from lindsayad/handle-emerald-rapids
Handle Intel's emerald rapids and do some formatting in the cpuid_x86 file
2026-03-03 20:06:49 +01:00
Martin Kroeker df29cc0205 Use AVX2 in the tail loop too for consistent FMA rounding 2026-03-03 15:51:51 +01:00
Alex Lindsay 692023e364 Switch case ordering for exmodel 12 to be sorted 2026-03-02 16:23:38 -07:00
Alex Lindsay 5a534a63e8 clang-format and make cpuid_x86.c more readable 2026-03-02 15:51:46 -07:00
Alex Lindsay 303903e29c Handle emerald rapids model 2026-03-02 15:39:41 -07:00
Martin Kroeker 18638c70ef Merge pull request #5656 from martin-frbg/issue5653
Add pragma to limit optimization in POWER10 DGEMV kernel
2026-02-22 15:54:59 +01:00
Martin Kroeker ef27ec6bed Add pragma to limit optimization level 2026-02-22 13:42:41 +01:00
Martin Kroeker da0e066c9e Merge pull request #5655 from martin-frbg/intel-default-cpuid
Add feature-based fallbacks for unknown/future Intel CPUIDs
2026-02-20 22:19:25 +01:00
Martin Kroeker 1d0ca19457 Add feature-based fallbacks for unknown/future Intel cpus 2026-02-20 16:45:08 +01:00
Martin Kroeker 30cf14c548 Merge pull request #5640 from ChipKerchner/RVV_Narrow_Accumulate_FP16_GEMM
Added ability to accumulate in FP16.  Convert BF16 to FP32.  For FP16 and BF16 GEMM in RISC-V (BF16 now works for pre-RVA23)
2026-02-20 14:22:27 +01:00
Martin Kroeker b4db4a1713 Merge pull request #5654 from martin-frbg/issue5627-2
Use generic SCAL kernels for PPC970 running FreeBSD
2026-02-20 12:45:42 +01:00
Martin Kroeker 43728ade59 Merge pull request #5651 from martin-frbg/issue5650
Fix gmake build with only a subset of precision types
2026-02-20 08:02:50 +01:00
Martin Kroeker 46b963b9a0 Use generic C kernels for SCAL on FreeBSD 2026-02-19 22:46:03 +01:00
Martin Kroeker 822aae6cab Merge pull request #5652 from martin-frbg/issue5649
Fix passing of C/ZDOTC results in C-converted LAPACK  on non-Windows systems
2026-02-19 22:05:10 +01:00
Martin Kroeker dccbf18c1f fix storing of ZDOTC result on non-Windows 2026-02-19 19:45:20 +01:00
Martin Kroeker 0cfb587fde fix storing of ZDOTU result on non-Windows 2026-02-19 18:46:20 +01:00
Martin Kroeker 92fcffff54 fix storing of CDOTC result on non-Windows 2026-02-19 18:45:07 +01:00
Martin Kroeker 5a07c1b61c Delete misplaced lapack-netlib/chpgst.c 2026-02-19 18:42:30 +01:00
Martin Kroeker 11986454b3 fix storing of CDOTU result on non-Windows 2026-02-19 18:05:59 +01:00
Martin Kroeker 946a2cffec fix storing of CDOTC result on non-Windows 2026-02-19 16:53:18 +01:00
Martin Kroeker ef1c06f5eb fix storing of CDOTC result on non-Windows 2026-02-19 14:08:59 +01:00
Martin Kroeker b7542ffb3d fix storing of CDOTC result on non-Windows 2026-02-19 13:34:48 +01:00
Martin Kroeker 5d29f88fed fix storing of CDOTC result on non-Windows 2026-02-19 13:18:20 +01:00
Martin Kroeker 61db4e8191 fix storing of CDOTC result on non-Windows 2026-02-19 13:07:15 +01:00
Martin Kroeker bf0d7eaacc fix storing of CDOTC result on non-Windows 2026-02-19 12:59:46 +01:00
Martin Kroeker 1da181dac6 fix storing of CDOTC result on non-Windows 2026-02-19 12:05:27 +01:00
Martin Kroeker 4389e1de70 fix storing of CDOTC result on non-Windows 2026-02-19 11:49:15 +01:00
Martin Kroeker 1defad49b6 fix storing of CDOTC result on non-Windows 2026-02-19 11:29:18 +01:00
Martin Kroeker be4ddc752f fix storing of CDOTC result on non-Windows systems 2026-02-19 10:56:27 +01:00
Martin Kroeker 7fe8bd8046 build comparison functions for complex cases too 2026-02-18 19:12:35 +01:00
Martin Kroeker 7d431f3bb0 fix conditional build for double and complex too 2026-02-18 19:10:53 +01:00
Martin Kroeker 0e28b427f3 Add slaed3/dlaed3 to complex builds 2026-02-18 19:09:36 +01:00
Martin Kroeker 92b4d1b6f3 make SLAED/DLAED definitions available to COMPLEX too 2026-02-18 19:06:17 +01:00
Martin Kroeker 7d7a6c6708 Build the comparison functions as needed to avoid missing references 2026-02-18 11:43:43 +01:00
Martin Kroeker d0a6e36896 Fix rules for running the GEMM3M tests 2026-02-18 11:41:25 +01:00
Martin Kroeker 6e3fb2ce52 fix conditional build rule 2026-02-18 11:40:20 +01:00
Chip Kerchner efe63e7970 Add pre-RVA23 to BF16 GEMM. 2026-02-15 15:49:59 +00:00
Martin Kroeker 1ef6319990 Merge pull request #5645 from martin-frbg/cortex925-cpuid
Add CPU autodetection for Arm Cortex X925/A725
2026-02-14 20:07:24 +01:00
Chip Kerchner 1d6aa0dc31 Add dummy memsets - just in case. 2026-02-13 20:03:35 +00:00
Chip Kerchner 7a1d23400f Add flag for not converting A & B - will be used in future to do conversion during packing. 2026-02-13 19:00:41 +00:00
Chip Kerchner 1cc377ef61 Only convert B if M is greater or equal to 4. 2026-02-13 18:14:11 +00:00
Chip Kerchner 0acb60aab3 Conversion from BF16 to FP32 only once. 2026-02-13 17:55:15 +00:00
Chip Kerchner 9701a80a9f One small change. 2026-02-12 20:35:41 +00:00
Chip Kerchner 4121a22c02 Convert BF16 values once (and vectorized). 2026-02-12 18:45:39 +00:00
Martin Kroeker 1690982cf1 Merge pull request #5644 from martin-frbg/issue5641
Work around llvm failing to compile the AVX512 sgemm kernel
2026-02-12 18:04:36 +01:00
Martin Kroeker 5613deb794 Merge pull request #5646 from mattip/azure-timeout
use 100 minute timeout for azure mingw32 job
2026-02-12 16:19:37 +01:00
Martin Kroeker ea82d802e6 fix typo 2026-02-12 15:55:55 +01:00
mattip e5ba61c344 use 100 minute timeout for azure mingw32 job 2026-02-12 12:09:59 +02:00
Martin Kroeker 387be46c42 Support Cortex X925 as NeoverseV2 2026-02-12 00:43:52 +01:00
Martin Kroeker 445b11148f work around llvm failing to compile the AVX512 sgemm kernel 2026-02-12 00:10:26 +01:00
Martin Kroeker db00d5c2c9 Fix SGEMM returning wrong results in multithreading on NeoverseV2 2026-02-12 00:02:13 +01:00
Chip Kerchner 33560437f5 Convert inputs from BF16 to FP32 and use FP32 vector madds. 18% faster. 2026-02-11 19:50:48 +00:00
Chip Kerchner e3cb067bf4 Fixed MADD to use float16 values. Use LMUL = 2 in main loop. Now 1.85X faster on BananaPi. 2026-02-11 00:27:27 +00:00
Chip Kerchner 74d9fe2832 Forget to add defintion. 2026-02-10 19:00:26 +00:00
Chip Kerchner aa1cebd45b 128-bit versions. 2026-02-10 18:30:02 +00:00
Chip Kerchner b5f2a50fe9 Added ability to accumulate in FP16 for GEMM. Widens once at the end of loops. 2026-02-10 17:30:05 +00:00
Chip Kerchner 7da983ebac Merge remote-tracking branch 'origin/develop' into develop 2026-02-10 17:27:51 +00:00
Martin Kroeker 986ba29493 Merge pull request #5637 from gula00/fix-typo
docs: fix minor spelling typos
2026-02-09 09:07:03 +01:00
Qingyu Li 37f7a2e00c docs: fix minor spelling typos 2026-02-09 08:25:31 +08:00
Martin Kroeker 08381cd2f0 Merge pull request #5636 from martin-frbg/arrowhu
Add CPUID identification for Intel Arrow Lake H/U
2026-02-08 17:44:11 +01:00
Martin Kroeker 35e8eeaad3 Merge pull request #5633 from cho-m/makefile-flangnew-macos
build: fix Makefile build with LLVM flang on macOS
2026-02-08 15:18:31 +01:00
Chip Kerchner 720654ace1 Merge remote-tracking branch 'origin/develop' into develop 2026-02-06 13:20:24 +00:00
Martin Kroeker 0ae18524cd Add Arrow Lake H/U 2026-02-05 20:18:46 +01:00
Martin Kroeker 20699b1812 Merge pull request #5634 from yuanjia111/develop
Fix: Remove invalid parentheses after endif
2026-02-05 08:39:46 +01:00
yuanjia 9e42e40884 Remove accidental file tream 2026-02-04 10:12:28 +08:00
yuanjia e955736005 Fix: Remove invalid parentheses after endif 2026-02-04 09:57:48 +08:00
Michael Cho 59da821b0d build: fix Makefile build with LLVM flang on macOS 2026-02-01 15:16:11 -05:00
Chip Kerchner cb4e4ce8bb Merge remote-tracking branch 'origin' into develop 2026-01-30 17:36:01 +00:00
Martin Kroeker 1a9cf8e291 Merge pull request #5631 from martin-frbg/issue5626n
Fix CMake/LLVM compilation issues seen under Windows-on-Arm
2026-01-30 10:17:46 +01:00
Martin Kroeker 27e35d639d Merge pull request #5630 from martin-frbg/pantherlake
Add Intel Panther Lake CPUID
2026-01-30 08:24:18 +01:00
Martin Kroeker 69d92490c1 move inclusion of sme_abi header into the conditional section 2026-01-29 22:24:00 +01:00
Martin Kroeker ebc3eaf80b Need strings.h for strncasecmp prototype 2026-01-29 22:21:46 +01:00
Martin Kroeker 0d6b7fe07b Fix gcc version check in absence of gcc compiler 2026-01-29 22:20:32 +01:00
Martin Kroeker 2ddcdafc0b Merge pull request #5628 from martin-frbg/issue5627
Fix stack address of flag parameter in (pre-POWER6) POWER ?SCAL kernels
2026-01-29 20:32:46 +01:00
Martin Kroeker 2ef9819803 Add Intel Panther Lake CPUID 2026-01-29 18:51:21 +01:00
Martin Kroeker 601bdde8ec fix stack location of dummy2 flag 2026-01-27 22:40:50 +01:00
Martin Kroeker d53d2b11a9 fix stack location of dummy2 flag 2026-01-27 22:39:37 +01:00
Martin Kroeker bc3b7e749a Merge pull request #5623 from martin-frbg/issue5366
Rename the DllMain copy used in static linking to OpenBLASDllMain
2026-01-23 22:48:53 +01:00
Martin Kroeker 80995622dd Rename the DllMain copy used in static linking to OpenBLASDllMain 2026-01-22 11:19:04 +01:00
Martin Kroeker dafb996425 Merge pull request #5621 from martin-frbg/woa_sum
Provide optimized ?SUM kernels for NeoverseN1 and related
2026-01-21 11:03:55 +01:00
Martin Kroeker b6aff4754a Merge pull request #5619 from lujiaweics/fix/serialize_parallelized_syrk_function_callers
Serialize accesses to parallelized syrk functions from multiple calle…
2026-01-20 23:28:18 +01:00
Martin Kroeker 861b3db733 Reuse ?SUM kernels from ThunderX2T99 2026-01-20 15:42:09 +01:00
Martin Kroeker 71261a7b3f Trivially derive optimized S/DSUM for existing SASUM/DASUM kernels 2026-01-20 15:38:50 +01:00
lujiaweics 1f3b81e562 Serialize accesses to parallelized syrk functions from multiple callers, like it was already done for GEMM in level3_thread.c and GEMM3M in level3_gemm3m_thread.c 2026-01-20 21:31:35 +08:00
Martin Kroeker 413e609f9c Merge pull request #5618 from vtjnash/jn/zdot_thunderx2t99-ICE
arm64: fix clang ICE on Windows for thunderx2t kernels
2026-01-20 14:02:30 +01:00
Martin Kroeker a10f535803 Merge pull request #5617 from martin-frbg/fix_apple_ranlib
CI, MacOS: fix missing ranlib with latest llvm
2026-01-20 14:02:07 +01:00
Martin Kroeker 2b4eaad2a0 try to make do without ranlib on OSX 2026-01-19 21:18:12 +01:00
Martin Kroeker 5ffbf38b41 Merge pull request #5616 from 7schroet/develop
Fix Intel OpenMP flag
2026-01-19 21:04:37 +01:00
Martin Kroeker 331b9ef11f Use llvm-ranlib in gmake/llvm builds on Mac 2026-01-19 18:10:58 +01:00
Jameson NashandClaude Opus 4.5 a18a4ee08a arm64: fix clang ICE on Windows for zdot_thunderx2t99.c
Guard .align directive to avoid internal compiler error on
AArch64 Windows with clang.

See: https://github.com/llvm/llvm-project/issues/149547
See: #5076

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2026-01-19 15:36:14 +00:00
Martin Kroeker 60d03c3600 fix missing ranlib 2026-01-19 15:40:58 +01:00
Martin Kroeker d5a5c7d319 Merge pull request #5598 from moluopro/develop
build: skip tests when building for iOS
2026-01-19 14:18:15 +01:00
Niclas Schroeter 3c9858cfa0 Fix Intel OpenMP flag 2026-01-19 11:53:27 +01:00
Martin Kroeker 14594773a0 Merge pull request #5615 from martin-frbg/issue5607
Fix building without multithreading or LAPACK
2026-01-19 00:35:33 +01:00
Martin Kroeker a8a2238848 Merge pull request #5611 from martin-frbg/issue5602
Fix too small DGEMM_R for some Loongson LA464 cpus
2026-01-18 23:26:11 +01:00
Martin Kroeker 8870cfc750 Merge pull request #5609 from eschnett/patch-2
Avoid integer overflow in dynamic_riscv64.c
2026-01-18 23:25:34 +01:00
Martin Kroeker d40e19ef41 Merge pull request #5606 from botantony/openblas_config-fix-arm-gcc-build
fix: don't use `_Float16` type on GCC 12
2026-01-18 23:24:51 +01:00
Martin Kroeker 566e315f4f Make test_post_fork_async depend on LAPACK as it uses getrf 2026-01-18 19:59:49 +01:00
Martin Kroeker 8742434212 Include thread callback replacement hook in singlethreaded builds as well 2026-01-18 19:55:40 +01:00
Martin Kroeker 8ea938f03a Merge pull request #5613 from al3xtjames/gemm-smp
Fix ARMV9SME/VORTEXM4 GEMM compilation with SMP disabled
2026-01-18 14:01:59 +01:00
Martin Kroeker 67c0675cf0 Merge branch 'OpenMathLib:develop' into issue5602 2026-01-18 13:50:02 +01:00
Martin Kroeker 01657b356f Merge pull request #5614 from martin-frbg/fixcirrusbsd
Cirrus CI: fix softlink for libgfortran on freebsd
2026-01-18 13:49:39 +01:00
Martin Kroeker 70ecde3e49 Merge pull request #5610 from OpenMathLib/release-0.3.0
Merge back from release 0.3.31 to develop to copy tag
2026-01-18 13:47:36 +01:00
Martin Kroeker 3628f35251 fix libgfortran link on freebsd 2026-01-18 12:55:55 +01:00
Alex James d2906e8787 Fix ARMV9SME/VORTEXM4 GEMM compilation with SMP disabled
gemm.c currently declares gotoblas_corename in SMP-enabled builds, but
the ARMV9SME and VORTEXM4 targets call gotoblas_corename even when SMP
is disabled. Fix compilation of the ARMV9SME and VORTEXM4 targets with
SMP disabled by unconditionally declaring gotoblas_corename for
DYNAMIC_ARCH builds.
2026-01-17 22:03:29 -08:00
Martin Kroeker f298361f98 Document size restriction on GEMM_R 2026-01-17 20:58:49 +01:00
Martin Kroeker 4001d7a74f Increase LA464/16MB DGEMM_R for minimal spacing of 64 to MAX(p,q) 2026-01-17 20:53:51 +01:00
Erik Schnetter 55e853a698 Avoid integer overflow in dynamic_riscv64.c
Closes https://github.com/OpenMathLib/OpenBLAS/issues/5608.
2026-01-16 10:36:53 -05:00
botantony c077708852 fix: don't use _Float16 type on GCC 12
`_Float16` is not supported by GCC 12 on Arm64 architectures:
https://godbolt.org/z/nKbrjPTvG

Related to:
https://github.com/Homebrew/homebrew-core/pull/263008
https://github.com/Homebrew/homebrew-core/pull/263009

Signed-off-by: botantony <antonsm21@gmail.com>
2026-01-16 03:04:13 +01:00
Martin Kroeker 45e9820118 Update version to 0.3.31.dev 2026-01-16 00:09:51 +01:00
Martin Kroeker f8a9c067d8 Update version to 0.3.31.dev 2026-01-16 00:09:15 +01:00
Martin Kroeker 76f1be470c Merge pull request #5605 from OpenMathLib/develop
apple m / build (cmake, gfortran, 0, 0) (push) Waiting to run
apple m / build (cmake, gfortran, 0, 1) (push) Waiting to run
apple m / build (cmake, gfortran, 1, 0) (push) Waiting to run
apple m / build (cmake, gfortran, 1, 1) (push) Waiting to run
apple m / build (make, gfortran, 0, 0) (push) Waiting to run
apple m / build (make, gfortran, 0, 1) (push) Waiting to run
apple m / build (make, gfortran, 1, 0) (push) Waiting to run
apple m / build (make, gfortran, 1, 1) (push) Waiting to run
c910v qemu test / TEST (riscv64-linux-gnu, NO_SHARED=1 TARGET=C910V, C910V, riscv64-unknown-linux-gnu) (push) Waiting to run
c910v qemu test / TEST (riscv64-linux-gnu, NO_SHARED=1 TARGET=RISCV64_GENERIC, RISCV64_GENERIC, riscv64-linux-gnu) (push) Waiting to run
Run codspeed benchmarks / benchmarks (make, gfortran, ubuntu-22.04, 3.12) (push) Waiting to run
continuous build / build (cmake, clang, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, macos-latest) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, clang, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang-21, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, clang-21, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, clang-21, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, gcc, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (cmake, gcc, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (cmake, gcc, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang, gfortran, macos-latest) (push) Waiting to run
continuous build / build (make, clang, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, clang, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang-21, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, clang-21, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, clang-21, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / build (make, gcc, flang, ubuntu-latest) (push) Waiting to run
continuous build / build (make, gcc, gfortran, ubuntu-24.04-arm) (push) Waiting to run
continuous build / build (make, gcc, gfortran, ubuntu-latest) (push) Waiting to run
continuous build / msys2 (None, fc, int32, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, CLANG64, mingw-w64-clang-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, MINGW32, mingw-w64-i686) (push) Waiting to run
continuous build / msys2 (Release, fc, int32, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int64, -DBINARY=64 -DINTERFACE64=1, CLANG64, mingw-w64-clang-x86_64) (push) Waiting to run
continuous build / msys2 (Release, fc, int64, -DBINARY=64 -DINTERFACE64=1, UCRT64, mingw-w64-ucrt-x86_64) (push) Waiting to run
continuous build / cross_build (DYNAMIC_ARCH=1 TARGET=GENERIC, mips64el, mips64el-linux-gnuabi64) (push) Waiting to run
continuous build / cross_build (TARGET=EV4, alpha, alpha-linux-gnu) (push) Waiting to run
continuous build / cross_build (TARGET=MIPS1004K, mipsel, mipsel-linux-gnu) (push) Waiting to run
continuous build / cross_build (TARGET=RISCV64_GENERIC, riscv64, riscv64-linux-gnu) (push) Waiting to run
continuous build / neoverse_build (push) Waiting to run
harmonyos / build (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=GENERIC, DYNAMIC_ARCH, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA264, LA264, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA464, LA464, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA64_GENERIC, LA64_GENERIC, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON2K1000, LOONGSON2K1000, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON3R5, LOONGSON3R5, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSONGENERIC, LOONGSONGENERIC, loongarch64-linux-gnu) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=GENERIC, DYNAMIC_ARCH) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA264, LA264) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA464, LA464) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LA64_GENERIC, LA64_GENERIC) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON2K1000, LOONGSON2K1000) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSON3R5, LOONGSON3R5) (push) Waiting to run
loongarch64 clang qemu test / TEST (NO_SHARED=1 DYNAMIC_ARCH=1 TARGET=LOONGSONGENERIC, LOONGSONGENERIC) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=I6400, I6400, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=I6500, I6500, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=MIPS64_GENERIC, MIPS64_GENERIC, mips64el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=P6600, P6600, mipsisa64r6el-linux-gnuabi64) (push) Waiting to run
mips64 qemu test / TEST (NO_SHARED=1 TARGET=SICORTEX, SICORTEX, mips64el-linux-gnuabi64) (push) Waiting to run
Nightly-Homebrew-Build / build-OpenBLAS-with-Homebrew (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_GENERIC BINARY=64 ARCH=riscv64 DYNAMIC_ARCH=1, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=256,elen=64, DYNAMIC_ARCH=1) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_ZVL128B BINARY=64 ARCH=riscv64, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=128,elen=64, RISCV64_ZVL128B) (push) Waiting to run
riscv64 zvl256b qemu test / TEST (TARGET=RISCV64_ZVL256B BINARY=64 ARCH=riscv64 BUILD_BFLOAT16=1 BUILD_HFLOAT16=1, rv64,g=true,c=true,v=true,vext_spec=v1.0,vlen=256,elen=64,zfh=true,zvfh=true,zvfbfwma=true, RISCV64_ZVL256B) (push) Waiting to run
Merge from develop for  0.3.31 release
2026-01-15 23:57:26 +01:00
Martin Kroeker 0e7b11fbc3 Merge branch 'release-0.3.0' into develop 2026-01-15 23:57:09 +01:00
Martin Kroeker ca1aefc5f9 Update version to 0.3.31 2026-01-15 23:54:29 +01:00
Martin Kroeker 366847fc10 Update version to 0.3.31 2026-01-15 23:53:53 +01:00
Martin Kroeker 10bd0ec4c3 Merge pull request #5604 from martin-frbg/changelog0331
Update the Changelog for version 0.3.31
2026-01-15 23:47:52 +01:00
Martin Kroeker 5bb7ef1466 Update the Changelog for version 0.3.31 2026-01-15 23:47:16 +01:00
Martin Kroeker 4cd575c20f Merge pull request #5423 from martin-frbg/issue5414
Split VORTEXM4 from VORTEX target and fix SGEMM_DIRECT support for SME-capable targets
2026-01-15 23:35:44 +01:00
Martin Kroeker 6f225daf94 make VORTEXM4 MacOS-only for now 2026-01-15 19:33:29 +01:00
Martin Kroeker 55a10c748d Make VortexM4 available in DYNAMIC_ARCH on MacOS only 2026-01-15 16:31:51 +01:00
Martin Kroeker 5133aac055 Make VORTEXM4 available in DYNAMIC_ARCH on Apple 2026-01-15 16:26:03 +01:00
Martin Kroeker faa18750d6 typo fix 2026-01-15 00:04:29 +01:00
Martin Kroeker 7acf919836 typo 2026-01-15 00:03:24 +01:00
Martin Kroeker 93cd7b9238 Force linking to clang_rt_builtins when using LLVM for AppleM4 2026-01-14 23:11:13 +01:00
Martin Kroeker d49df4c579 force linking to clang_rt_builtins when using LLVM for AppleM4 2026-01-14 23:08:58 +01:00
Martin Kroeker 7ffce1c788 fix spurious change of (S)BGEMM parameters for NeoverseV1 2026-01-14 19:03:06 +01:00
Martin Kroeker 88c583ed49 Update Makefile 2026-01-14 17:00:45 +01:00
Martin Kroeker d3e4b41136 remove cpu=apple-m4 as not required and less portable 2026-01-14 13:55:16 +01:00
Martin Kroeker 6735872092 drop the cpu=apple-m4 part as nonessential 2026-01-14 11:28:19 +01:00
Martin Kroeker 6137236c0a fix os variable reference 2026-01-13 23:51:34 +01:00
Martin Kroeker fa021e1887 fix missing endif() and add AppleClang options for M4 2026-01-13 22:32:41 +01:00
Martin Kroeker bdcb9b7252 add prototype 2026-01-13 22:25:30 +01:00
Martin Kroeker 533cab235f add prototype 2026-01-13 22:24:27 +01:00
Martin Kroeker 31bb6ca7df Apple Clang requires +sme in the arch string for M4 2026-01-13 21:10:07 +01:00
Martin Kroeker 5e5f9a39ad Apple Clang absolutely needs the +sme in the arch string 2026-01-13 21:05:11 +01:00
Martin Kroeker 770ad6883d Distinguish AppleClang from LLVM on ARM64 2026-01-13 20:48:02 +01:00
Martin Kroeker 10ba0e6044 fix missing parentheses on endif 2026-01-12 23:30:46 +01:00
Martin Kroeker 3149408165 Merge branch 'OpenMathLib:develop' into issue5414 2026-01-12 23:25:39 +01:00
Martin Kroeker e07bea17c8 Merge pull request #5601 from martin-frbg/issue5336-2
Use linker response files with CMake on all Apple hardware
2026-01-12 23:23:59 +01:00
Martin Kroeker e04df1941d Use linker response files on all Apple hardware 2026-01-12 20:54:51 +01:00
Martin Kroeker 31150eb1e6 Move early exit up; don't rely on support_sme() for now 2026-01-12 15:47:33 +01:00
Martin Kroeker 0a53d91789 Move early exit up; don't rely on support_sme() for now 2026-01-12 15:44:49 +01:00
Martin Kroeker aafd3cb0db Merge branch 'OpenMathLib:develop' into issue5414 2026-01-12 00:51:25 +01:00
Martin Kroeker 01cc6df92e Merge pull request #5600 from martin-frbg/lapack1179
Fix out-of-bounds accesses in the LAPACK LIN tests (Reference-LAPACK PR 1179)
2026-01-12 00:48:23 +01:00
Martin Kroeker e776297bf9 Fix out-of-bounds accesses to the TAU array (Reference-LAPACK PR 1179) 2026-01-11 22:10:16 +01:00
Martin Kroeker 05d7c18894 Merge pull request #5599 from martin-frbg/issue5552
Temporarily use the generic C kernel for DNRM2 on SPARC
2026-01-11 22:06:59 +01:00
Martin Kroeker 4d08156266 Use the generic C kernel for DNRM2 2026-01-11 21:58:31 +01:00
Martin Kroeker 6de062cfc2 Merge branch 'OpenMathLib:develop' into issue5414 2026-01-11 17:45:11 +01:00
Martin Kroeker 52ec7faf31 Merge pull request #5554 from hideaki-motoki/issue5553_gemm_default_pqr_for_a64fx
Setting optimized `[SD]GEMM_DEFAULT_[PQR]` parameters for `A64FX`
2026-01-11 15:59:51 +01:00
Martin Kroeker 1ffea2b8c1 Merge pull request #5597 from martin-frbg/issue5503
Improve the precision of ARM64 S/CNRM2 by summing in double precision
2026-01-11 14:47:39 +01:00
moluopro a3af4cadcc build: skip tests when building for iOS 2026-01-11 20:16:42 +08:00
Martin Kroeker d1de282a4e Improve the precision of S/CNRM2 by summing in double precision 2026-01-11 13:04:00 +01:00
Martin Kroeker e5aebeaf93 Merge pull request #5596 from moluopro/develop
docs: fix iOS build script & use xcrun SDK path
2026-01-10 19:51:04 +01:00
moluopro a514760e06 Change 'make libs' back to 'make' 2026-01-10 21:14:05 +08:00
moluopro d7d1088d21 docs: fix iOS build script & use xcrun SDK path 2026-01-09 23:35:58 +08:00
Martin Kroeker a9a6edaf17 Adapt for DYNAMIC_ARCH with multiple ...preprocess symbols 2026-01-09 15:29:36 +01:00
Martin Kroeker 2d46f1ec65 Merge branch 'develop' into issue5414 2026-01-09 15:04:06 +01:00
Martin Kroeker c040d5ed86 Merge pull request #5591 from quic/topic/ssyr2k_direct_sme1
Support for SME1 based ssyr2k_direct kernel for cblas_ssyr2k level 3 API
2026-01-08 15:47:38 +01:00
Zhiqing xie 6939a43c3b Support for SME1 based ssyr2k_direct kernel for cblas_ssyr2k level 3 API 2026-01-08 11:09:04 +08:00
Martin Kroeker 20ae36ba75 Merge pull request #5595 from amritahs-ibm/fix_dgemm_warnings
Fixing warning messages in dgemm and dgemv kernels
2026-01-07 12:17:28 +01:00
Amrita H S b53d18b3ad Fixing warning messages in dgemm and dgemv kernels
Signed-off-by: Amrita H S <amritahs@linux.vnet.ibm.com>
2026-01-06 10:20:56 -06:00
Martin Kroeker 7e612b640f Merge pull request #5594 from lujiaweics/fix/symbol-suffix-missing-threads-callback-function
Fix bug where openblas_set_threads_callback_function does not support modyfing symbol prefix and suffix in shared library
2026-01-06 11:54:39 +01:00
Martin Kroeker e384396a51 Use the armv9 capability set in the compiler test for SME 2026-01-05 23:37:31 +01:00
Martin Kroeker 02bc005306 reset SVE and SME capabilities between targets 2026-01-05 19:14:50 +01:00
Martin Kroeker a18a53605e Adjust M4 options to avoid unresolved reference with non-Apple LLVM 2026-01-05 19:10:52 +01:00
Martin Kroeker 618bcbd7c0 adjust M4 options to avoid undefined references with non-Apple LLVM 2026-01-05 19:09:09 +01:00
Martin Kroeker badf4c09e2 Merge pull request #5592 from RajalakshmiSR/sgemm-p10-unroll
POWER10: Reduce sgemm loop unrolling
2026-01-05 16:19:55 +01:00
lujiaweics 879497990f Fix bug where openblas_set_threads_callback_function does not support modyfing symbol prefix and suffix in shared library 2026-01-05 21:49:04 +08:00
Rajalakshmi Srinivasaraghavan 2283fcbbe7 POWER10: Reduce sgemm loop unrolling
With GCC 14, unnecessary move and lxvp instructions appear when unrolling the inner loop for larger sizes.
Reducing the loop unroll factor restores performance to GCC 11.
2026-01-04 17:01:01 -06:00
Martin Kroeker 67fd33e729 syntax fix 2025-12-31 19:46:20 +01:00
Martin Kroeker f4383d0235 syntax fix 2025-12-31 19:41:46 +01:00
Martin Kroeker 7beba94023 Add workaround for current LLVM SME bug on Windows 2025-12-31 15:49:43 +01:00
Martin Kroeker b183182e61 Add workaround for current LLVM SME bug on Windows 2025-12-31 15:43:15 +01:00
Martin Kroeker 5c8cf37d83 Add workaround for current LLVM SME bug on Windows 2025-12-31 15:33:15 +01:00
Martin Kroeker 275eb6f7f3 Add workaround for current LLVM SME bug on Windows 2025-12-31 15:29:27 +01:00
Martin Kroeker e4344def6a Merge pull request #5505 from martin-frbg/issue5493
Rewrite lapacke headers with pre/postfixes if necessary
2025-12-31 14:19:41 +01:00
Martin Kroeker 80951a2acc Merge pull request #5534 from bartoldeman/fix-flang-fcheck
Fix f_check detection of LLVM 21 flang
2025-12-30 18:15:37 +01:00
Martin Kroeker 772741e2b1 Merge pull request #5586 from martin-frbg/issue5337
Update instructions for setting up a conda-based build environment on Windows
2025-12-30 12:38:03 +01:00
Martin Kroeker 54f7b76f20 Merge pull request #5588 from martin-frbg/cooperlake_cast
Cast the alignment parameter for Cooper Lake and Sapphire Rapids to BLASLONG
2025-12-30 10:45:13 +01:00
Martin Kroeker 579eda3778 Name openmp packages in Windows/conda build recipe 2025-12-30 00:03:40 +01:00
Martin Kroeker 83a788c387 Add BLASLONG cast to the DEFAULT_ALIGN parameter of Cooper Lake and Sapphire Rapids 2025-12-29 23:23:34 +01:00
Martin Kroeker 5e3a9922bd replace mentions of miniconda with miniforge 2025-12-27 22:50:38 +01:00
Martin Kroeker e548bda1ba Update Windows/LLVM build to use miniforge and flang_win-64 package 2025-12-27 22:33:01 +01:00
Martin Kroeker cd02751b12 Merge pull request #5548 from mayeut/ppc64le-clang
ci: add build with clang on ppc64le
2025-12-25 18:02:16 +01:00
Martin Kroeker 5766adbcad Merge pull request #5569 from OpenMathLib/revert-5479-forklock
Revert "[WIP,Testing] remove the lock around the thread shutdown function again"
2025-12-25 17:10:29 +01:00
Martin Kroeker 1f2bffb4fe Merge pull request #5551 from almayne/sgemv_ramps
Updated SGEMV ramps.
2025-12-25 12:53:36 +01:00
Martin Kroeker 067e43c1e1 Merge pull request #5575 from martin-frbg/woa-neozdot
Make the thunderx2 zdot kernel compatible with LLVM21 in Windows on Arm
2025-12-25 12:00:58 +01:00
Martin Kroeker 0ff51a40f1 Merge pull request #5579 from vtjnash/jn/YIELDING
fix define for YIELDING
2025-12-25 09:51:44 +01:00
Jameson Nash 0b2b583223 POSIX.1-2008 2025-12-24 21:47:03 -05:00
Jameson Nash fed16d638c Update common.h 2025-12-24 17:47:48 -05:00
Martin Kroeker d39b77748f Make .align conditional on not being on WoA and strip CRLF endings 2025-12-24 20:00:45 +01:00
Jameson Nash 371663f0c2 fix define for YIELDING
The intent is to define this as nop, but previously it was then
immediately overriding it for various architectures, causing a compiler
warning on Windows.
2025-12-23 13:59:29 -05:00
Martin Kroeker 097d2d98fd Merge pull request #5578 from martin-frbg/mingw-getenv
Fix potential crash on startup in CYGWIN or MINGW builds with DYNAMIC_ARCH
2025-12-23 18:56:53 +01:00
Martin Kroeker 5b0884d8e7 Use getenv for readenv_atoi in CYGWIN or MINGW builds 2025-12-22 16:31:33 +01:00
Martin Kroeker c7b0304ba3 Merge pull request #5576 from martin-frbg/issue5562-4
Fix variable shadowing in the c/z_div macros of f2c-converted LAPACK
2025-12-21 09:09:38 +01:00
Martin Kroeker 652bf6b51b Initialize local variable to remove a compiler warning 2025-12-20 18:08:03 +01:00
Martin Kroeker 9fa64b9a3d Initialise string length variables 2025-12-20 18:04:30 +01:00
Martin Kroeker b8163b65cb Fix compilation error caused by inadvertent shadowing of variables in the MSVC c/z_div macros 2025-12-20 18:02:28 +01:00
Martin Kroeker ac2c66321d remove special handling of C/ZDOT for LLVM on WoA 2025-12-19 17:04:21 +01:00
Martin Kroeker cfa28bcf71 Support compilation with LLVM for Windows on Arm 2025-12-19 17:00:47 +01:00
Martin Kroeker 6bc4276f90 Merge pull request #5574 from martin-frbg/issue5562-3
Fix MSVC macros for complex division in the f2c-translated LAPACK
2025-12-19 16:28:49 +01:00
Martin Kroeker bc52252cd5 Fix previous misedits in MSVC complex dot and fix MSVC macros for complex division 2025-12-19 01:09:10 +01:00
Martin Kroeker 5aff62eb96 Merge pull request #5572 from martin-frbg/issue5562-2
Fix cut-n-paste error introduced into the f2c-converted LAPACK with PR 5567
2025-12-16 10:08:34 +01:00
Martin Kroeker e155bc0061 Merge pull request #5571 from mattip/issue5570
fix regression due to adding bgemv interfaces
2025-12-15 23:52:41 +01:00
Martin Kroeker e1d2411545 Fix cut-n-paste error introduced with previous fix for zdotc/zdotu 2025-12-15 22:46:15 +01:00
mattip cbecf98308 fix regression due to adding bgemv interfaces 2025-12-15 21:49:14 +02:00
Jameson Nash 1607a49cb9 build: fix rule for building dynamic files
Previously the architecture-specific dynamic files were relying on the
built-in rules alone.
2025-12-11 15:11:40 -05:00
Martin Kroeker e85efb8d86 remove za from clobber lists 2025-12-03 22:40:02 +01:00
Martin Kroeker 825d3ad12e AppleClang does not define feature local_streaming 2025-11-28 23:54:16 +01:00
h-motoki 5f0735832b fix param.h: turn [sd]gemm_default_[pqr] parameters for a64fx 2025-11-28 13:27:23 +09:00
Martin Kroeker c3c857c95e fix sequence 2025-11-24 22:38:49 +01:00
Martin Kroeker 7ab8dc125d rework ARM64 SME dependency handling 2025-11-24 22:36:02 +01:00
Martin Kroeker 705259c344 remove redundant HAVE_SME 2025-11-24 22:30:36 +01:00
Martin Kroeker a683287006 rework for dynamic_arch 2025-11-24 22:24:06 +01:00
Martin Kroeker b185c9a4ce small fixes for separating sme and dummy parts 2025-11-24 22:22:14 +01:00
Martin Kroeker 4af187080a Only add dedicated VORTEXM4 if building with LLVM 2025-11-24 22:15:45 +01:00
Martin Kroeker b0bd49a064 Add compiler guard around the M4 HAVE_SME property 2025-11-24 22:07:38 +01:00
Martin Kroeker 7e44f62a09 fix sequence of arm64 sgemm_direct_performance and sgemm_direct_ab 2025-11-24 22:02:34 +01:00
Anna Mayne 8da0a1fb9c Updated SGEMV ramps. 2025-11-24 14:43:43 +00:00
Martin Kroeker 7d35bf61ba Add cpuid for Apple M5 (from a PR to the archspec project) 2025-11-24 08:21:37 +01:00
Martin Kroeker 8c0b13c41c Merge branch 'OpenMathLib:develop' into issue5414 2025-11-23 23:12:49 +01:00
Martin Kroeker 9c0965b884 Merge branch 'OpenMathLib:develop' into issue5414 2025-11-23 19:45:53 +01:00
Martin Kroeker ea85b6696f Merge branch 'OpenMathLib:develop' into issue5414 2025-11-23 10:14:07 +01:00
mayeut 4867c421ed ci: add build with clang on ppc64le 2025-11-22 20:14:55 +01:00
Bart Oldeman 71c6016206 Fix f_check detection of LLVM 21 flang
The check for GCC is confused by the GNU-stack in

```
	.file	"FIRModule"
	.text
	.globl	zhoge_
	.p2align	4
	.type	zhoge_,@function
zhoge_:
	xorps	%xmm0, %xmm0
	xorps	%xmm1, %xmm1
	retq
.Lfunc_end0:
	.size	zhoge_, .Lfunc_end0-zhoge_

	.ident	"flang version 21.1.5"
	.section	".note.GNU-stack","",@progbits
```

And displays:
```
./f_check: line 102: [: 	: integer expression expected
```

Since it expects a string with GCC anyway, better to only match
GCC and not GNU.
2025-11-15 16:06:38 +00:00
Martin Kroeker 8882409131 Merge branch 'OpenMathLib:develop' into issue5493 2025-11-04 03:54:28 -08:00
Martin Kroeker 92fe96b460 fix processing of lapacke.h 2025-11-02 13:17:31 -08:00
Martin Kroeker 682f61e8b8 Add prototype for gotoblas_corename 2025-10-19 14:37:39 -07:00
Martin Kroeker 83d3e0ed1a fix copy/paste 2025-10-19 14:16:46 -07:00
Martin Kroeker 1b591ea4ed export HAVE_SME setting and exclude VortexM4 from DYNAMIC_ARCH if gcc-compiled 2025-10-19 13:49:26 -07:00
Martin Kroeker f4ee3aec88 Allow VortexM4 on the SME fast path only with non-gcc compilers 2025-10-19 13:43:11 -07:00
Martin Kroeker e01b1094de Allow VortexM4 on the same fast path only with non-gcc compilers 2025-10-19 13:42:17 -07:00
Martin Kroeker 643a0b53b0 Allow VortexM4 on the direct_SME fast path only for clang-based compilers 2025-10-19 13:37:38 -07:00
Martin Kroeker d7b0fccbb4 Enable SME-based kernels for VortexM4 with clang-based compilers only 2025-10-19 13:34:26 -07:00
Martin Kroeker 2346d0bdc4 Add HAVE_SME for VortexM4 only with non-gcc compilers 2025-10-19 13:32:54 -07:00
Martin Kroeker 8211db6203 Don't enable SME for VortexM4 when the compiler is gcc (which does not support it w/out SVE) 2025-10-19 13:31:30 -07:00
Martin Kroeker 3d5010bf37 Fix test for pre/postfix 2025-10-16 11:35:27 +02:00
Martin Kroeker 96f34621fb Add symbol pre- and/or postfixes to lapack.h and lapacke.h 2025-10-15 09:02:18 -07:00
Martin Kroeker d539685c49 rewrite lapacke headers with pre/postfixes if necessary 2025-10-12 14:04:34 -07:00
Martin Kroeker 9bfc3612f9 Merge branch 'OpenMathLib:develop' into issue5414 2025-10-12 09:18:06 -07:00
Martin Kroeker 47a66aef0f Update limits based on benchmarking the SME code on Apple M4 2025-10-08 14:36:17 +02:00
Martin Kroeker 20f5ed1a94 Merge branch 'OpenMathLib:develop' into issue5414 2025-10-08 05:27:28 -07:00
Martin Kroeker c889558317 Rework for DYNAMIC_ARCH use and use of SGEMM functions by SSYMM 2025-10-02 07:39:24 -07:00
Martin Kroeker 4ae3e37b45 restore 2VLx2VL naming 2025-10-02 07:35:30 -07:00
Martin Kroeker b3d0bc40e9 Update Makefile.L3 2025-10-02 05:13:58 -07:00
Martin Kroeker ba9d2d29f3 remove sme from M4 Fortran flags as gfortran couples it with sve 2025-10-02 05:11:34 -07:00
Martin Kroeker fc516af155 Merge branch 'develop' into issue5414 2025-10-01 14:12:59 -07:00
Martin Kroeker 2b5d8c789d remove debugging printout 2025-08-24 13:50:08 -07:00
Martin Kroeker 1b88c9c742 remove debugging printouts 2025-08-24 13:48:22 -07:00
Martin Kroeker b4fc09e9e1 Add registers d8 to d15 to clobber lists as the code does not expressly save them 2025-08-23 14:39:27 -07:00
Martin Kroeker 8e50b8d525 Add d8 to d15 to clobber lists as the code does not expressly save them 2025-08-23 14:36:49 -07:00
Martin Kroeker 7f89c6f353 smh-based direct sgemm currently requires leading dimensions to be same as matrix dimension 2025-08-23 14:20:15 -07:00
Martin Kroeker 1ee8879c78 Add VORTEXM4 2025-08-20 09:59:32 -07:00
Martin Kroeker edaa73fd24 Hide the local 2VLx2VL symbol as static is insufficient for this with gcc 2025-08-20 06:33:28 -07:00
Martin Kroeker 501728a354 adjust register 20 accesses to 21 after moving x18 2025-08-20 06:24:38 -07:00
Martin Kroeker 107c883c8a Update SME-related kernels 2025-08-19 05:13:28 -07:00
Martin Kroeker 05dbb54362 Delete misplaced file 2025-08-19 05:12:09 -07:00
Martin Kroeker 4609732e69 Relax version number requirement for AppleClang 2025-08-18 14:54:20 -07:00
Martin Kroeker bf98e448eb Add VORTEXM4 to DYNAMIC_ARCH list 2025-08-18 14:43:08 -07:00
Martin Kroeker 0bc19a1335 Update SME kernel details 2025-08-18 14:38:16 -07:00
Martin Kroeker 426b5f23ed Add compiler options for VORTEXM4 2025-08-18 14:35:36 -07:00
Martin Kroeker 4328c91e27 relax requirements in compiler SME capability check 2025-08-18 14:34:51 -07:00
Martin Kroeker c794d0a4ce Add VORTEXM4 2025-08-18 14:33:24 -07:00
Martin Kroeker a4f5fec46e Add compiler options for VORTEXM4 2025-08-18 14:32:07 -07:00
Martin Kroeker ca542f319f Add VORTEXM4 2025-08-18 08:41:38 -07:00
Martin Kroeker 18f9582f3e Add VORTEXM4 2025-08-18 01:54:09 -07:00
Martin Kroeker 4e2a8c18e5 Split VORTEXM4 from VORTEX target due to SME support 2025-08-18 01:53:04 -07:00
Martin Kroeker 30970460b8 Add VORTEXM4 target 2025-08-18 01:52:05 -07:00
Martin Kroeker b0a00fbd62 Add minimal compiler flags for VORTEXM4 2025-08-18 01:51:10 -07:00
Martin Kroeker ccfd0170fb Enable SME on MacOS and add VORTEXM4 to DYNAMIC_ARCH list 2025-08-18 01:50:13 -07:00
Martin Kroeker ef0b883dff Add sgemm_direct_performant for ARM64 2025-08-18 01:48:08 -07:00
Martin Kroeker e76c39099a Add sgemm_direct_performant for ARM64 2025-08-18 01:47:17 -07:00
Martin Kroeker 202a7a0e2a Separate VORTEXM4 from VORTEX and ARMV9SME 2025-08-18 01:45:40 -07:00
Martin Kroeker de91afd2ae Move SGEMM_DIRECT after the CBLAS parameter check and add sgemm_direct_performant for ARM64 2025-08-18 01:44:21 -07:00
Martin Kroeker 0203657f40 Add sgemm_direct_performant for ARM64 2025-08-18 01:42:32 -07:00
Martin Kroeker e82bcd2740 Update ARM64 sgemm_direct object generation 2025-08-18 01:41:13 -07:00
Martin Kroeker 731f4dd686 Add VORTEXM4 settings 2025-08-18 01:39:35 -07:00
Martin Kroeker 53d3bb50cc Get symbol name from build system; change b.first to b.mi for AppleClang compatibility 2025-08-18 01:37:50 -07:00
Martin Kroeker 08a00326a4 Build symbol name from build system variables 2025-08-18 01:35:41 -07:00
Martin Kroeker 89898fc499 Add sgemm_direct_performant for switching between direct and regular kernels 2025-08-18 01:31:40 -07:00
Martin Kroeker 22c6607db9 Use ASMNAME to get symbol name from build system; leave x18 unused as reserved on MacOS 2025-08-18 01:30:10 -07:00
Martin Kroeker ca22e28ca1 Rename sgemm_direct_sme1.S to sgemm_direct_sme1_2VLx2VL.S 2025-08-18 01:25:44 -07:00
1435 changed files with 24344 additions and 19826 deletions
+5 -3
View File
@@ -89,14 +89,16 @@ task:
type: text/plain
macos_instance:
image: ghcr.io/cirruslabs/macos-sonoma-xcode:latest
image: ghcr.io/cirruslabs/macos-tahoe-xcode:latest
task:
name: AppleM1/LLVM armv7-androidndk xbuild
compile_script:
- brew install --cask android-ndk
- export ANDROID_NDK_HOME="/opt/homebrew/share/android-ndk"
- export CC=/opt/homebrew/share/android-ndk/toolchains/llvm/prebuilt/darwin-x86_64/bin/armv7a-linux-androideabi23-clang
- make TARGET=ARMV7 ARM_SOFTFP_ABI=1 NUM_THREADS=32 HOSTCC=clang NOFORTRAN=1 RANLIB="ls -l"
- export AR=/opt/homebrew/share/android-ndk/toolchains/llvm/prebuilt/darwin-x86_64/bin/llvm-ar
- export RANLIB=/opt/homebrew/share/android-ndk/toolchains/llvm/prebuilt/darwin-x86_64/bin/llvm-ranlib
- make TARGET=ARMV7 ARM_SOFTFP_ABI=1 NUM_THREADS=32 HOSTCC=clang NOFORTRAN=1
always:
config_artifacts:
path: "*conf*"
@@ -151,7 +153,7 @@ FreeBSD_task:
image_family: freebsd-14-3
install_script:
- pkg update -f && pkg upgrade -y && pkg install -y gmake gcc
- ln -s /usr/local/lib/gcc13/libgfortran.so.5.0.0 /usr/lib/libgfortran.so
- ln -s /usr/local/lib/gcc14/libgfortran.so.5.0.0 /usr/lib/libgfortran.so
compile_script:
- gmake CC=clang FC=gfortran USE_OPENMP=1 CPP_THREAD_SAFETY_TEST=1
+1
View File
@@ -99,6 +99,7 @@ jobs:
run: |
export CPPFLAGS="-I/opt/homebrew/opt/llvm/include"
export CC="/opt/homebrew/opt/llvm/bin/clang"
export RANLIB=llvm-ranlib
case "${{ matrix.build }}" in
"make")
make -j$(nproc) DYNAMIC_ARCH=1 USE_OPENMP=${{matrix.openmp}} INTERFACE64=${{matrix.ilp64}} FC="ccache ${{ matrix.fortran }}"
+36 -3
View File
@@ -9,7 +9,7 @@ project(OpenBLAS C ASM)
set(OpenBLAS_MAJOR_VERSION 0)
set(OpenBLAS_MINOR_VERSION 3)
set(OpenBLAS_PATCH_VERSION 30.dev)
set(OpenBLAS_PATCH_VERSION 32.dev)
set(OpenBLAS_VERSION "${OpenBLAS_MAJOR_VERSION}.${OpenBLAS_MINOR_VERSION}.${OpenBLAS_PATCH_VERSION}")
@@ -308,8 +308,8 @@ if (USE_OPENMP)
endif()
endif()
# Fix "Argument list too long" for macOS with POWERPC or Intel CPUs
if(APPLE AND (NOT CMAKE_HOST_SYSTEM_PROCESSOR STREQUAL "arm64"))
# Fix "Argument list too long" for macOS - mostly seen with older OS versions on POWERPC or Intel CPUs
if(APPLE)
# Use response files
set(CMAKE_C_USE_RESPONSE_FILE_FOR_OBJECTS 1)
# Always build static library first
@@ -708,6 +708,39 @@ if(NOT NO_LAPACKE)
COMMAND ${CMAKE_COMMAND} -E copy ${CMAKE_CURRENT_SOURCE_DIR}/lapack-netlib/LAPACKE/include/lapacke_mangling_with_flags.h.in "${CMAKE_BINARY_DIR}/lapacke_mangling.h"
)
install (FILES ${CMAKE_BINARY_DIR}/lapacke_mangling.h DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
if (NOT (x${SYMBOLPREFIX}${SYMBOLSUFFIX} STREQUAL "x"))
message (STATUS "Generating lapacke.h in ${CMAKE_INSTALL_INCLUDEDIR}")
set(LAPACKE_H ${CMAKE_BINARY_DIR}/generated/lapacke.h)
file(READ ${CMAKE_CURRENT_SOURCE_DIR}/lapack-netlib/LAPACKE/include/lapacke.h LAPACKE_H_CONTENTS)
if (NOT ${SYMBOLPREFIX} STREQUAL "")
string(REGEX REPLACE "(LAPACKE_*)" " ${SYMBOLPREFIX}\\1" LAPACKE_H_CONTENTS_NEW "${LAPACKE_H_CONTENTS}")
string(REPLACE "_ ${SYMBOLPREFIX}LAPACKE_H_" "_LAPACKE_H_" LAPACKE_H_CONTENTS ${LAPACKE_H_CONTENTS_NEW})
string(REPLACE "${SYMBOLPREFIX}LAPACKE_malloc" "LAPACKE_malloc" LAPACKE_H_CONTENTS_NEW ${LAPACKE_H_CONTENTS})
string(REPLACE "${SYMBOLPREFIX}LAPACKE_free" "LAPACKE_free" LAPACKE_H_CONTENTS ${LAPACKE_H_CONTENTS_NEW})
set(LAPACKE_H_CONTENTS_NEW ${LAPACKE_H_CONTENTS})
endif()
if (NOT ${SYMBOLSUFFIX} STREQUAL "")
string(REGEX REPLACE "(${SYMBOLPREFIX}LAPACKE_[a-z1-9]*[^ (]*)" "\\1${SYMBOLSUFFIX}" LAPACKE_H_CONTENTS_NEW "${LAPACKE_H_CONTENTS}")
string(REPLACE "#define${SYMBOLSUFFIX}" "#define" LAPACKE_H_CONTENTS ${LAPACKE_H_CONTENTS_NEW})
string(REPLACE "LAPACKE_malloc${SYMBOLSUFFIX}" "LAPACKE_malloc" LAPACKE_H_CONTENTS_NEW ${LAPACKE_H_CONTENTS})
string(REPLACE "LAPACKE_free${SYMBOLSUFFIX}" "LAPACKE_free" LAPACKE_H_CONTENTS ${LAPACKE_H_CONTENTS_NEW})
set(LAPACKE_H_CONTENTS_NEW ${LAPACKE_H_CONTENTS})
endif()
file(WRITE ${LAPACKE_H} "${LAPACKE_H_CONTENTS_NEW}")
install (FILES ${LAPACKE_H} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
message (STATUS "Generating lapack.h in ${CMAKE_INSTALL_INCLUDEDIR}")
set(LAPACK_H ${CMAKE_BINARY_DIR}/generated/lapack.h)
file(READ ${CMAKE_CURRENT_SOURCE_DIR}/lapack-netlib/LAPACKE/include/lapack.h LAPACK_H_CONTENTS)
if (NOT ${SYMBOLPREFIX} STREQUAL "")
string(REGEX REPLACE "(LAPACK_[a-z1-9]*[ \(][.\)]*)" "${SYMBOLPREFIX}\\1" LAPACK_H_CONTENTS_NEW "${LAPACK_H_CONTENTS}")
set(LAPACK_H_CONTENTS ${LAPACK_H_CONTENTS_NEW})
endif()
if (NOT ${SYMBOLSUFFIX} STREQUAL "")
string(REGEX REPLACE "(${SYMBOLPREFIX}LAPACK_[a-z1-9]*)([ \(].\)" "\\1${SYMBOLSUFFIX}\\2" LAPACK_H_CONTENTS_NEW "${LAPACK_H_CONTENTS}")
endif()
file(WRITE ${LAPACK_H} "${LAPACK_H_CONTENTS_NEW}")
install (FILES ${LAPACK_H} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
endif()
endif()
# Install pkg-config files
+8
View File
@@ -29,6 +29,9 @@
* Annop Wongwathanarat <annop.wongwathanarat@arm.com>
* Optimizations and other improvements targeting AArch64
* Anna Mayne <anna.mayne@arm.com>
* Optimizations and other improvements targeting AArch64
## Previous Developers
* Zaheer Chothia <zaheer.chothia@gmail.com>
@@ -267,3 +270,8 @@ In chronological order:
* [2025-05-29] Optimise axpby kernel for RISCV64_ZVL256B
* [2025-06-05] Optimise hbmv kernel for RISCV64_ZVL256B
* Anna Mayne <anna.mayne@arm.com>
* [2025-11-19] Update thread throttling profile for SGEMV on NEOVERSEV1 and NEOVERSEV2
* Fadi Arafeh <fadi.arafeh@arm.com>
* [2026-03-05] Accelerate SVE128 SBGEMM/BGEMM
+199
View File
@@ -1,4 +1,203 @@
OpenBLAS ChangeLog
====================================================================
Version 0.3.32
23-Mar-2026
general:
- Moved the preliminary support for a Web Assembly target to its own WASM
architecture and WASM128_GENERIC target
- Fixed a potential performance difference between dedicated compilation for
a target and its representation in DYNAMIC_ARCH builds by making additional
cpu-specific parameters available to the DYNAMIC_ARCH configuration
- Fixed the reimplementation of LAPACK ?GESV to conform to the reference (i.e.
compute the LU factorization even when NRHS is zero)
- Improved the error message that is displayed when the compile-time allocation
of memory buffers is exceeded
- Fixed a problem with non-serialized accesses to parallelized SYRK by concurrent
callers
- Fixed an ABI mismatch in the internal version of CDOT/ZDOT used by the C fallback
versions of the LAPACK source
- Improved the f_check script for detecting the Fortran compiler to handle embedded
dashes in path names
- Fixed several memory access issues in the utests that were detected by Address
Sanitizer
- Fixed Makefile errors in cases where only a subset of precision types was selected
- Fixed missing function errors in Makefile builds without LAPACK or without threads
- Fixed a syntax error in the benchmarks Makefile
- Fixed compiler warnings in the CBLAS testsuite
- Fixed the OpenMP compiler option used with the Intel Ifx compiler
- Updated the README sections on supported cpus and operating systems, and added
notes pertaining to JAVA
- Updated the documentation page for supported BLAS-like extensions
- included fixes from the Reference-LAPACK project:
- Improved step length selection in the fallback path of ?LAED4
(Reference-LAPACK PR 1191)
- Rounding up of LWORK and removal of redundant type conversions in the GVD
functions (Reference-LAPACK PR 1202)
- internal errors were getting ignored in calculation of selected eigenvalues
(Reference-LAPACK PR 1204)
arm64:
- Fixed a potential miscompilation of the SDOT/DDOT/DSDOT kernels
- Fixed DYNAMIC_ARCH compilation with CMake and compilers lacking SVE support
- Improved the performance of BGEMM and SBGEMM kernels for Neoverse V2
- Added optimized SSUM and DSUM kernels for Neoverse N1
- Added preliminary support for Neoverse V3 cpus as NEOVERSEV2
- Added cpu autodetection of Cortex A725 and X925 cpus
- Fixed a CMake build problem with flang on Mac OS
- Fixed build problems with gcc versions 12 and earlier that do not support fp16
- Fixed compilation of GEMM kernels for VORTEXM4/ARMV9SME without multithreading
- Fixed the optimized CDOT/ZDOT kernel to compile with LLVM under Windows on Arm
- Renamed the copy of the DllMain function used in static linking on MS Windows to
OpenBLASDllMain to avoid symbol name conflicts with other libraries
ioongarch64:
- fixed POTRF returning wrong results on LA464 due to a wrong parameter setting
power:
- Fixed compilation problems caused by missing support for half-precision floats (FP16)
- Fixed a potential miscompilation of the POWER10 DGEMV kernel by limiting its optimization
level
- Fixed a SCAL issue on PPCG4/PPC970 running Linux
- Worked around a SCAL issue on PPC970 running FreeBSD by switching to the generic C kernels
riscv64:
- Optimized the CROT/ZROT kernel for vector length 128 in the non-unit stride path
- Improved SBGEMM/SHGEMM and related helper functions for type conversion
- Fixed probing for BFLOAT16 support in DYNAMIC_ARCH cpu detection at runtime
x86_64:
- Fixed a potential miscompilation (by gcc 15.x) of the AVX512 SGEMM kernel for "small"
matrix sizes
- Fixed the SROT and DROT kernels for Haswell to have consistent (FMA) rounding
in the main loop and tail call
- Added automatic detection of Intel Arrow Lake H/U, Panther Lake and Jasper Lake
- Added automatic detection of Intel Emerald Rapids and upcoming cpu models
- Updated the cache size translation table in the cpu model autodetection code
- Improved cpu detection fallback to also include Nehalem as a non-AVX option
- Fixed a Makefile build issue with clang and the SkylakeX SGEMM kernel
- Renamed the copy of the DllMain function used in static linking on MS Windows to
OpenBLASDllMain to avoid symbol name conflicts with other libraries
wasm:
- Added optimized intrinsics kernels for SGEMM and DGEMM as well as DOT, ROT and SUM
====================================================================
Version 0.3.31
15-Jan-2026
general:
- reverted a matrix partitioning optimization from 0.3.30 that could lead to
race conditions and subsequent invalid results in GEMM
- added the bfloat16 extensions BGEMM and BGEMV
- added a BLAS interface for the ?GEMM_BATCH extensions
- added the BLAS extensions ?GEMM_BATCH_STRIDED and their CBLAS interface
- added the basic infrastructure for half-precision float (FP16) format
using SH prefix
- reimplemented the LAPACK SLAED3/DLAED3 function using multithreading, thereby
improving the performance of the SSYEVD/DSYEVD eigensolver for symmetric matrices
on all platforms
- limited the number of retries for initial memory allocation to avoid infinite
hanging on low-memory systems
- fixed a thread lockup situation encountered with python 3.9 or older and numpy
- introduced a problem size threshold for multithreading in STRMV/DTRMV
- introduced a problem size threshold for multithreading in CHER/CHER2/CHPR/CHPR2
and ZHER/ZHER2/ZHPR/ZHPR2
- improved the problem size thresholds for multithreading in SGER/DGER
- improved autodetection of the Fortran compiler
- fixed passing of the INTERFACE64=1 option to the flang-new compiler
- fixed a potential deadlock in multithreaded code after calling fork()
- fixed builds using CMake on FreeBSD
- fixed builds using CMake from within Cygwin on Windows
- fixed builds using CMake and the NVHPC compiler on ARM64
- fixed CMake build error from misdetecting compiler or OpenMP versions
- improved contents of the CMake-generated OpenBLASConfig.cmake file
- added support for cross-compilation to RISCV targets via CMake
- fixed cross-compilation to x86 targets from non-x86 architectures
- fixed failure to install cblas.h if NO_CBLAS=0 was specified
- fixed missing user-defined pre- and postfixes on functions in lapack.h,lapacke.h
- included fixes from the Reference-LAPACK project:
- fix ordering bug in ?LAED/?LASD (Reference-LAPACK PR 1140)
- revert changes in ?GEEV from PR 1129 (Reference-LAPACK PR 1142)
- fix workspace allocation in LAPACKE_?TRSEN (Reference-LAPACK PR 1144)
riscv:
- added optimized SBGEMM kernels for ZVL128B and ZVL256B targets
- added optimized SHGEMM kernels for ZVL128B and ZVL256B targets
- added optimized SBGEMV and SHGEMV kernels for ZVL128B/ZVL256B
- improved performance of the GEMV kernel for ZVL256B
- improved the performance of the CROT and ZROT kernels for ZVL128B and x280
- improved the detection of RVV1.0 capability
- improved performance of the matrix packing helper functions for ZVL128B and ZVL256B
- improved performance of OMATCOPY for ZVL128B and ZVL256B
arm:
- fixed spurious executable stack in the getarch utility
arm64:
- fixed spurious executable stack in the getarch utility
- fixed compiler warnings arising from the timer macro RPCC
- fixed cache size detection for Qualcomm Oryon under Windows on Arm
- fixed argument handling in the default SVE kernel for SDOT/DDOT
- building the BFLOAT16 kernels is now enabled by default
- improved the overall performance of GEMM,SYMM and HEMM on A64FX
- improved the performance of SDOT/DDOT on A64FX
- improved the multithreading performance of SDOT/DDOT on A64FX by
introduction of a throttling table matching thread count to problem size
- improved the performance of SGER/DGER on A64FX and NEOVERSEV1
- improved the multithreading performance of GEMM on A64FX and NEOVERSEV1
- improved the performance of the GEMV kernel for SVE-capable targets
- improved the multithreading performance of SGEMM on NEOVERSEV1 and V2
- added optimized SAXPY/DAXPY SVE kernels for A64FX and NEOVERSEV1
- added optimized BGEMM and BGEMV kernels for NEOVERSEV1
- added an optimized BGEMM kernel for NEOVERSEN2
- added support for the NEOVERSEV2 cpu
- added dedicated support for the Apple M4 cpu as VORTEXM4
- added optimized SGEMM/SSYMM/STRMM/SSYRK/SSYR2K for SME-capable targets
(ARMV9SME and VORTEXM4)
- improved the precision of the SNRM2 kernel
- added cpu autodetection and compiler settings for Ampere One processors
- fixed cpu autodetection for Apple M systems running Linux
- fixed building on MacOS with AppleClang,gfortran and xcode v16 or newer
- fixed several errors in the C code replacements for the complex and double
precision complex LAPACK functions that get used (only) when compiling with
Microsoft C and NOFORTRAN=1 under MS Windows
power:
- added initial support for the POWER11 architecture
- improved performance of DGEMM and DGEMV on POWER10
- fixed the default compiler flags to use "-O3" instead of the possibly unsafe
"-Ofast"
- fixed building under MacOS (for old G4 Macs) with CMake
- fixed potential miscompilation of DGEMV and other assembly kernels by gcc15.1
- fixed compilation with recent versions of flang
loongarch64:
- fixed warnings and potential inaccuracies arising from incorrect saving of registers
- fixed enumeration of logical cores on big NUMA servers
- fixed building with LLVM and the INTERFACE64=1 option
x86:
- fixed building the GEMM3M kernels for the GENERIC target
- fixed several errors in the C code replacements for the complex and double
precision complex LAPACK functions that get used (only) when compiling with
Microsoft C and NOFORTRAN=1 under MS Windows
x86_64:
- added cpu autodetection for Intel Lunar Lake (Core Ultra 200V)
- changed all ?MIN and ?MAX assembly kernels to use unaligned operations
- fixed several errors in the C code replacements for the complex and double
precision complex LAPACK functions that get used (only) when compiling with
Microsoft C and NOFORTRAN=1 under MS Windows
- fixed potential crashes in builds for Cooper Lake, Sapphire Rapids or Zen5 cpus
under MS Windows
zarch:
- added support for building with CMake
sparc:
- fixed a potential crash in the DNRM2 kernel
====================================================================
Version 0.3.30
19-Jun-2025
+21 -6
View File
@@ -1,16 +1,31 @@
pipeline {
agent {
docker {
image 'osuosl/ubuntu-ppc64le:18.04'
}
}
agent none
stages {
stage('Build') {
stage('GCC build') {
agent {
docker {
image 'osuosl/ubuntu-ppc64le:18.04' // gcc 7, gfortran 7
}
}
steps {
checkout scm
sh 'sudo apt update'
sh 'sudo apt install gfortran -y'
sh 'make clean && make'
}
}
stage('Clang build') {
agent {
docker {
image 'osuosl/ubuntu-ppc64le:20.04' // clang 10, gfortran 9
}
}
steps {
checkout scm
sh 'sudo apt update'
sh 'sudo apt install -y clang gfortran'
sh 'make clean && make CC=clang'
}
}
}
}
+19
View File
@@ -61,6 +61,11 @@ endif
ifeq ($(CORE), ARMV9SME)
CCOMMON_OPT += -march=armv9-a+sve2+sme
FCOMMON_OPT += -march=armv9-a+sve2
ifdef OS_WINDOWS
ifeq ($(C_COMPILER), CLANG)
CCOMMON_OPT += --aarch64-stack-hazard-size=0
endif
endif
endif
ifeq ($(CORE), CORTEXA53)
@@ -303,6 +308,20 @@ FCOMMON_OPT += -march=armv8.3-a
endif
endif
ifeq ($(CORE), VORTEXM4)
ifneq ($(C_COMPILER), GCC)
ifeq ($(APPLECLANG),1)
CCOMMON_OPT += -march=armv8.4-a+sme
else
CCOMMON_OPT += -march=armv8.4-a+sme
override LDFLAGS += -lclang_rt_builtins-aarch64
endif
else
CCOMMON_OPT += -march=armv8.4-a
endif
FCOMMON_OPT += -march=armv8.4-a
endif
ifeq (1, $(filter 1,$(GCCVERSIONGTEQ9) $(ISCLANG)))
ifeq ($(CORE), TSV110)
CCOMMON_OPT += -march=armv8.2-a -mtune=tsv110
+20 -2
View File
@@ -93,9 +93,27 @@ endif
ifneq ($(OSNAME), AIX)
ifneq ($(NO_LAPACKE), 1)
@cp $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapacke.h lapacke_h.tmp
ifdef SYMBOLPREFIX
@sed 's/LAPACKE_[a-z1-9].[^() ]*/$(SYMBOLPREFIX)&/g' lapacke_h.tmp > lapacke.tmp2
@mv lapacke.tmp2 lapacke_h.tmp
endif
ifdef SYMBOLSUFFIX
@sed 's/LAPACKE_[a-z1-9].[^() ]*/&$(SYMBOLSUFFIX)/g' lapacke_h.tmp > lapacke.tmp2
@mv lapacke.tmp2 lapacke_h.tmp
endif
@-install -m644 lapacke_h.tmp "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapacke.h"
@echo Copying LAPACKE header files to $(DESTDIR)$(OPENBLAS_INCLUDE_DIR)
@-install -m644 $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapack.h "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapack.h"
@-install -m644 $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapacke.h "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapacke.h"
@cp $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapack.h lapack_h.tmp
ifdef SYMBOLPREFIX
@sed 's/LAPACK_[a-z1-9]*(\.\.\.)/$(SYMBOLPREFIX)&/g' lapack_h.tmp > lapack.tmp2
@mv lapack.tmp2 lapack_h.tmp
endif
ifdef SYMBOLSUFFIX
@sed 's/\(#define $(SYMBOLPREFIX)LAPACK_[a-z1-9].*\)\((...)\)/\1$(SYMBOLSUFFIX)\2/g' lapack_h.tmp > lapack.tmp2
@mv lapack.tmp2 lapack_h.tmp
endif
@-install -m644 lapack_h.tmp "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapack.h"
@-install -m644 $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapacke_config.h "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapacke_config.h"
@-install -m644 $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapacke_mangling_with_flags.h.in "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapacke_mangling.h"
@-install -m644 $(NETLIB_LAPACK_DIR)/LAPACKE/include/lapacke_utils.h "$(DESTDIR)$(OPENBLAS_INCLUDE_DIR)/lapacke_utils.h"
+1 -1
View File
@@ -3,7 +3,7 @@
#
# This library's version
VERSION = 0.3.30.dev
VERSION = 0.3.32.dev
# If you set this prefix, the library name will be lib$(LIBNAMESUFFIX)openblas.a
# and lib$(LIBNAMESUFFIX)openblas.so, with a matching soname in the shared library
+8 -1
View File
@@ -331,6 +331,7 @@ HAVE_SSE5=
HAVE_AVX=
HAVE_AVX2=
HAVE_FMA3=
HAVE_SME=
include $(TOPDIR)/Makefile_kernel.conf
endif
@@ -427,7 +428,7 @@ ifndef MACOSX_DEPLOYMENT_TARGET
ifeq ($(ARCH), arm64)
export MACOSX_DEPLOYMENT_TARGET=11.0
export NO_SVE = 1
export NO_SME = 1
# export NO_SME = 1
else
export MACOSX_DEPLOYMENT_TARGET=10.8
endif
@@ -721,6 +722,11 @@ DYNAMIC_CORE += A64FX
endif
ifneq ($(NO_SME), 1)
DYNAMIC_CORE += ARMV9SME
ifeq ($(OSNAME), Darwin)
ifneq ($(C_COMPILER), GCC)
DYNAMIC_CORE += VORTEXM4
endif
endif
endif
DYNAMIC_CORE += THUNDERX
DYNAMIC_CORE += THUNDERX2T99
@@ -1896,6 +1902,7 @@ ifndef NO_MSA
export HAVE_MSA
export MSA_FLAGS
endif
export HAVE_SME
export KERNELDIR
export FUNCTION_PROFILE
export TARGET_CORE
+1
View File
@@ -0,0 +1 @@
CCOMMON_OPT += -msimd128
+5
View File
@@ -61,6 +61,9 @@ endif
ifeq ($(CORE), SKYLAKEX)
ifndef NO_AVX512
CCOMMON_OPT += -march=skylake-avx512
ifeq ($(C_COMPILER), CLANG)
CCOMMON_OPT += -mllvm -exhaustive-register-search
endif
ifneq ($(F_COMPILER), NAG)
FCOMMON_OPT += -march=skylake-avx512
endif
@@ -93,6 +96,7 @@ ifeq ($(C_COMPILER), GCC)
endif
endif
else ifeq ($(C_COMPILER), CLANG)
CCOMMON_OPT += -mllvm -exhaustive-register-search
# cooperlake support was added in clang 9
ifeq ($(CLANGVERSIONGTEQ9), 1)
CCOMMON_OPT += -march=cooperlake
@@ -135,6 +139,7 @@ ifeq ($(C_COMPILER), GCC)
endif
endif
else ifeq ($(C_COMPILER), CLANG)
CCOMMON_OPT += -mllvm -exhaustive-register-search
# sapphire rapids support was added in clang 12
ifeq ($(CLANGVERSIONGTEQ12), 1)
CCOMMON_OPT += -march=sapphirerapids
+24 -7
View File
@@ -148,11 +148,12 @@ Please read `GotoBLAS_01Readme.txt` for older CPU models already supported by th
- **Intel Haswell**: Optimized Level-3 and Level-2 BLAS with AVX2 and FMA on x86-64.
- **Intel Skylake-X**: Optimized Level-3 and Level-2 BLAS with AVX512 and FMA on x86-64.
- **Intel Cooper Lake**: as Skylake-X with improved BFLOAT16 support.
- **Intel Sapphire Rapids**: as Cooper Lake with improved BFLOAT16 SBGEMM kernel.
- **AMD Bobcat**: Used GotoBLAS2 Barcelona codes.
- **AMD Bulldozer**: x86-64 ?GEMM FMA4 kernels. (Thanks to Werner Saar)
- **AMD PILEDRIVER**: Uses Bulldozer codes with some optimizations.
- **AMD STEAMROLLER**: Uses Bulldozer codes with some optimizations.
- **AMD ZEN**: Uses Haswell codes with some optimizations for Zen 2/3 (use SkylakeX for Zen4)
- **AMD ZEN**: Uses Haswell codes with some optimizations for Zen 2/3, SkylakeX for Zen4, Cooperlake for Zen5
#### MIPS32
@@ -186,9 +187,13 @@ Please read `GotoBLAS_01Readme.txt` for older CPU models already supported by th
- **EMAG 8180**: preliminary support based on A57
- **Neoverse N1**: (AWS Graviton2) preliminary support
- **Neoverse V1**: (AWS Graviton3) optimized Level-3 BLAS
- **Neoverse N2**: preliminary support
- **Neoverse V2**: preliminary support
- **Apple Vortex**: preliminary support based on ThunderX2/3
- **Apple VortexM4**: preliminary support based on ThunderX2/3, SME kernels for SGEMM,SSYMM,STRMM,SSYRK,SSYR2K
- **A64FX**: preliminary support, optimized Level-3 BLAS
- **ARMV8SVE**: any ARMV8 cpu with SVE extensions
- **ARMV9SME**: any ARMV9 cpu with SVE and SME extensions
#### PPC/PPC64
@@ -249,9 +254,15 @@ e.g.:
```
The old-style TARGET=LOONGSON3R5 is still supported
#### WASM
Not a cpu target in the strict sense, but portable WebAssembly for browser-based applications and the like. See emscripten.org for the compiler and related information
- **WASM128_GENERIC**: Optimized SGEMM,DGEMM, DAXPY, SSUM/DSUM, SDOT/DDOT and SROT/DROT
### Support for multiple targets in a single library
OpenBLAS can be built for multiple targets with runtime detection of the target cpu by specifiying `DYNAMIC_ARCH=1` in Makefile.rule, on the gmake command line or as `-DDYNAMIC_ARCH=TRUE` in cmake.
OpenBLAS can be built for multiple targets with runtime detection of the target cpu by specifying `DYNAMIC_ARCH=1` in Makefile.rule, on the gmake command line or as `-DDYNAMIC_ARCH=TRUE` in cmake.
For **x86_64**, the list of targets this activates contains Prescott, Core2, Nehalem, Barcelona, Sandybridge, Bulldozer, Piledriver, Steamroller, Excavator, Haswell, Zen, SkylakeX, Cooper Lake, Sapphire Rapids. For cpu generations not included in this list, the corresponding older model is used. If you also specify `DYNAMIC_OLDER=1`, specific support for Penryn, Dunnington, Opteron, Opteron/SSE3, Bobcat, Atom and Nano is added. Finally there is an option `DYNAMIC_LIST` that allows to specify an individual list of targets to include instead of the default.
@@ -277,23 +288,29 @@ Please note that it is not possible to combine support for different architectur
### Supported OS
- **GNU/Linux**
- **MinGW or Visual Studio (CMake)/Windows**: Please read <https://github.com/xianyi/OpenBLAS/wiki/How-to-use-OpenBLAS-in-Microsoft-Visual-Studio>.
- **Darwin/macOS/OSX/iOS**: Experimental. Although GotoBLAS2 already supports Darwin, we are not OSX/iOS experts.
- **MinGW or Visual Studio (CMake)/Windows**: Please read <https://github.com/OpenMathLib/OpenBLAS/docs/nstall.md#visual-studio-native-windows-abi>.
- **Darwin/macOS/OSX/iOS**: Already supported on PPC and x86 by the original GotoBLAS, now also on ARM64 but we are not OSX/iOS experts.
- **FreeBSD**: Supported by the community. We don't actively test the library on this OS.
- **OpenBSD**: Supported by the community. We don't actively test the library on this OS.
- **NetBSD**: Supported by the community. We don't actively test the library on this OS.
- **DragonFly BSD**: Supported by the community. We don't actively test the library on this OS.
- **Android**: Supported by the community. Please read <https://github.com/xianyi/OpenBLAS/wiki/How-to-build-OpenBLAS-for-Android>.
- **AIX**: Supported on PPC up to POWER10
- **Android**: Supported by the community. Please read <https://github.com/OpenMathLib/OpenBLAS/docs/install.md#android>.
- **AIX**: Supported on PPC up to POWER10 but testing is increasingly problematic due to lack of publicly available systems
- **Haiku**: Supported by the community. We don't actively test the library on this OS.
- **SunOS**: Supported by the community. We don't actively test the library on this OS.
- **Cortex-M**: Supported by the community. Please read <https://github.com/xianyi/OpenBLAS/wiki/How-to-use-OpenBLAS-on-Cortex-M>.
- **Cortex-M**: Supported by the community. Please read <https://github.com/OpenMathLib/OpenBLAS/docs/install.md#cortex-m>.
## Usage
Statically link with `libopenblas.a` or dynamically link with `-lopenblas` if OpenBLAS was
compiled as a shared library.
### Considerations for using the library from Java
The default stack size of only 1MB may be too small, especially if you built OpenBLAS to support larger matrix sizes than provided for by the default settings. Use the -Xss option to request a larger stack size if you encounter problems.
When a Windows build of OpenBLAS was created using the MINGW gfortran (for the LAPACK parts), the java application may hang on startup due to a deadlock between the gfortran runtime library initialization and any pipes created by a Win11/SBT/Play Framework environment. Use -Djdk.console=jdk.internal.le to work around this.
### Setting the number of threads using environment variables
Environment variables are used to specify a maximum number of threads.
+5
View File
@@ -111,6 +111,7 @@ THUNDERX2T99
TSV110
THUNDERX3T110
VORTEX
VORTEXM4
A64FX
ARMV8SVE
ARMV9SME
@@ -152,3 +153,7 @@ EV6
14.CSKY
CSKY
CK860FV
15. WebAssembly/Emscripten:
WASM128_GENERIC
+4
View File
@@ -91,6 +91,7 @@ jobs:
openblas_utest.exe
- job: Windows_mingw_gmake
timeoutInMinutes: 100
pool:
vmImage: 'windows-latest'
steps:
@@ -185,6 +186,7 @@ jobs:
variables:
LD_LIBRARY_PATH: /usr/local/opt/llvm/lib
LIBRARY_PATH: /usr/local/opt/llvm/lib
RANLIB: touch
steps:
- script: |
brew update
@@ -197,6 +199,7 @@ jobs:
variables:
LD_LIBRARY_PATH: /usr/local/opt/llvm/lib
LIBRARY_PATH: /usr/local/opt/llvm/lib
RANLIB: touch
steps:
- script: |
brew update
@@ -240,6 +243,7 @@ jobs:
LD_LIBRARY_PATH: /usr/local/opt/llvm/lib
MACOS_HPCKIT_URL: https://registrationcenter-download.intel.com/akdlm/IRC_NAS/edb4dc2f-266f-47f2-8d56-21bc7764e119/m_HPCKit_p_2023.2.0.49443.dmg
LIBRARY_PATH: /usr/local/opt/llvm/lib
RANLIB: touch
MACOS_FORTRAN_COMPONENTS: intel.oneapi.mac.ifort-compiler
steps:
- script: |
+1 -1
View File
@@ -3155,7 +3155,7 @@ bgemv.$(SUFFIX) : gemv.c
$(CC) $(CFLAGS) -c -DBFLOAT16 -DBGEMM -UCOMPLEX -UDOUBLE -o $(@F) $^
sbgemv.$(SUFFIX) : gemv.c
$(CC) $(CFLAGS) -c -DBFLOAT16 -UCOMPLEX -UDOUBLE -o $(@F) $^
endif ()
endif
zgemv.$(SUFFIX) : gemv.c
$(CC) $(CFLAGS) -c -DCOMPLEX -DDOUBLE -o $(@F) $^
+16 -1
View File
@@ -23,6 +23,7 @@ config="$2"
compiler_name="$3"
shift 3
flags="$*"
is_ios=false
# First, we need to know the target OS and compiler name
{
@@ -78,6 +79,7 @@ case "$data" in *OS_CYGWIN_NT*) os=CYGWIN_NT ;; esac
case "$data" in *OS_INTERIX*) os=Interix ;; esac
case "$data" in *OS_ANDROID*) os=Android ;; esac
case "$data" in *OS_HAIKU*) os=Haiku ;; esac
case "$data" in *OS_IOS*) is_ios=true ;; esac
case "$data" in
*ARCH_X86_64*) architecture=x86_64 ;;
@@ -95,6 +97,7 @@ case "$data" in
*ARCH_RISCV64*) architecture=riscv64 ;;
*ARCH_LOONGARCH64*) architecture=loongarch64 ;;
*ARCH_CSKY*) architecture=csky ;;
*ARCH_WASM*) architecture=wasm ;;
esac
defined=0
@@ -128,7 +131,7 @@ case "$architecture" in
defined=1
;;
arm|arm64) defined=1 ;;
zarch|e2k|alpha|ia64|riscv64|loonarch64)
zarch|e2k|alpha|ia64|riscv64|loongarch64|wasm)
defined=1
BINARY=64
;;
@@ -252,6 +255,7 @@ case "$data" in
*ARCH_ZARCH*) architecture=zarch ;;
*ARCH_LOONGARCH64*) architecture=loongarch64 ;;
*ARCH_CSKY*) architecture=csky ;;
*ARCH_WASM*) architecture=wasm ;;
esac
binformat='bin32'
@@ -335,7 +339,14 @@ if [ "$architecture" = "arm64" ]; then
fi
no_sme=0
is_appleclang=0
if [ "$architecture" = "arm64" ]; then
if [ "$compiler" = "CLANG" ]; then
data=`$compiler_name --version`
case "$data" in Apple*)
is_appleclang=1
esac
fi
tmpd=$(mktemp -d 2>/dev/null || mktemp -d -t 'OBC')
tmpf="$tmpd/a.S"
printf ".text \n.global sme_test\n\nsme_test:\nsmstart\nsmstop\nret\n">> "$tmpf"
@@ -410,6 +421,8 @@ fi
[ "$os" = "Android" ] && [ "$hostos" = "Linux" ] && [ -n "$TERMUX_APP_PID" ] \
&& cross=0
[ "$is_ios" = true ] && cross=1
[ "$USE_OPENMP" != 1 ] && openmp=''
linker_L=""
@@ -469,6 +482,7 @@ done
[ "$no_avx512bf" -eq 1 ] && printf "NO_AVX512BF16=1\n"
[ "$no_avx2" -eq 1 ] && printf "NO_AVX2=1\n"
[ "$oldgcc" -eq 1 ] && printf "OLDGCC=1\n"
[ "$is_appleclang" -eq 1 ] && printf "APPLECLANG=1\n"
exit 0
}
@@ -499,6 +513,7 @@ done
[ "$no_avx512bf" -eq 1 ] && printf "NO_AVX512BF16=1\n"
[ "$no_avx2" -eq 1 ] && printf "NO_AVX2=1\n"
[ "$oldgcc" -eq 1 ] && printf "OLDGCC=1\n"
[ "$is_appleclang" -eq 1 ] && printf "APPLECLANG=1\n"
[ "$no_lsx" -eq 1 ] && printf "NO_LSX=1\n"
[ "$no_lasx" -eq 1 ] && printf "NO_LASX=1\n"
} >> "$makefile"
+7 -2
View File
@@ -40,14 +40,19 @@ if (DYNAMIC_ARCH)
endif ()
if (${CMAKE_C_COMPILER_VERSION} VERSION_GREATER_EQUAL 14) # SME ACLE supported in GCC >= 14
set(DYNAMIC_CORE ${DYNAMIC_CORE} ARMV9SME)
endif()
if (${CMAKE_C_COMPILER_ID} MATCHES "Clang" AND ${CMAKE_SYSTEM_NAME} STREQUAL "Darwin")
set(DYNAMIC_CORE ${DYNAMIC_CORE} VORTEXM4)
endif()
elseif (${CMAKE_C_COMPILER_ID} MATCHES "Clang")
if (${CMAKE_C_COMPILER_VERSION} VERSION_GREATER_EQUAL 11) # SVE ACLE supported in LLVM >= 11
set(DYNAMIC_CORE ${DYNAMIC_CORE} NEOVERSEV1 NEOVERSEN2 ARMV8SVE A64FX)
endif ()
if (${CMAKE_C_COMPILER_VERSION} VERSION_GREATER_EQUAL 19) # SME ACLE supported in LLVM >= 19
set(DYNAMIC_CORE ${DYNAMIC_CORE} ARMV9SME)
if (NOT ${CMAKE_SYSTEM_NAME} STREQUAL "Windows")
if (${CMAKE_C_COMPILER_VERSION} VERSION_GREATER_EQUAL 19 OR (${CMAKE_C_COMPILER_ID} MATCHES AppleClang AND ${CMAKE_C_COMPILER_VERSION} VERSION_GREATER_EQUAL 17) ) # SME ACLE supported in LLVM >= 19 and AppleClang >= 17
set(DYNAMIC_CORE ${DYNAMIC_CORE} ARMV9SME VORTEXM4)
endif()
endif()
endif ()
if (DYNAMIC_LIST)
set(DYNAMIC_CORE ARMV8 ${DYNAMIC_LIST})
+17
View File
@@ -315,7 +315,24 @@ if (${CORE} STREQUAL ARMV9SME)
set (CCOMMON_OPT "${CCOMMON_OPT} -tp=host")
else ()
set (CCOMMON_OPT "${CCOMMON_OPT} -march=armv9-a+sme")
if (${OSNAME} STREQUAL Windows AND ${CMAKE_C_COMPILER_ID} MATCHES "Clang" )
set (CCOMMON_OPT "${CCOMMON_OPT} --aarch64-stack-hazard-size=0")
endif ()
endif ()
endif ()
endif ()
if (${CORE} STREQUAL VORTEXM4)
if (NOT DYNAMIC_ARCH)
if (${CMAKE_C_COMPILER_ID} STREQUAL "NVC" AND NOT NO_SVE)
set (CCOMMON_OPT "${CCOMMON_OPT} -tp=host")
else ()
if (${CMAKE_C_COMPILER_ID} STREQUAL "AppleClang")
set (CCOMMON_OPT "${CCOMMON_OPT} -march=armv8.4-a+sme -mcpu=apple-m4")
else ()
set (CCOMMON_OPT "${CCOMMON_OPT} -march=armv8.4-a -mcpu=apple-m4")
endif ()
endif ()
endif ()
endif ()
+1 -1
View File
@@ -128,7 +128,7 @@ if (${F_COMPILER} STREQUAL "INTEL" OR CMAKE_Fortran_COMPILER_ID MATCHES "Intel")
endif ()
set(FCOMMON_OPT "${FCOMMON_OPT} -recursive -fp-model=consistent")
if (USE_OPENMP)
set(OpenMP_Fortran_FLAGS "-openmp" CACHE STRING "OpenMP Fortran compiler flags")
set(OpenMP_Fortran_FLAGS "-qopenmp" CACHE STRING "OpenMP Fortran compiler flags")
endif ()
endif ()
+8 -6
View File
@@ -71,7 +71,7 @@ set(SLASRC
slaqr0.f slaqr1.f slaqr2.f slaqr3.f slaqr4.f slaqr5.f
slaqtr.f slar1v.f slar2v.f ilaslr.f ilaslc.f
slarf.f slarfb.f slarfb_gett.f slarfg.f slarfgp.f slarft.f slarfx.f slarfy.f slargv.f
slarrv.f slartv.f
slarf1f.f slarf1l.f slarrv.f slartv.f
slarz.f slarzb.f slarzt.f slasy2.f
slasyf.f slasyf_rook.f slasyf_rk.f slasyf_aa.f
slatbs.f slatdf.f slatps.f slatrd.f slatrs.f slatrz.f
@@ -178,6 +178,7 @@ set(CLASRC
claqz0.f claqz1.f claqz2.f claqz3.f
claqsp.f claqsy.f clar1v.f clar2v.f ilaclr.f ilaclc.f
clarf.f clarfb.f clarfb_gett.f clarfg.f clarfgp.f clarft.f
clarf1f.f clarf1l.f
clarfx.f clarfy.f clargv.f clarnv.f clarrv.f clartg.f90 clartv.f
clarz.f clarzb.f clarzt.f clascl.f claset.f clasr.f classq.f90
clasyf.f clasyf_rook.f clasyf_rk.f clasyf_aa.f
@@ -262,7 +263,7 @@ set(DLASRC
dlaqr0.f dlaqr1.f dlaqr2.f dlaqr3.f dlaqr4.f dlaqr5.f
dlaqtr.f dlar1v.f dlar2v.f iladlr.f iladlc.f
dlarf.f dlarfb.f dlarfb_gett.f dlarfg.f dlarfgp.f dlarft.f dlarfx.f dlarfy.f
dlargv.f dlarrv.f dlartv.f
dlarf1f.f dlarf1l.f dlargv.f dlarrv.f dlartv.f
dlarz.f dlarzb.f dlarzt.f dlasy2.f
dlasyf.f dlasyf_rook.f dlasyf_rk.f dlasyf_aa.f
dlatbs.f dlatdf.f dlatps.f dlatrd.f dlatrs.f dlatrz.f
@@ -371,7 +372,7 @@ set(ZLASRC
zlaqr0.f zlaqr1.f zlaqr2.f zlaqr3.f zlaqr4.f zlaqr5.f
zlaqsp.f zlaqsy.f zlar1v.f zlar2v.f ilazlr.f ilazlc.f
zlarcm.f zlarf.f zlarfb.f zlarfb_gett.f
zlarfg.f zlarfgp.f zlarft.f
zlarfg.f zlarfgp.f zlarft.f zlarf1f.f zlarf1l.f
zlarfx.f zlarfy.f zlargv.f zlarnv.f zlarrv.f zlartg.f90 zlartv.f
zlarz.f zlarzb.f zlarzt.f zlascl.f zlaset.f zlasr.f
zlassq.f90 zlasyf.f zlasyf_rook.f zlasyf_rk.f zlasyf_aa.f
@@ -575,7 +576,7 @@ set(SLASRC
slaqr0.c slaqr1.c slaqr2.c slaqr3.c slaqr4.c slaqr5.c
slaqtr.c slar1v.c slar2v.c ilaslr.c ilaslc.c
slarf.c slarfb.c slarfb_gett.c slarfg.c slarfgp.c slarft.c slarfx.c slarfy.c slargv.c
slarrv.c slartv.c
slarf1f.c slarf1l.c slarrv.c slartv.c
slarz.c slarzb.c slarzt.c slasy2.c
slasyf.c slasyf_rook.c slasyf_rk.c slasyf_aa.c
slatbs.c slatdf.c slatps.c slatrd.c slatrs.c slatrz.c
@@ -681,6 +682,7 @@ set(CLASRC
claqr0.c claqr1.c claqr2.c claqr3.c claqr4.c claqr5.c
claqsp.c claqsy.c clar1v.c clar2v.c ilaclr.c ilaclc.c
clarf.c clarfb.c clarfb_gett.c clarfg.c clarfgp.c clarft.c
clarf1f.c clarf1l.c
clarfx.c clarfy.c clargv.c clarnv.c clarrv.c clartg.c clartv.c
clarz.c clarzb.c clarzt.c clascl.c claset.c clasr.c classq.c
clasyf.c clasyf_rook.c clasyf_rk.c clasyf_aa.c
@@ -764,7 +766,7 @@ set(DLASRC
dlaqr0.c dlaqr1.c dlaqr2.c dlaqr3.c dlaqr4.c dlaqr5.c
dlaqtr.c dlar1v.c dlar2v.c iladlr.c iladlc.c
dlarf.c dlarfb.c dlarfb_gett.c dlarfg.c dlarfgp.c dlarft.c dlarfx.c dlarfy.c
dlargv.c dlarrv.c dlartv.c
dlarf1f.c dlarf1l.c dlargv.c dlarrv.c dlartv.c
dlarz.c dlarzb.c dlarzt.c dlasy2.c
dlasyf.c dlasyf_rook.c dlasyf_rk.c dlasyf_aa.c
dlatbs.c dlatdf.c dlatps.c dlatrd.c dlatrs.c dlatrz.c
@@ -871,7 +873,7 @@ set(ZLASRC
zlaqhb.c zlaqhe.c zlaqhp.c zlaqp2.c zlaqp2rk.c zlaqp3rk.c zlaqps.c zlaqsb.c
zlaqr0.c zlaqr1.c zlaqr2.c zlaqr3.c zlaqr4.c zlaqr5.c
zlaqsp.c zlaqsy.c zlar1v.c zlar2v.c ilazlr.c ilazlc.c
zlarcm.c zlarf.c zlarfb.c zlarfb_gett.c
zlarcm.c zlarf.c zlarfb.c zlarfb_gett.c zlarf1f.c zlarf1l.c
zlarfg.c zlarfgp.c zlarft.c
zlarfx.c zlarfy.c zlargv.c zlarnv.c zlarrv.c zlartg.c zlartv.c
zlarz.c zlarzb.c zlarzt.c zlascl.c zlaset.c zlasr.c
+16 -1
View File
@@ -98,6 +98,10 @@ if (${COMPILER_ID} STREQUAL "GNU")
set(COMPILER_ID "GCC")
endif ()
if (HOST_OS STREQUAL "EMSCRIPTEN")
set (ARCH wasm)
endif()
string(TOUPPER ${ARCH} UC_ARCH)
file(WRITE ${TARGET_CONF_TEMP}
"#define OS_${HOST_OS}\t1\n"
@@ -1255,7 +1259,7 @@ endif ()
set(ZGEMM_UNROLL_M 4)
set(ZGEMM_UNROLL_N 4)
set(SYMV_P 16)
elseif ("${TCORE}" STREQUAL "VORTEX")
elseif ("${TCORE}" STREQUAL "VORTEX" OR "${TCORE}" STREQUAL "VORTEXM4")
file(APPEND ${TARGET_CONF_TEMP}
"#define ARMV8\n"
"#define L1_CODE_SIZE\t32768\n"
@@ -1500,6 +1504,15 @@ endif ()
"#define DTB_DEFAULT_ENTRIES 128\n"
"#define DTB_SIZE 4096\n"
"#define L2_ASSOCIATIVE 4\n")
elseif ("${TCORE}" STREQUAL "WASM128_GENERIC")
file(APPEND ${TARGET_CONF_TEMP}
"#define L1_DATA_SIZE 32768\n"
"#define L1_DATA_LINESIZE 32\n"
"#define L2_SIZE 1048576\n"
"#define L2_LINESIZE 32 \n"
"#define DTB_DEFAULT_ENTRIES 128\n"
"#define DTB_SIZE 4096\n"
"#define L2_ASSOCIATIVE 4\n")
elseif ("${TCORE}" STREQUAL "LA64_GENERIC")
file(APPEND ${TARGET_CONF_TEMP}
"#define DTB_DEFAULT_ENTRIES 64\n")
@@ -1639,6 +1652,8 @@ else(NOT CMAKE_CROSSCOMPILING)
unset (HAVE_VFP)
unset (HAVE_VFPV3)
unset (HAVE_VFPV4)
unset (HAVE_SVE)
unset (HAVE_SME)
message(STATUS "Running getarch")
# use the cmake binary w/ the -E param to run a shell command in a cross-platform way
+14
View File
@@ -367,11 +367,21 @@ if (${TARGET} STREQUAL NEOVERSEV1)
endif()
if (${TARGET} STREQUAL ARMV9SME)
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=armv9-a+sme -O3")
if (${CMAKE_SYSTEM_NAME} STREQUAL Windows AND ${CMAKE_C_COMPILER_ID} MATCHES "Clang")
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} --aarch64-stack-hazard-size=0")
endif()
endif()
if (${TARGET} STREQUAL VORTEXM4)
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=armv8.4-a+sme -O3")
if (${CMAKE_SYSTEM_NAME} STREQUAL Windows AND ${CMAKE_C_COMPILER_ID} MATCHES "Clang")
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} --aarch64-stack-hazard-size=0")
endif()
endif()
if (${TARGET} STREQUAL A64FX)
if (${CMAKE_C_COMPILER_ID} STREQUAL "PGI" AND NOT NO_SVE)
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -Msve-intrinsics -march=armv8.2-a+sve -mtune=a64fx")
else ()
set (GCC_VERSION 0.0)
execute_process(COMMAND ${CMAKE_C_COMPILER} -dumpversion OUTPUT_VARIABLE GCC_VERSION)
if (${GCC_VERSION} VERSION_GREATER 10.4 OR ${GCC_VERSION} VERSION_EQUAL 10.4)
set (KERNEL_DEFINITIONS "${KERNEL_DEFINITIONS} -march=armv8.2-a+sve -mtune=a64fx")
@@ -869,6 +879,10 @@ if (DEFINED ARCH)
set(USE_GEMM3M 1)
endif ()
if (EMSCRIPTEN)
set(USE_GEMM3M 0)
endif ()
if (${CORE} STREQUAL "generic")
set(USE_GEMM3M 0)
endif ()
+11
View File
@@ -40,6 +40,8 @@ if(CMAKE_CL_64 OR MINGW64)
else()
set(X86_64 1)
endif()
elseif(OS_EMSCRIPTEN)
set(WASM 1)
elseif(MINGW OR (MSVC AND NOT CMAKE_CROSSCOMPILING))
set(X86 1)
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "ppc.*|power.*|Power.*" OR (CMAKE_SYSTEM_NAME MATCHES "Darwin" AND CMAKE_OSX_ARCHITECTURES MATCHES "ppc.*"))
@@ -145,6 +147,15 @@ endif()
endif()
if (ARM64)
if (NOT NO_SVE)
file(WRITE ${PROJECT_BINARY_DIR}/sve.c "#include <arm_sve.h>\n\n int main(void){}\n")
execute_process(COMMAND ${CMAKE_C_COMPILER} -march=armv8-a+sve -c -o ${PROJECT_BINARY_DIR}/sve.o ${PROJECT_BINARY_DIR}/sve.c OUTPUT_QUIET ERROR_QUIET RESULT_VARIABLE NO_SVE)
if (NO_SVE EQUAL 1)
set (CCOMMON_OPT "${CCOMMON_OPT} -DNO_SVE")
endif()
file(REMOVE "${PROJECT_BINARY_DIR}/sve.c" "${PROJECT_BINARY_DIR}/sve.o")
endif()
if (NOT NO_SME)
file(WRITE ${PROJECT_BINARY_DIR}/sme.c ".text \n.global sme_test\n\nsme_test:\nsmstart\nsmstop\nret\n")
execute_process(COMMAND ${CMAKE_C_COMPILER} -march=armv9-a+sve2+sme -c -v -o ${PROJECT_BINARY_DIR}/sme.o ${PROJECT_BINARY_DIR}/sme.c OUTPUT_QUIET ERROR_QUIET RESULT_VARIABLE NO_SME)
+1 -1
View File
@@ -51,7 +51,7 @@ macro(ParseMakefileVars MAKEFILE_IN)
if (${OSNAME} STREQUAL Windows)
set (OSNAME WINNT)
endif ()
message(STATUS OS ${OSNAME} COMPILER ${C_COMPILER})
#message(STATUS OS ${OSNAME} COMPILER ${C_COMPILER})
set (IfElse 0)
set (ElseSeen 0)
set (SkipIfs 0)
+21 -15
View File
@@ -362,18 +362,6 @@ typedef int blasint;
#define MAX_CPU_NUMBER 2
#endif
#if defined(OS_SUNOS)
#define YIELDING thr_yield()
#endif
#if defined(OS_WINDOWS)
#if defined(_MSC_VER) && !defined(__clang__)
#define YIELDING YieldProcessor()
#else
#define YIELDING SwitchToThread()
#endif
#endif
#if defined(ARMV7) || defined(ARMV6) || defined(ARMV8) || defined(ARMV5)
#define YIELDING __asm__ __volatile__ ("nop;nop;nop;nop;nop;nop;nop;nop; \n");
#endif
@@ -398,14 +386,28 @@ typedef int blasint;
#endif
#endif
#ifdef __EMSCRIPTEN__
#if defined(ARCH_WASM)
#ifndef YIELDING
#define YIELDING
#endif
#endif
#if defined(_MSC_VER) && !defined(__clang__)
#undef YIELDING // MSVC doesn't support assembly code
#define YIELDING YieldProcessor()
#endif
#ifndef YIELDING
#if defined(OS_SUNOS)
#define YIELDING thr_yield()
#elif defined(OS_WINDOWS)
#define YIELDING SwitchToThread()
#else // assume POSIX.1-2008
#define YIELDING sched_yield()
#endif
#endif
/***
To alloc job_t on heap or stack.
@@ -498,6 +500,10 @@ please https://github.com/xianyi/OpenBLAS/issues/246
#include "common_csky.h"
#endif
#ifdef ARCH_WASM
#include "common_wasm.h"
#endif
#ifndef ASSEMBLER
#ifdef OS_WINDOWSSTORE
typedef char env_var_t[MAX_PATH];
@@ -765,7 +771,7 @@ static __inline int readenv_atoi(char *env) {
return 0;
}
#else
#ifdef OS_WINDOWS
#if defined(OS_WINDOWS) && !defined(OS_CYGWIN_NT)
static __inline int readenv_atoi(char *env) {
env_var_t p;
return readenv(p,env) ? 0 : atoi(p);
+25
View File
@@ -110,6 +110,31 @@ void ssyrk_direct_alpha_betaLT(BLASLONG N, BLASLONG K,
float beta,
float * C, BLASLONG strideC);
void ssyr2k_direct_alpha_betaUN(BLASLONG N, BLASLONG K,
float alpha,
float * A, BLASLONG strideA,
float * B, BLASLONG strideB,
float beta,
float * R, BLASLONG strideR);
void ssyr2k_direct_alpha_betaUT(BLASLONG N, BLASLONG K,
float alpha,
float * A, BLASLONG strideA,
float * B, BLASLONG strideB,
float beta,
float * R, BLASLONG strideR);
void ssyr2k_direct_alpha_betaLN(BLASLONG N, BLASLONG K,
float alpha,
float * A, BLASLONG strideA,
float * B, BLASLONG strideB,
float beta,
float * R, BLASLONG strideR);
void ssyr2k_direct_alpha_betaLT(BLASLONG N, BLASLONG K,
float alpha,
float * A, BLASLONG strideA,
float * B, BLASLONG strideB,
float beta,
float * R, BLASLONG strideR);
int sgemm_direct_performant(BLASLONG M, BLASLONG N, BLASLONG K);
int shgemm_beta(BLASLONG, BLASLONG, BLASLONG, float,
+4
View File
@@ -3159,6 +3159,8 @@ typedef struct {
#define NEG_TCOPY ZNEG_TCOPY
#define LARF_L ZLARF_L
#define LARF_R ZLARF_R
#define LAED3_SINGLE dlaed3_single
#define LAED3_PARALLEL dlaed3_parallel
#else
#define GETF2 CGETF2
#define GETRF CGETRF
@@ -3180,6 +3182,8 @@ typedef struct {
#define NEG_TCOPY CNEG_TCOPY
#define LARF_L CLARF_L
#define LARF_R CLARF_R
#define LAED3_SINGLE slaed3_single
#define LAED3_PARALLEL slaed3_parallel
#endif
#endif
+8
View File
@@ -47,6 +47,9 @@
typedef struct {
int dtb_entries;
int switch_ratio;
int divide_rate;
int divide_limit;
int preferred_size;
int offsetA, offsetB, align;
#if BUILD_HFLOAT16 == 1
int shgemm_p, shgemm_q, shgemm_r;
@@ -257,6 +260,7 @@ int (*shgemv_t) (BLASLONG, BLASLONG, float, hfloat16 *, BLASLONG, hfloat16 *, BL
#endif
#ifdef ARCH_ARM64
void (*sgemm_direct) (BLASLONG, BLASLONG, BLASLONG, float *, BLASLONG , float *, BLASLONG , float * , BLASLONG);
int (*sgemm_direct_performant) (BLASLONG M, BLASLONG N, BLASLONG K);
void (*sgemm_direct_alpha_beta) (BLASLONG, BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float * , BLASLONG);
void (*ssymm_direct_alpha_betaLU) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float * , BLASLONG);
void (*ssymm_direct_alpha_betaLL) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float * , BLASLONG);
@@ -268,6 +272,10 @@ int (*shgemv_t) (BLASLONG, BLASLONG, float, hfloat16 *, BLASLONG, hfloat16 *, BL
void (*ssyrk_direct_alpha_betaUT) (BLASLONG, BLASLONG, float, float *, BLASLONG, float, float *, BLASLONG);
void (*ssyrk_direct_alpha_betaLN) (BLASLONG, BLASLONG, float, float *, BLASLONG, float, float *, BLASLONG);
void (*ssyrk_direct_alpha_betaLT) (BLASLONG, BLASLONG, float, float *, BLASLONG, float, float *, BLASLONG);
void (*ssyr2k_direct_alpha_betaUN) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float *, BLASLONG);
void (*ssyr2k_direct_alpha_betaUT) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float *, BLASLONG);
void (*ssyr2k_direct_alpha_betaLN) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float *, BLASLONG);
void (*ssyr2k_direct_alpha_betaLT) (BLASLONG, BLASLONG, float, float *, BLASLONG, float *, BLASLONG, float, float *, BLASLONG);
#endif
+9 -1
View File
@@ -60,6 +60,10 @@
#define SSYRK_DIRECT_ALPHA_BETA_UT ssyrk_direct_alpha_betaUT
#define SSYRK_DIRECT_ALPHA_BETA_LN ssyrk_direct_alpha_betaLN
#define SSYRK_DIRECT_ALPHA_BETA_LT ssyrk_direct_alpha_betaLT
#define SSYR2K_DIRECT_ALPHA_BETA_UN ssyr2k_direct_alpha_betaUN
#define SSYR2K_DIRECT_ALPHA_BETA_UT ssyr2k_direct_alpha_betaUT
#define SSYR2K_DIRECT_ALPHA_BETA_LN ssyr2k_direct_alpha_betaLN
#define SSYR2K_DIRECT_ALPHA_BETA_LT ssyr2k_direct_alpha_betaLT
#define SGEMM_ONCOPY sgemm_oncopy
#define SGEMM_OTCOPY sgemm_otcopy
@@ -227,7 +231,7 @@
#define SGEMM_DIRECT_PERFORMANT gotoblas -> sgemm_direct_performant
#define SGEMM_DIRECT gotoblas -> sgemm_direct
#elif ARCH_ARM64
#define SGEMM_DIRECT_PERFORMANT sgemm_direct_performant
#define SGEMM_DIRECT_PERFORMANT gotoblas -> sgemm_direct_performant
#define SGEMM_DIRECT gotoblas -> sgemm_direct
#define SGEMM_DIRECT_ALPHA_BETA gotoblas -> sgemm_direct_alpha_beta
#define SSYMM_DIRECT_ALPHA_BETA_LU gotoblas -> ssymm_direct_alpha_betaLU
@@ -240,6 +244,10 @@
#define SSYRK_DIRECT_ALPHA_BETA_UT gotoblas -> ssyrk_direct_alpha_betaUT
#define SSYRK_DIRECT_ALPHA_BETA_LN gotoblas -> ssyrk_direct_alpha_betaLN
#define SSYRK_DIRECT_ALPHA_BETA_LT gotoblas -> ssyrk_direct_alpha_betaLT
#define SSYR2K_DIRECT_ALPHA_BETA_UN gotoblas -> ssyr2k_direct_alpha_betaUN
#define SSYR2K_DIRECT_ALPHA_BETA_UT gotoblas -> ssyr2k_direct_alpha_betaUT
#define SSYR2K_DIRECT_ALPHA_BETA_LN gotoblas -> ssyr2k_direct_alpha_betaLN
#define SSYR2K_DIRECT_ALPHA_BETA_LT gotoblas -> ssyr2k_direct_alpha_betaLT
#endif
#define SGEMM_ONCOPY gotoblas -> sgemm_oncopy
+91
View File
@@ -0,0 +1,91 @@
/*****************************************************************************
Copyright (c) 2011-2014, The OpenBLAS Project
All rights reserved.
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the following conditions are
met:
1. Redistributions of source code must retain the above copyright
notice, this list of conditions and the following disclaimer.
2. Redistributions in binary form must reproduce the above copyright
notice, this list of conditions and the following disclaimer in
the documentation and/or other materials provided with the
distribution.
3. Neither the name of the OpenBLAS project nor the names of
its contributors may be used to endorse or promote products
derived from this software without specific prior written
permission.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
**********************************************************************************/
/*********************************************************************/
/* Copyright 2009, 2010 The University of Texas at Austin. */
/* All rights reserved. */
/* */
/* Redistribution and use in source and binary forms, with or */
/* without modification, are permitted provided that the following */
/* conditions are met: */
/* */
/* 1. Redistributions of source code must retain the above */
/* copyright notice, this list of conditions and the following */
/* disclaimer. */
/* */
/* 2. Redistributions in binary form must reproduce the above */
/* copyright notice, this list of conditions and the following */
/* disclaimer in the documentation and/or other materials */
/* provided with the distribution. */
/* */
/* THIS SOFTWARE IS PROVIDED BY THE UNIVERSITY OF TEXAS AT */
/* AUSTIN ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, */
/* INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF */
/* MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE */
/* DISCLAIMED. IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT */
/* AUSTIN OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, */
/* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES */
/* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE */
/* GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR */
/* BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF */
/* LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT */
/* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT */
/* OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
/* POSSIBILITY OF SUCH DAMAGE. */
/* */
/* The views and conclusions contained in the software and */
/* documentation are those of the authors and should not be */
/* interpreted as representing official policies, either expressed */
/* or implied, of The University of Texas at Austin. */
/*********************************************************************/
#ifndef COMMON_WASM
#define COMMON_WASM
#define MB __sync_synchronize()
#define WMB __sync_synchronize()
#define RMB __sync_synchronize()
#ifndef ASSEMBLER
static inline int blas_quickdivide(blasint x, blasint y){
return x / y;
}
#endif
#define BUFFER_SIZE ( 16 << 20)
#define SEEK_ADDRESS
#endif
+31 -3
View File
@@ -82,6 +82,7 @@ size_t length64=sizeof(value64);
#define CPU_AMPERE1 25
// Apple
#define CPU_VORTEX 13
#define CPU_VORTEXM4 26
// Fujitsu
#define CPU_A64FX 15
// Phytium
@@ -113,7 +114,8 @@ static char *cpuname[] = {
"FT2000",
"CORTEXA76",
"NEOVERSEV2",
"AMPERE1"
"AMPERE1",
"VORTEXM4",
};
static char *cpuname_lower[] = {
@@ -143,7 +145,7 @@ static char *cpuname_lower[] = {
"cortexa76",
"neoversev2",
"ampere1",
"ampere1a"
"vortexm4"
};
static int cpulowperf=0;
@@ -321,6 +323,8 @@ int detect(void)
return CPU_CORTEXX2;
else if (strstr(cpu_part, "0xd4f")) //NVIDIA Grace et al.
return CPU_NEOVERSEV2;
else if (strstr(cpu_part, "0xd87") || strstr(cpu_part, "0xd85") || strstr(cpu_part, "0xd83")) // X925/A725
return CPU_NEOVERSEV2;
else if (strstr(cpu_part, "0xd0b"))
return CPU_CORTEXA76;
}
@@ -402,7 +406,8 @@ int detect(void)
if (value64 ==131287967|| value64 == 458787763 ) return CPU_VORTEX; //A12/M1
if (value64 == 3660830781) return CPU_VORTEX; //A15/M2
if (value64 == 2271604202) return CPU_VORTEX; //A16/M3
if (value64 == 1867590060) return CPU_VORTEX; //M4
if (value64 == 1867590060) return CPU_VORTEXM4; //M4
if (value64 == 492472296) return CPU_VORTEXM4; //M5
#else
#ifdef OS_WINDOWS
HKEY reghandle;
@@ -749,6 +754,29 @@ void get_cpuconfig(void)
break;
case CPU_VORTEX:
printf("#define VORTEX \n");
#ifdef __APPLE__
length64 = sizeof(value64);
sysctlbyname("hw.l1icachesize",&value64,&length64,NULL,0);
printf("#define L1_CODE_SIZE %lld \n",value64);
length64 = sizeof(value64);
sysctlbyname("hw.cachelinesize",&value64,&length64,NULL,0);
printf("#define L1_CODE_LINESIZE %lld \n",value64);
printf("#define L1_DATA_LINESIZE %lld \n",value64);
length64 = sizeof(value64);
sysctlbyname("hw.l1dcachesize",&value64,&length64,NULL,0);
printf("#define L1_DATA_SIZE %lld \n",value64);
length64 = sizeof(value64);
sysctlbyname("hw.l2cachesize",&value64,&length64,NULL,0);
printf("#define L2_SIZE %lld \n",value64);
#endif
printf("#define DTB_DEFAULT_ENTRIES 64 \n");
printf("#define DTB_SIZE 4096 \n");
break;
case CPU_VORTEXM4:
printf("#define VORTEXM4 \n");
#ifdef __clang__
printf("#define HAVE_SME 1 \n");
#endif
#ifdef __APPLE__
length64 = sizeof(value64);
sysctlbyname("hw.l1icachesize",&value64,&length64,NULL,0);
+40 -41
View File
@@ -1,5 +1,5 @@
/*****************************************************************************
Copyright (c) 2011-2014, The OpenBLAS Project
Copyright (c) 2011-2026, The OpenBLAS Project
All rights reserved.
Redistribution and use in source and binary forms, with or without
@@ -13,9 +13,9 @@ met:
notice, this list of conditions and the following disclaimer in
the documentation and/or other materials provided with the
distribution.
3. Neither the name of the OpenBLAS project nor the names of
its contributors may be used to endorse or promote products
derived from this software without specific prior written
3. Neither the name of the OpenBLAS project nor the names of
its contributors may be used to endorse or promote products
derived from this software without specific prior written
permission.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
@@ -109,7 +109,7 @@ int detect(void){
return CPU_1004K;
} else if (strstr(p, " 24K")) {
return CPU_24K;
} else
} else
return CPU_UNKNOWN;
}
#endif
@@ -136,6 +136,40 @@ void get_subdirname(void){
printf("mips");
}
int get_feature(char *search) {
#ifdef __linux
FILE *infile;
char buffer[2048], *p, *t;
p = (char *)NULL;
infile = fopen("/proc/cpuinfo", "r");
while (fgets(buffer, sizeof(buffer), infile)) {
if (!strncmp("Features", buffer, 8) ||
!strncmp("ASEs implemented", buffer, 16)) {
p = strchr(buffer, ':') + 2;
break;
}
}
fclose(infile);
if (p == NULL)
return 0;
t = strtok(p, " ");
while (t = strtok(NULL, " ")) {
if (strstr(t, search)) {
return (1);
}
}
#endif
return (0);
}
void get_cpuconfig(void){
if(detect()==CPU_P5600){
printf("#define P5600\n");
@@ -165,7 +199,7 @@ void get_cpuconfig(void){
}else{
printf("#define UNKNOWN\n");
}
#ifndef NO_MSA
#ifndef NO_MSA
if (get_feature("msa")) printf("#define HAVE_MSA\n");
#endif
}
@@ -181,38 +215,3 @@ void get_libname(void){
printf("mips\n");
}
}
int get_feature(char *search)
{
#ifdef __linux
FILE *infile;
char buffer[2048], *p,*t;
p = (char *) NULL ;
infile = fopen("/proc/cpuinfo", "r");
while (fgets(buffer, sizeof(buffer), infile))
{
if (!strncmp("Features", buffer, 8) || !strncmp("ASEs implemented", buffer, 16))
{
p = strchr(buffer, ':') + 2;
break;
}
}
fclose(infile);
if( p == NULL ) return 0;
t = strtok(p," ");
while( t = strtok(NULL," "))
{
if (strstr(t, search)) { return(1); }
}
#endif
return(0);
}
+39 -40
View File
@@ -1,5 +1,5 @@
/*****************************************************************************
Copyright (c) 2011-2014, The OpenBLAS Project
Copyright (c) 2011-2026, The OpenBLAS Project
All rights reserved.
Redistribution and use in source and binary forms, with or without
@@ -13,9 +13,9 @@ met:
notice, this list of conditions and the following disclaimer in
the documentation and/or other materials provided with the
distribution.
3. Neither the name of the OpenBLAS project nor the names of
its contributors may be used to endorse or promote products
derived from this software without specific prior written
3. Neither the name of the OpenBLAS project nor the names of
its contributors may be used to endorse or promote products
derived from this software without specific prior written
permission.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
@@ -145,13 +145,47 @@ void get_subarchitecture(void){
printf("SICORTEX");
}else{
printf("MIPS64_GENERIC");
}
}
}
void get_subdirname(void){
printf("mips64");
}
int get_feature(char *search) {
#ifdef __linux
FILE *infile;
char buffer[2048], *p, *t;
p = (char *)NULL;
infile = fopen("/proc/cpuinfo", "r");
while (fgets(buffer, sizeof(buffer), infile)) {
if (!strncmp("Features", buffer, 8) ||
!strncmp("ASEs implemented", buffer, 16)) {
p = strchr(buffer, ':') + 2;
break;
}
}
fclose(infile);
if (p == NULL)
return 0;
t = strtok(p, " ");
while (t = strtok(NULL, " ")) {
if (strstr(t, search)) {
return (1);
}
}
#endif
return (0);
}
void get_cpuconfig(void){
if(detect()==CPU_LOONGSON3R3) {
printf("#define LOONGSON3R3\n");
@@ -228,38 +262,3 @@ void get_libname(void){
printf("mips64_generic\n");
}
}
int get_feature(char *search)
{
#ifdef __linux
FILE *infile;
char buffer[2048], *p,*t;
p = (char *) NULL ;
infile = fopen("/proc/cpuinfo", "r");
while (fgets(buffer, sizeof(buffer), infile))
{
if (!strncmp("Features", buffer, 8) || !strncmp("ASEs implemented", buffer, 16))
{
p = strchr(buffer, ':') + 2;
break;
}
}
fclose(infile);
if( p == NULL ) return 0;
t = strtok(p," ");
while( t = strtok(NULL," "))
{
if (strstr(t, search)) { return(1); }
}
#endif
return(0);
}
+1694 -1755
View File
File diff suppressed because it is too large Load Diff
+4 -1
View File
@@ -178,7 +178,10 @@ ARCH_CSKY
#endif
#if defined(__EMSCRIPTEN__)
ARCH_RISCV64
ARCH_WASM
OS_WINDOWS
#endif
#if defined(TARGET_OS_IPHONE)
OS_IOS
#endif
+8 -15
View File
@@ -23,17 +23,10 @@ typedef struct { real r, i; } complex;
typedef struct { doublereal r, i; } doublecomplex;
#ifdef _MSC_VER
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
#else
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
#endif
#define pCf(z) (*_pCf(z))
#define pCd(z) (*_pCd(z))
typedef int logical;
typedef short int shortlogical;
typedef char logical1;
@@ -440,12 +433,12 @@ static real c_b43 = (float)1.;
extern /* Subroutine */ int ctest_(integer*, complex*, complex*, complex*, real*);
static complex mwpcs[5], mwpct[5];
extern /* Subroutine */ int itest1_(integer*, integer*), stest1_(real*,real*,real*,real*);
extern /* Subroutine */ int cscaltest_(integer*, complex*, complex*, integer*);
extern /* Subroutine */ void cscaltest_(integer*, complex*, complex*, integer*);
static complex cx[8];
extern real scnrm2test_(integer*, complex*, integer*);
static integer np1;
extern integer icamaxtest_(integer*, complex*, integer*);
extern /* Subroutine */ int csscaltest_(integer*, real*, complex*, integer*);
extern /* Subroutine */ void csscaltest_(integer*, real*, complex*, integer*);
extern real scasumtest_(integer*, complex*, integer*);
static integer len;
@@ -468,7 +461,7 @@ static real c_b43 = (float)1.;
i__1 = len;
for (i__ = 1; i__ <= i__1; ++i__) {
i__2 = i__ - 1;
i__3 = i__ + (np1 + combla_1.incx * 5 << 3) - 49;
i__3 = i__ + ((np1 + combla_1.incx * 5) << 3) - 49;
cx[i__2].r = cv[i__3].r, cx[i__2].i = cv[i__3].i;
/* L20: */
}
@@ -483,13 +476,13 @@ static real c_b43 = (float)1.;
} else if (combla_1.icase == 8) {
/* .. CSCAL .. */
cscaltest_(&combla_1.n, &ca, cx, &combla_1.incx);
ctest_(&len, cx, &ctrue5[(np1 + combla_1.incx * 5 << 3) - 48],
&ctrue5[(np1 + combla_1.incx * 5 << 3) - 48], sfac);
ctest_(&len, cx, &ctrue5[((np1 + combla_1.incx * 5) << 3) - 48],
&ctrue5[((np1 + combla_1.incx * 5) << 3) - 48], sfac);
} else if (combla_1.icase == 9) {
/* .. CSSCALTEST .. */
csscaltest_(&combla_1.n, &sa, cx, &combla_1.incx);
ctest_(&len, cx, &ctrue6[(np1 + combla_1.incx * 5 << 3) - 48],
&ctrue6[(np1 + combla_1.incx * 5 << 3) - 48], sfac);
ctest_(&len, cx, &ctrue6[((np1 + combla_1.incx * 5) << 3) - 48],
&ctrue6[((np1 + combla_1.incx * 5) << 3) - 48], sfac);
} else if (combla_1.icase == 10) {
/* .. ICAMAXTEST .. */
i__1 = icamaxtest_(&combla_1.n, cx, &combla_1.incx);
@@ -737,7 +730,7 @@ static real c_b43 = (float)1.;
static complex ctemp;
extern /* Subroutine */ int ctest_(integer*, complex*, complex*, complex*, real*);
static integer ksize;
extern /* Subroutine */ int cdotctest_(integer*, complex*, integer*, complex*, integer*,complex*), ccopytest_(integer*, complex*, integer*, complex*, integer*), cdotutest_(integer*, complex*, integer*, complex*, integer*, complex*),
extern /* Subroutine */ void cdotctest_(integer*, complex*, integer*, complex*, integer*,complex*), ccopytest_(integer*, complex*, integer*, complex*, integer*), cdotutest_(integer*, complex*, integer*, complex*, integer*, complex*),
cswaptest_(integer*, complex*, integer*, complex*, integer*), caxpytest_(integer*, complex*, complex*, integer*, complex*, integer*);
static integer ki, kn;
static complex cx[7], cy[7];
+32 -46
View File
@@ -23,17 +23,12 @@ typedef struct { real r, i; } complex;
typedef struct { doublereal r, i; } doublecomplex;
#ifdef _MSC_VER
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
#else
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
#endif
#define pCf(z) (*_pCf(z))
#define pCd(z) (*_pCd(z))
typedef int logical;
typedef short int shortlogical;
typedef char logical1;
@@ -319,7 +314,7 @@ static logical c_false = FALSE_;
static char snamet[12];
static real thresh;
static logical rorder;
extern /* Subroutine */ void cc2chke_(char*, ftnlen);
extern /* Subroutine */ void cc2chke_(char*);
static integer layout;
static logical ltestt, tsterr;
static complex alf[7];
@@ -712,7 +707,7 @@ L100:
ftnlen)12);
/* Test error exits. */
if (tsterr) {
cc2chke_(snames[isnum - 1], (ftnlen)12);
cc2chke_(snames[isnum - 1]);
}
/* Test computations. */
infoc_1.infot = 0;
@@ -892,8 +887,8 @@ L240:
static integer ia, ib, ic;
static logical banded;
static integer nc, nd, im, in, kl, ml, nk, nl, ku, ix, iy, ms, lx, ly, ns;
extern /* Subroutine */ int ccgbmv_(integer*, char*, integer*, integer*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*, ftnlen);
extern /* Subroutine */ void ccgemv_(integer*, char*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*, ftnlen);
extern /* Subroutine */ void ccgbmv_(integer*, char*, integer*, integer*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*);
extern /* Subroutine */ void ccgemv_(integer*, char*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*);
extern logical lceres_(char*, char*, integer*, integer*, complex*, complex*, integer*, ftnlen, ftnlen);
static char ctrans[14];
static real errmax;
@@ -1142,8 +1137,7 @@ L240:
}
ccgemv_(iorder, trans, &m, &n, &alpha,
&aa[1], &lda, &xx[1], &incx,
&beta, &yy[1], &incy, (ftnlen)
1);
&beta, &yy[1], &incy);
} else if (banded) {
if (*trace) {
/*
@@ -1158,8 +1152,7 @@ L240:
}
ccgbmv_(iorder, trans, &m, &n, &kl, &
ku, &alpha, &aa[1], &lda, &xx[
1], &incx, &beta, &yy[1], &
incy, (ftnlen)1);
1], &incx, &beta, &yy[1], &incy);
}
/* Check if error-exit was taken incorrectly. */
@@ -1347,10 +1340,10 @@ L140:
static integer nc, ik, in;
static logical packed;
static integer nk, ks, ix, iy, ns, lx, ly;
extern /* Subroutine */ void cchbmv_(integer*, char*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*, ftnlen);
extern /* Subroutine */ void cchemv_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*, ftnlen);
extern /* Subroutine */ void cchbmv_(integer*, char*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*);
extern /* Subroutine */ void cchemv_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*);
extern logical lceres_(char*, char*, integer*, integer*, complex*, complex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cchpmv_(integer*, char*, integer*, complex*, complex*, complex*, integer*, complex*, complex*, integer*, ftnlen);
extern /* Subroutine */ void cchpmv_(integer*, char*, integer*, complex*, complex*, complex*, integer*, complex*, complex*, integer*);
static real errmax;
static complex transl;
static integer laa, lda;
@@ -1566,7 +1559,7 @@ L140:
}
cchemv_(iorder, uplo, &n, &alpha, &aa[1],
&lda, &xx[1], &incx, &beta, &yy[1]
, &incy, (ftnlen)1);
, &incy);
} else if (banded) {
if (*trace) {
/*
@@ -1581,7 +1574,7 @@ L140:
}
cchbmv_(iorder, uplo, &n, &k, &alpha, &aa[
1], &lda, &xx[1], &incx, &beta, &
yy[1], &incy, (ftnlen)1);
yy[1], &incy);
} else if (packed) {
if (*trace) {
/*
@@ -1596,7 +1589,7 @@ L140:
}
cchpmv_(iorder, uplo, &n, &alpha, &aa[1],
&xx[1], &incx, &beta, &yy[1], &
incy, (ftnlen)1);
incy);
}
/* Check if error-exit was taken incorrectly. */
@@ -1792,15 +1785,15 @@ L130:
static logical packed;
static integer nk, ks, ix, ns, lx;
extern logical lceres_(char*, char*, integer*, integer*, complex*, complex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cctbmv_(integer*, char*, char*, char*, integer*, integer*, complex*, integer*, complex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cctbsv_(integer*, char*, char*, char*, integer*, integer*, complex*, integer*, complex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cctbmv_(integer*, char*, char*, char*, integer*, integer*, complex*, integer*, complex*, integer*);
extern /* Subroutine */ void cctbsv_(integer*, char*, char*, char*, integer*, integer*, complex*, integer*, complex*, integer*);
static char ctrans[14];
extern /* Subroutine */ void cctpmv_(integer*, char*, char*, char*, integer*, complex*, complex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cctpmv_(integer*, char*, char*, char*, integer*, complex*, complex*, integer*);
static real errmax;
extern /* Subroutine */ void cctrmv_(integer*, char*, char*, char*, integer*, complex*, integer*, complex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cctpsv_(integer*, char*, char*, char*, integer*, complex*, complex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cctrmv_(integer*, char*, char*, char*, integer*, complex*, integer*, complex*, integer*);
extern /* Subroutine */ void cctpsv_(integer*, char*, char*, char*, integer*, complex*, complex*, integer*);
static complex transl;
extern /* Subroutine */ void cctrsv_(integer*, char*, char*, char*, integer*, complex*, integer*, complex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cctrsv_(integer*, char*, char*, char*, integer*, complex*, integer*, complex*, integer*);
static char transs[1];
static integer laa, icd, lda;
extern logical lce_(complex*, complex*, integer*);
@@ -2010,8 +2003,7 @@ L130:
f_rew(&al__1);*/
}
cctrmv_(iorder, uplo, trans, diag, &n, &
aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
aa[1], &lda, &xx[1], &incx);
} else if (banded) {
if (*trace) {
/*
@@ -2025,8 +2017,7 @@ L130:
f_rew(&al__1);*/
}
cctbmv_(iorder, uplo, trans, diag, &n, &k,
&aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
&aa[1], &lda, &xx[1], &incx);
} else if (packed) {
if (*trace) {
/*
@@ -2040,8 +2031,7 @@ L130:
f_rew(&al__1);*/
}
cctpmv_(iorder, uplo, trans, diag, &n, &
aa[1], &xx[1], &incx, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
aa[1], &xx[1], &incx);
}
} else if (s_cmp(sname + 9, "sv", (ftnlen)2, (
ftnlen)2) == 0) {
@@ -2058,8 +2048,7 @@ L130:
f_rew(&al__1);*/
}
cctrsv_(iorder, uplo, trans, diag, &n, &
aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
aa[1], &lda, &xx[1], &incx);
} else if (banded) {
if (*trace) {
/*
@@ -2073,8 +2062,7 @@ L130:
f_rew(&al__1);*/
}
cctbsv_(iorder, uplo, trans, diag, &n, &k,
&aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
&aa[1], &lda, &xx[1], &incx);
} else if (packed) {
if (*trace) {
/*
@@ -2088,8 +2076,7 @@ L130:
f_rew(&al__1);*/
}
cctpsv_(iorder, uplo, trans, diag, &n, &
aa[1], &xx[1], &incx, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
aa[1], &xx[1], &incx);
}
}
@@ -2634,10 +2621,10 @@ L150:
static char uplo[1];
static integer i__, j, n;
extern /* Subroutine */ int cmake_(char*, char*, char*, integer*, integer*, complex*, integer*, complex*, integer*, integer*, integer*, logical*, complex*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void ccher_(integer*, char*, integer*, real*, complex*, integer*, complex*, integer*, ftnlen);
extern /* Subroutine */ void ccher_(integer*, char*, integer*, real*, complex*, integer*, complex*, integer*);
static complex alpha, w[1];
static logical isame[13];
extern /* Subroutine */ void cchpr_(integer*, char*, integer*, real*, complex*, integer*, complex*, ftnlen);
extern /* Subroutine */ void cchpr_(integer*, char*, integer*, real*, complex*, integer*, complex*);
extern /* Subroutine */ int cmvch_(char*, integer*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, complex*, integer*, complex*, real*, complex*, real*, real*, logical*, integer*, logical*, ftnlen);
static integer nargs;
static logical reset;
@@ -2812,7 +2799,7 @@ L150:
f_rew(&al__1);*/
}
ccher_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[
1], &lda, (ftnlen)1);
1], &lda);
} else if (packed) {
if (*trace) {
/*
@@ -2825,8 +2812,7 @@ L150:
al__1.aunit = *ntra;
f_rew(&al__1);*/
}
cchpr_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[
1], (ftnlen)1);
cchpr_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[1]);
}
/* Check if error-exit was taken incorrectly. */
@@ -3005,8 +2991,8 @@ L130:
static integer incxs, incys;
static logical upper;
static char uplos[1];
extern /* Subroutine */ void ccher2_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, integer*, ftnlen);
extern /* Subroutine */ void cchpr2_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, ftnlen);
extern /* Subroutine */ void ccher2_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*, integer*);
extern /* Subroutine */ void cchpr2_(integer*, char*, integer*, complex*, complex*, integer*, complex*, integer*, complex*);
static integer ia, ja, ic, nc, jj, lj, in;
static logical packed;
static integer ix, iy, ns, lx, ly;
@@ -3202,7 +3188,7 @@ L130:
f_rew(&al__1);*/
}
ccher2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
yy[1], &incy, &aa[1], &lda, (ftnlen)1);
yy[1], &incy, &aa[1], &lda);
} else if (packed) {
if (*trace) {
/*
@@ -3216,7 +3202,7 @@ L130:
f_rew(&al__1);*/
}
cchpr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
yy[1], &incy, &aa[1], (ftnlen)1);
yy[1], &incy, &aa[1]);
}
/* Check if error-exit was taken incorrectly. */
+35 -40
View File
@@ -23,17 +23,12 @@ typedef struct { real r, i; } complex;
typedef struct { doublereal r, i; } doublecomplex;
#ifdef _MSC_VER
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
#else
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
#endif
#define pCf(z) (*_pCf(z))
#define pCd(z) (*_pCd(z))
typedef int logical;
typedef short int shortlogical;
typedef char logical1;
@@ -284,10 +279,10 @@ int /* Main program */ main(void)
real r__1;
/* Local variables */
integer nalf, idim[9];
logical same;
integer nbet, ntra;
logical rewi;
static integer nalf, idim[9];
static logical same;
static integer nbet, ntra;
static logical rewi;
extern /* Subroutine */ int cchk1_(char *, real *, real *, integer *,
integer *, logical *, logical *, logical *, integer *, integer *,
integer *, complex *, integer *, complex *, integer *, complex *,
@@ -311,35 +306,35 @@ int /* Main program */ main(void)
integer *, complex *, integer *, complex *, integer *, complex *,
complex *, complex *, complex *, complex *, complex *, complex *,
complex *, complex *, real *, complex *, integer *);
complex c__[4225] /* was [65][65] */;
real g[65];
integer i__, j, n;
logical fatal;
complex w[130];
static complex c__[4225] /* was [65][65] */;
static real g[65];
static integer i__, j, n;
static logical fatal;
static complex w[130];
extern /* Subroutine */ int cmmch_(char *, char *, integer *, integer *,
integer *, complex *, complex *, integer *, complex *, integer *,
complex *, complex *, integer *, complex *, real *, complex *,
integer *, real *, real *, logical *, integer *, logical *);
extern real sdiff_(real *, real *);
logical trace;
integer nidim;
char snaps[32];
integer isnum;
logical ltest[9];
complex aa[4225], ab[8450] /* was [65][130] */, bb[4225], cc[4225], as[
static logical trace;
static integer nidim;
static char snaps[32];
static integer isnum;
static logical ltest[9];
static complex aa[4225], ab[8450] /* was [65][130] */, bb[4225], cc[4225], as[
4225], bs[4225], cs[4225], ct[65];
logical sfatal, corder;
char snamet[12], transa[1], transb[1];
real thresh;
logical rorder;
extern /* Subroutine */ int cc3chke_(char *);
integer layout;
logical ltestt, tsterr;
complex alf[7];
static logical sfatal, corder;
static char snamet[12], transa[1], transb[1];
static real thresh;
static logical rorder;
extern /* Subroutine */ void cc3chke_(char *);
static integer layout;
static logical ltestt, tsterr;
static complex alf[7];
extern logical lce_(complex *, complex *, integer *);
complex bet[7];
real eps, err;
char tmpchar;
static complex bet[7];
static real eps, err;
static char tmpchar;
/* Test program for the COMPLEX Level 3 Blas. */
@@ -856,7 +851,7 @@ L230:
*, char *, char *, integer *, integer *, integer *, complex *,
integer *, integer *, complex *, integer *);
integer ia, ib, ma, mb, na, nb, nc, ik, im, in;
extern /* Subroutine */ int ccgemm_(integer *, char *, char *, integer *,
extern /* Subroutine */ void ccgemm_(integer *, char *, char *, integer *,
integer *, integer *, complex *, complex *, integer *, complex *,
integer *, complex *, complex *, integer *);
integer ks, ms, ns;
@@ -1268,13 +1263,13 @@ L130:
*, char *, char *, integer *, integer *, complex *, integer *,
integer *, complex *, integer *);
integer ia, ib, na, nc, im, in;
extern /* Subroutine */ int cchemm_(integer *, char *, char *, integer *,
extern /* Subroutine */ void cchemm_(integer *, char *, char *, integer *,
integer *, complex *, complex *, integer *, complex *, integer *,
complex *, complex *, integer *);
integer ms, ns;
extern logical lceres_(char *, char *, integer *, integer *, complex *,
complex *, integer *);
extern /* Subroutine */ int ccsymm_(integer *, char *, char *, integer *,
extern /* Subroutine */ void ccsymm_(integer *, char *, char *, integer *,
integer *, complex *, complex *, integer *, complex *, integer *,
complex *, complex *, integer *);
real errmax;
@@ -1668,11 +1663,11 @@ L120:
integer ia, na, nc, im, in, ms, ns;
extern logical lceres_(char *, char *, integer *, integer *, complex *,
complex *, integer *);
extern /* Subroutine */ int cctrmm_(integer *, char *, char *, char *,
extern /* Subroutine */ void cctrmm_(integer *, char *, char *, char *,
char *, integer *, integer *, complex *, complex *, integer *,
complex *, integer *);
char tranas[1], transa[1];
extern /* Subroutine */ int cctrsm_(integer *, char *, char *, char *,
extern /* Subroutine */ void cctrsm_(integer *, char *, char *, char *,
char *, integer *, integer *, complex *, complex *, integer *,
complex *, integer *);
real errmax;
@@ -2143,7 +2138,7 @@ L160:
integer *, char *, integer *, char *, char *, integer *, integer *
, real *, integer *, real *, integer *);
integer ia, ib, jc, ma, na, nc, ik, in, jj, lj, ks;
extern /* Subroutine */ int ccherk_(integer *, char *, char *, integer *,
extern /* Subroutine */ void ccherk_(integer *, char *, char *, integer *,
integer *, real *, complex *, integer *, real *, complex *,
integer *);
integer ns;
@@ -2151,7 +2146,7 @@ L160:
extern logical lceres_(char *, char *, integer *, integer *, complex *,
complex *, integer *);
real errmax;
extern /* Subroutine */ int ccsyrk_(integer *, char *, char *, integer *,
extern /* Subroutine */ void ccsyrk_(integer *, char *, char *, integer *,
integer *, complex *, complex *, integer *, complex *, complex *,
integer *);
char transs[1], transt[1];
@@ -2643,12 +2638,12 @@ L130:
complex *, integer *);
real errmax;
char transs[1], transt[1];
extern /* Subroutine */ int ccher2k_(integer *, char *, char *, integer *,
extern /* Subroutine */ void ccher2k_(integer *, char *, char *, integer *,
integer *, complex *, complex *, integer *, complex *, integer *,
real *, complex *, integer *);
integer laa, lbb, lda, lcc, ldb, ldc;
extern logical lce_(complex *, complex *, integer *);
extern /* Subroutine */ int ccsyr2k_(integer *, char *, char *, integer *,
extern /* Subroutine */ void ccsyr2k_(integer *, char *, char *, integer *,
integer *, complex *, complex *, integer *, complex *, integer *,
complex *, complex *, integer *);
complex als;
+1 -1
View File
@@ -54,7 +54,7 @@ void F77_drot( const int *N, double *X, const int *incX, double *Y,
}
void F77_drotm(const int *N, double *X, const int *incX, double *Y,
const int *incY, const double *dparam)
const int *incY, double *dparam)
{
cblas_drotm(*N, X, *incX, Y, *incY, dparam);
return;
+13 -8
View File
@@ -332,7 +332,8 @@ static doublereal c_b34 = 1.;
/* Local variables */
static integer k;
extern /* Subroutine */ int drotgtest_(doublereal*,doublereal*,doublereal*,doublereal*), stest1_(doublereal*,doublereal*,doublereal*,doublereal*);
extern /* Subroutine */ void drotgtest_(doublereal*,doublereal*,doublereal*,doublereal*);
extern int stest1_(doublereal*,doublereal*,doublereal*,doublereal*);
static doublereal sa, sb, sc, ss;
/* .. Parameters .. */
@@ -404,7 +405,8 @@ L40:
static integer i__;
extern doublereal dnrm2test_(integer*, doublereal*, integer*);
static doublereal stemp[1], strue[8];
extern /* Subroutine */ int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*), dscaltest_(integer*,doublereal*,doublereal*,integer*);
extern /* Subroutine */ int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*);
extern void dscaltest_(integer*,doublereal*,doublereal*,integer*);
extern doublereal dasumtest_(integer*,doublereal*,integer*);
extern /* Subroutine */ int itest1_(integer*,integer*), stest1_(doublereal*,doublereal*,doublereal*,doublereal*);
static doublereal sx[8];
@@ -430,7 +432,7 @@ L40:
/* .. Set vector arguments .. */
i__1 = len;
for (i__ = 1; i__ <= i__1; ++i__) {
sx[i__ - 1] = dv[i__ + (np1 + combla_1.incx * 5 << 3) - 49];
sx[i__ - 1] = dv[i__ + ((np1 + combla_1.incx * 5) << 3) - 49];
/* L20: */
}
@@ -450,7 +452,7 @@ L40:
, sx, &combla_1.incx);
i__1 = len;
for (i__ = 1; i__ <= i__1; ++i__) {
strue[i__ - 1] = dtrue5[i__ + (np1 + combla_1.incx * 5 <<
strue[i__ - 1] = dtrue5[i__ + ((np1 + combla_1.incx * 5) <<
3) - 49];
/* L40: */
}
@@ -517,8 +519,10 @@ L40:
static integer lenx, leny;
extern doublereal ddottest_(integer*,doublereal*,integer*,doublereal*,integer*);
static integer i__, j, ksize;
extern /* Subroutine */ int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*), dcopytest_(integer*,doublereal*,integer*,doublereal*,integer*), dswaptest_(integer*,doublereal*,integer*,doublereal*,integer*),
daxpytest_(integer*,doublereal*,doublereal*,integer*,doublereal*,integer*), stest1_(doublereal*,doublereal*,doublereal*,doublereal*);
extern /* Subroutine */ int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*);
extern void dcopytest_(integer*,doublereal*,integer*,doublereal*,integer*), dswaptest_(integer*,doublereal*,integer*,doublereal*,integer*),
daxpytest_(integer*,doublereal*,doublereal*,integer*,doublereal*,integer*);
extern int stest1_(doublereal*,doublereal*,doublereal*,doublereal*);
static integer ki, kn, mx, my;
static doublereal sx[7], sy[7], stx[7], sty[7];
@@ -618,9 +622,10 @@ L40:
;
/* Local variables */
extern /* Subroutine */ int drottest_(integer*,doublereal*,integer*,doublereal*,integer*,doublereal*,doublereal*);
extern /* Subroutine */ void drottest_(integer*,doublereal*,integer*,doublereal*,integer*,doublereal*,doublereal*);
static integer i__, k, ksize;
extern /* Subroutine */int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*), drotmtest_(integer*,doublereal*,integer*,doublereal*,integer*,doublereal*);
extern /* Subroutine */int stest_(integer*,doublereal*,doublereal*,doublereal*,doublereal*);
extern void drotmtest_(integer*,doublereal*,integer*,doublereal*,integer*,doublereal*);
static integer ki, kn;
static doublereal dparam[5], sx[10], sy[10], stx[10], sty[10];
+32 -53
View File
@@ -21,19 +21,6 @@ typedef float real;
typedef double doublereal;
typedef struct { real r, i; } complex;
typedef struct { doublereal r, i; } doublecomplex;
#ifdef _MSC_VER
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
#else
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
#endif
#define pCf(z) (*_pCf(z))
#define pCd(z) (*_pCd(z))
typedef int logical;
typedef short int shortlogical;
typedef char logical1;
@@ -318,7 +305,7 @@ static logical c_false = FALSE_;
static char snamet[12];
static doublereal thresh;
static logical rorder;
extern /* Subroutine */ void cd2chke_(char*, ftnlen);
extern /* Subroutine */ void cd2chke_(char*);
static integer layout;
static logical ltestt, tsterr;
static doublereal alf[7];
@@ -706,7 +693,7 @@ L100:
ftnlen)12);
/* Test error exits. */
if (tsterr) {
cd2chke_(snames[isnum - 1], (ftnlen)12);
cd2chke_(snames[isnum - 1]);
}
/* Test computations. */
infoc_1.infot = 0;
@@ -885,8 +872,8 @@ L240:
static integer ia, ib, ic;
static logical banded;
static integer nc, nd, im, in, kl, ml, nk, nl, ku, ix, iy, ms, lx, ly, ns;
extern /* Subroutine */ void cdgbmv_(integer*, char*, integer*, integer*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen);
extern /* Subroutine */ void cdgemv_(integer*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen);
extern /* Subroutine */ void cdgbmv_(integer*, char*, integer*, integer*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
extern /* Subroutine */ void cdgemv_(integer*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
static char ctrans[14];
static doublereal errmax, transl;
@@ -1118,8 +1105,7 @@ L240:
}
cdgemv_(iorder, trans, &m, &n, &alpha,
&aa[1], &lda, &xx[1], &incx,
&beta, &yy[1], &incy, (ftnlen)
1);
&beta, &yy[1], &incy);
} else if (banded) {
if (*trace) {
/*
@@ -1135,7 +1121,7 @@ L240:
cdgbmv_(iorder, trans, &m, &n, &kl, &
ku, &alpha, &aa[1], &lda, &xx[
1], &incx, &beta, &yy[1], &
incy, (ftnlen)1);
incy);
}
/* Check if error-exit was taken incorrectly. */
@@ -1329,10 +1315,10 @@ L140:
static logical packed;
static integer nk, ks, ix, iy, ns, lx, ly;
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cdsbmv_(integer*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen);
extern /* Subroutine */ void cdspmv_(integer*, char*, integer*, doublereal*, doublereal*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen);
extern /* Subroutine */ void cdsbmv_(integer*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
extern /* Subroutine */ void cdspmv_(integer*, char*, integer*, doublereal*, doublereal*, doublereal*, integer*, doublereal*, doublereal*, integer*);
static doublereal errmax, transl;
extern /* Subroutine */ void cdsymv_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen);
extern /* Subroutine */ void cdsymv_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
static integer laa, lda;
extern logical lde_(doublereal*, doublereal*, integer*);
static doublereal als, bls, err;
@@ -1536,7 +1522,7 @@ L140:
}
cdsymv_(iorder, uplo, &n, &alpha, &aa[1],
&lda, &xx[1], &incx, &beta, &yy[1]
, &incy, (ftnlen)1);
, &incy);
} else if (banded) {
if (*trace) {
/*
@@ -1551,7 +1537,7 @@ L140:
}
cdsbmv_(iorder, uplo, &n, &k, &alpha, &aa[
1], &lda, &xx[1], &incx, &beta, &
yy[1], &incy, (ftnlen)1);
yy[1], &incy);
} else if (packed) {
if (*trace) {
/*
@@ -1566,7 +1552,7 @@ L140:
}
cdspmv_(iorder, uplo, &n, &alpha, &aa[1],
&xx[1], &incx, &beta, &yy[1], &
incy, (ftnlen)1);
incy);
}
/* Check if error-exit was taken incorrectly. */
@@ -1770,15 +1756,15 @@ L130:
static logical packed;
static integer nk, ks, ix, ns, lx;
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cdtbmv_(integer*, char*, char*, char*, integer*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cdtbsv_(integer*, char*, char*, char*, integer*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cdtbmv_(integer*, char*, char*, char*, integer*, integer*, doublereal*, integer*, doublereal*, integer*);
extern /* Subroutine */ void cdtbsv_(integer*, char*, char*, char*, integer*, integer*, doublereal*, integer*, doublereal*, integer*);
static char ctrans[14];
static doublereal errmax;
extern /* Subroutine */ void cdtpmv_(integer*, char*, char*, char*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cdtrmv_(integer*, char*, char*, char*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cdtpmv_(integer*, char*, char*, char*, integer*, doublereal*, doublereal*, integer*);
extern /* Subroutine */ void cdtrmv_(integer*, char*, char*, char*, integer*, doublereal*, integer*, doublereal*, integer*);
static doublereal transl;
extern /* Subroutine */ void cdtpsv_(integer*, char*, char*, char*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cdtrsv_(integer*, char*, char*, char*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cdtpsv_(integer*, char*, char*, char*, integer*, doublereal*, doublereal*, integer*);
extern /* Subroutine */ void cdtrsv_(integer*, char*, char*, char*, integer*, doublereal*, integer*, doublereal*, integer*);
static char transs[1];
static integer laa, icd, lda;
extern logical lde_(doublereal*, doublereal*, integer*);
@@ -1978,8 +1964,7 @@ L130:
f_rew(&al__1);*/
}
cdtrmv_(iorder, uplo, trans, diag, &n, &
aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
aa[1], &lda, &xx[1], &incx);
} else if (banded) {
if (*trace) {
/*
@@ -1993,8 +1978,7 @@ L130:
f_rew(&al__1);*/
}
cdtbmv_(iorder, uplo, trans, diag, &n, &k,
&aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
&aa[1], &lda, &xx[1], &incx);
} else if (packed) {
if (*trace) {
/*
@@ -2008,8 +1992,7 @@ L130:
f_rew(&al__1);*/
}
cdtpmv_(iorder, uplo, trans, diag, &n, &
aa[1], &xx[1], &incx, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
aa[1], &xx[1], &incx);
}
} else if (s_cmp(sname + 9, "sv", (ftnlen)2, (
ftnlen)2) == 0) {
@@ -2026,8 +2009,7 @@ L130:
f_rew(&al__1);*/
}
cdtrsv_(iorder, uplo, trans, diag, &n, &
aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
aa[1], &lda, &xx[1], &incx);
} else if (banded) {
if (*trace) {
/*
@@ -2041,8 +2023,7 @@ L130:
f_rew(&al__1);*/
}
cdtbsv_(iorder, uplo, trans, diag, &n, &k,
&aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
&aa[1], &lda, &xx[1], &incx);
} else if (packed) {
if (*trace) {
/*
@@ -2056,8 +2037,7 @@ L130:
f_rew(&al__1);*/
}
cdtpsv_(iorder, uplo, trans, diag, &n, &
aa[1], &xx[1], &incx, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
aa[1], &xx[1], &incx);
}
}
@@ -2587,11 +2567,11 @@ L150:
static logical isame[13];
extern /* Subroutine */ int dmvch_(char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, doublereal*, doublereal*, doublereal*, doublereal*, doublereal*, logical*, integer*, logical*, ftnlen);
static integer nargs;
extern /* Subroutine */ void cdspr_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, ftnlen);
extern /* Subroutine */ void cdspr_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*);
static logical reset;
static char cuplo[14];
static integer incxs;
extern /* Subroutine */ void cdsyr_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, ftnlen);
extern /* Subroutine */ void cdsyr_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*);
static logical upper;
static char uplos[1];
static integer ia, ja, ic, nc, jj, lj, in;
@@ -2751,7 +2731,7 @@ L150:
f_rew(&al__1);*/
}
cdsyr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]
, &lda, (ftnlen)1);
, &lda);
} else if (packed) {
if (*trace) {
/*
@@ -2764,8 +2744,7 @@ L150:
al__1.aunit = *ntra;
f_rew(&al__1);*/
}
cdspr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]
, (ftnlen)1);
cdspr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]);
}
/* Check if error-exit was taken incorrectly. */
@@ -2948,8 +2927,8 @@ L130:
static integer incxs, incys;
static logical upper;
static char uplos[1];
extern /* Subroutine */ void cdspr2_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, ftnlen);
extern /* Subroutine */ void cdsyr2_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen);
extern /* Subroutine */ void cdspr2_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*);
extern /* Subroutine */ void cdsyr2_(integer*, char*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, integer*);
static integer ia, ja, ic, nc, jj, lj, in;
static logical packed;
static integer ix, iy, ns, lx, ly;
@@ -3132,7 +3111,7 @@ L130:
f_rew(&al__1);*/
}
cdsyr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
yy[1], &incy, &aa[1], &lda, (ftnlen)1);
yy[1], &incy, &aa[1], &lda);
} else if (packed) {
if (*trace) {
/*
@@ -3146,7 +3125,7 @@ L130:
f_rew(&al__1);*/
}
cdspr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
yy[1], &incy, &aa[1], (ftnlen)1);
yy[1], &incy, &aa[1]);
}
/* Check if error-exit was taken incorrectly. */
+14 -32
View File
@@ -21,19 +21,6 @@ typedef float real;
typedef double doublereal;
typedef struct { real r, i; } complex;
typedef struct { doublereal r, i; } doublecomplex;
#ifdef _MSC_VER
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
#else
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
#endif
#define pCf(z) (*_pCf(z))
#define pCd(z) (*_pCd(z))
typedef int logical;
typedef short int shortlogical;
typedef char logical1;
@@ -309,7 +296,7 @@ static logical c_false = FALSE_;
static char snamet[12], transa[1], transb[1];
static doublereal thresh;
static logical rorder;
extern /* Subroutine */ void cd3chke_(char*, ftnlen);
extern /* Subroutine */ void cd3chke_(char*);
static integer layout;
static logical ltestt, tsterr;
static doublereal alf[7];
@@ -658,7 +645,7 @@ L80:
ftnlen)12);
/* Test error exits. */
if (tsterr) {
cd3chke_(snames[isnum - 1], (ftnlen)12);
cd3chke_(snames[isnum - 1]);
}
/* Test computations. */
infoc_1.infot = 0;
@@ -807,7 +794,7 @@ L230:
static logical reset;
extern /* Subroutine */ void dprcn1_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, integer*, doublereal*, integer*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
static integer ia, ib, ma, mb, na, nb, nc, ik, im, in;
extern /* Subroutine */ void cdgemm_(integer*, char*, char*, integer*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cdgemm_(integer*, char*, char*, integer*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
static integer ks, ms, ns;
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
static char tranas[1], tranbs[1], transa[1], transb[1];
@@ -1012,8 +999,7 @@ L230:
}
cdgemm_(iorder, transa, transb, &m, &n, &k, &
alpha, &aa[1], &lda, &bb[1], &ldb, &
beta, &cc[1], &ldc, (ftnlen)1, (
ftnlen)1);
beta, &cc[1], &ldc);
/* Check if error-exit was taken incorrectly. */
@@ -1204,7 +1190,7 @@ L130:
extern /* Subroutine */ void dprcn2_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, doublereal*, integer*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
static integer ia, ib, na, nc, im, in, ms, ns;
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cdsymm_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cdsymm_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
static doublereal errmax;
static integer laa, lbb, lda, lcc, ldb, ldc;
extern logical lde_(doublereal*, doublereal*, integer*);
@@ -1386,8 +1372,7 @@ L130:
f_rew(&al__1);*/
}
cdsymm_(iorder, side, uplo, &m, &n, &alpha, &aa[1]
, &lda, &bb[1], &ldb, &beta, &cc[1], &ldc,
(ftnlen)1, (ftnlen)1);
, &lda, &bb[1], &ldb, &beta, &cc[1], &ldc);
/* Check if error-exit was taken incorrectly. */
@@ -1580,9 +1565,9 @@ L120:
extern /* Subroutine */ void dprcn3_(integer*, integer*, char*, integer*, char*, char*, char*, char*, integer*, integer*, doublereal*, integer*, integer*, ftnlen, ftnlen, ftnlen, ftnlen, ftnlen);
static integer ia, na, nc, im, in, ms, ns;
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cdtrmm_(integer*, char*, char*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cdtrmm_(integer*, char*, char*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*);
static char tranas[1], transa[1];
extern /* Subroutine */ void cdtrsm_(integer*, char*, char*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cdtrsm_(integer*, char*, char*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*);
static doublereal errmax;
static integer laa, icd, lbb, lda, ldb;
extern logical lde_(doublereal*, doublereal*, integer*);
@@ -1762,8 +1747,7 @@ L120:
}
cdtrmm_(iorder, side, uplo, transa, diag,
&m, &n, &alpha, &aa[1], &lda, &bb[
1], &ldb, (ftnlen)1, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
1], &ldb);
} else if (s_cmp(sname + 9, "sm", (ftnlen)2, (
ftnlen)2) == 0) {
if (*trace) {
@@ -1780,8 +1764,7 @@ L120:
}
cdtrsm_(iorder, side, uplo, transa, diag,
&m, &n, &alpha, &aa[1], &lda, &bb[
1], &ldb, (ftnlen)1, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
1], &ldb);
}
/* Check if error-exit was taken incorrectly. */
@@ -2038,7 +2021,7 @@ L160:
static integer ia, ib, jc, ma, na, nc, ik, in, jj, lj, ks, ns;
extern logical lderes_(char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
static doublereal errmax;
extern /* Subroutine */ void cdsyrk_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cdsyrk_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, doublereal*, integer*);
static char transs[1];
static integer laa, lda, lcc, ldc;
extern logical lde_(doublereal*, doublereal*, integer*);
@@ -2199,8 +2182,7 @@ L160:
f_rew(&al__1);*/
}
cdsyrk_(iorder, uplo, trans, &n, &k, &alpha, &aa[
1], &lda, &beta, &cc[1], &ldc, (ftnlen)1,
(ftnlen)1);
1], &lda, &beta, &cc[1], &ldc);
/* Check if error-exit was taken incorrectly. */
@@ -2420,7 +2402,7 @@ L130:
static char transs[1];
static integer laa, lbb, lda, lcc, ldb, ldc;
extern logical lde_(doublereal*, doublereal*, integer*);
extern /* Subroutine */ void cdsyr2k_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cdsyr2k_(integer*, char*, char*, integer*, integer*, doublereal*, doublereal*, integer*, doublereal*, integer*, doublereal*, doublereal*, integer*);
static doublereal als;
static integer ict, icu;
static doublereal err;
@@ -2604,7 +2586,7 @@ L130:
}
cdsyr2k_(iorder, uplo, trans, &n, &k, &alpha, &aa[
1], &lda, &bb[1], &ldb, &beta, &cc[1], &
ldc, (ftnlen)1, (ftnlen)1);
ldc);
/* Check if error-exit was taken incorrectly. */
+10 -6
View File
@@ -342,7 +342,8 @@ static real c_b34 = (float)1.;
/* Local variables */
static integer k;
extern /* Subroutine */ int srotgtest_(real*,real*,real*,real*), stest1_(real*,real*,real*,real*);
extern /* Subroutine */ void srotgtest_(real*,real*,real*,real*);
extern int stest1_(real*,real*,real*,real*);
static real sa, sb, sc, ss;
/* .. Parameters .. */
@@ -435,7 +436,8 @@ L40:
static integer i__;
extern real snrm2test_(integer*,real*,integer*);
static real stemp[1], strue[8];
extern /* Subroutine */ int stest_(integer*, real*,real*,real*,real*), sscaltest_(integer*,real*,real*,integer*);
extern /* Subroutine */ int stest_(integer*, real*,real*,real*,real*);
extern void sscaltest_(integer*,real*,real*,integer*);
extern real sasumtest_(integer*,real*,integer*);
extern /* Subroutine */ int itest1_(integer*,integer*), stest1_(real*,real*,real*,real*);
static real sx[8];
@@ -462,7 +464,7 @@ L40:
/* .. Set vector arguments .. */
i__1 = len;
for (i__ = 1; i__ <= i__1; ++i__) {
sx[i__ - 1] = dv[i__ + (np1 + combla_1.incx * 5 << 3) - 49];
sx[i__ - 1] = dv[i__ + ((np1 + combla_1.incx * 5) << 3) - 49];
/* L20: */
}
@@ -482,7 +484,7 @@ L40:
, sx, &combla_1.incx);
i__1 = len;
for (i__ = 1; i__ <= i__1; ++i__) {
strue[i__ - 1] = dtrue5[i__ + (np1 + combla_1.incx * 5 <<
strue[i__ - 1] = dtrue5[i__ + ((np1 + combla_1.incx * 5) <<
3) - 49];
/* L40: */
}
@@ -592,7 +594,8 @@ L40:
static integer lenx, leny;
extern real sdottest_(integer*,real*,integer*,real*,integer*);
static integer i__, j, ksize;
extern /* Subroutine */ int stest_(integer*,real*,real*,real*,real*), scopytest_(integer*,real*,integer*,real*,integer*), sswaptest_(integer*,real*,integer*,real*,integer*),
extern /* Subroutine */ int stest_(integer*,real*,real*,real*,real*);
extern void scopytest_(integer*,real*,integer*,real*,integer*), sswaptest_(integer*,real*,integer*,real*,integer*),
saxpytest_(integer*,real*,real*,integer*,real*,integer*);
static integer ki;
extern /* Subroutine */ int stest1_(real*,real*,real*,real*);
@@ -710,7 +713,8 @@ L40:
/* Local variables */
extern /* Subroutine */ void srottest_(integer*,real*,integer*,real*,integer*,real*,real*);
static integer i__, k, ksize;
extern /* Subroutine */ int stest_(integer*,real*,real*,real*,real*), srotmtest_(integer*,real*,integer*,real*,integer*,real*);
extern /* Subroutine */ int stest_(integer*,real*,real*,real*,real*);
extern void srotmtest_(integer*,real*,integer*,real*,integer*,real*);
static integer ki, kn;
static real sx[19], sy[19], sparam[5], stx[19], sty[19];
+34 -55
View File
@@ -21,19 +21,6 @@ typedef float real;
typedef double doublereal;
typedef struct { real r, i; } complex;
typedef struct { doublereal r, i; } doublecomplex;
#ifdef _MSC_VER
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
#else
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
#endif
#define pCf(z) (*_pCf(z))
#define pCd(z) (*_pCd(z))
typedef int logical;
typedef short int shortlogical;
typedef char logical1;
@@ -319,7 +306,7 @@ extern /* Subroutine */ int schk6_(char* sname, real* eps, real* thresh, integer
static logical rorder;
static integer layout;
static logical ltestt;
extern /* Subroutine */ int cs2chke_(char*, ftnlen);
extern /* Subroutine */ void cs2chke_(char*);
static logical tsterr;
static real alf[7];
static integer inc[7], nkb;
@@ -702,7 +689,7 @@ L100:
ftnlen)12);
/* Test error exits. */
if (tsterr) {
cs2chke_(snames[isnum - 1], (ftnlen)12);
cs2chke_(snames[isnum - 1]);
}
/* Test computations. */
infoc_1.infot = 0;
@@ -880,8 +867,8 @@ L240:
static integer ia, ib, ic;
static logical banded;
static integer nc, nd, im, in, kl, ml, nk, nl, ku, ix, iy, ms, lx, ly, ns;
extern /* Subroutine */ void csgbmv_(integer*, char*, integer*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen);
extern /* Subroutine */ void csgemv_(integer*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen);
extern /* Subroutine */ void csgbmv_(integer*, char*, integer*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
extern /* Subroutine */ void csgemv_(integer*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
static char ctrans[14];
static real errmax;
extern logical lseres_(char* type__, char* uplo, integer* m, integer* n, real* aa, real* as, integer* lda, ftnlen ltype_len, ftnlen uplo_len);
@@ -1115,8 +1102,7 @@ L240:
}
csgemv_(iorder, trans, &m, &n, &alpha,
&aa[1], &lda, &xx[1], &incx,
&beta, &yy[1], &incy, (ftnlen)
1);
&beta, &yy[1], &incy);
} else if (banded) {
if (*trace) {
/*
@@ -1132,7 +1118,7 @@ L240:
csgbmv_(iorder, trans, &m, &n, &kl, &
ku, &alpha, &aa[1], &lda, &xx[
1], &incx, &beta, &yy[1], &
incy, (ftnlen)1);
incy);
}
/* Check if error-exit was taken incorrectly. */
@@ -1327,10 +1313,10 @@ L140:
static integer nk, ks, ix, iy, ns, lx, ly;
static real errmax;
extern logical lseres_(char* , char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cssbmv_(integer*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen);
extern /* Subroutine */ void cssbmv_(integer*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
static real transl;
extern /* Subroutine */ void csspmv_(integer*, char*, integer*, real*, real*, real*, integer*, real*, real*, integer*, ftnlen);
extern /* Subroutine */ void cssymv_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen);
extern /* Subroutine */ void csspmv_(integer*, char*, integer*, real*, real*, real*, integer*, real*, real*, integer*);
extern /* Subroutine */ void cssymv_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
static integer laa, lda;
static real als, bls;
extern logical lse_(real*, real*, integer*);
@@ -1531,7 +1517,7 @@ L140:
}
cssymv_(iorder, uplo, &n, &alpha, &aa[1],
&lda, &xx[1], &incx, &beta, &yy[1]
, &incy, (ftnlen)1);
, &incy);
} else if (banded) {
if (*trace) {
/*
@@ -1546,7 +1532,7 @@ L140:
}
cssbmv_(iorder, uplo, &n, &k, &alpha, &aa[
1], &lda, &xx[1], &incx, &beta, &
yy[1], &incy, (ftnlen)1);
yy[1], &incy);
} else if (packed) {
if (*trace) {
/*
@@ -1561,7 +1547,7 @@ L140:
}
csspmv_(iorder, uplo, &n, &alpha, &aa[1],
&xx[1], &incx, &beta, &yy[1], &
incy, (ftnlen)1);
incy);
}
/* Check if error-exit was taken incorrectly. */
@@ -1767,14 +1753,14 @@ L130:
static char ctrans[14];
static real errmax;
extern logical lseres_(char*, char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cstbmv_(integer*, char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cstbmv_(integer*, char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*);
static real transl;
extern /* Subroutine */ void cstbsv_(integer*, char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cstbsv_(integer*, char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*);
static char transs[1];
extern /* Subroutine */ void cstpmv_(integer*, char*, char*, char*, integer*, real*, real*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cstrmv_(integer*, char*, char*, char*, integer*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cstpsv_(integer*, char*, char*, char*, integer*, real*, real*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cstrsv_(integer*, char*, char*, char*, integer*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cstpmv_(integer*, char*, char*, char*, integer*, real*, real*, integer*);
extern /* Subroutine */ void cstrmv_(integer*, char*, char*, char*, integer*, real*, integer*, real*, integer*);
extern /* Subroutine */ void cstpsv_(integer*, char*, char*, char*, integer*, real*, real*, integer*);
extern /* Subroutine */ void cstrsv_(integer*, char*, char*, char*, integer*, real*, integer*, real*, integer*);
static integer laa, icd, lda, ict, icu;
extern logical lse_(real*, real*, integer*);
static real err;
@@ -1972,8 +1958,7 @@ L130:
f_rew(&al__1);*/
}
cstrmv_(iorder, uplo, trans, diag, &n, &
aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
aa[1], &lda, &xx[1], &incx);
} else if (banded) {
if (*trace) {
/*
@@ -1987,8 +1972,7 @@ L130:
f_rew(&al__1);*/
}
cstbmv_(iorder, uplo, trans, diag, &n, &k,
&aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
&aa[1], &lda, &xx[1], &incx);
} else if (packed) {
if (*trace) {
/*
@@ -2002,8 +1986,7 @@ L130:
f_rew(&al__1);*/
}
cstpmv_(iorder, uplo, trans, diag, &n, &
aa[1], &xx[1], &incx, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
aa[1], &xx[1], &incx);
}
} else if (s_cmp(sname + 9, "sv", (ftnlen)2, (
ftnlen)2) == 0) {
@@ -2020,8 +2003,7 @@ L130:
f_rew(&al__1);*/
}
cstrsv_(iorder, uplo, trans, diag, &n, &
aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
aa[1], &lda, &xx[1], &incx);
} else if (banded) {
if (*trace) {
/*
@@ -2035,8 +2017,7 @@ L130:
f_rew(&al__1);*/
}
cstbsv_(iorder, uplo, trans, diag, &n, &k,
&aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
&aa[1], &lda, &xx[1], &incx);
} else if (packed) {
if (*trace) {
/*
@@ -2050,8 +2031,7 @@ L130:
f_rew(&al__1);*/
}
cstpsv_(iorder, uplo, trans, diag, &n, &
aa[1], &xx[1], &incx, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
aa[1], &xx[1], &incx);
}
}
@@ -2585,10 +2565,10 @@ L150:
static logical reset;
static char cuplo[14];
static integer incxs;
extern /* Subroutine */ void csspr_(integer*, char*, integer*, real*, real*, integer*, real*, ftnlen);
extern /* Subroutine */ void csspr_(integer*, char*, integer*, real*, real*, integer*, real*);
static logical upper;
static char uplos[1];
extern /* Subroutine */ void cssyr_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, ftnlen);
extern /* Subroutine */ void cssyr_(integer*, char*, integer*, real*, real*, integer*, real*, integer*);
static integer ia, ja, ic, nc, jj, lj, in;
static logical packed;
static integer ix, ns, lx;
@@ -2747,7 +2727,7 @@ L150:
f_rew(&al__1);*/
}
cssyr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]
, &lda, (ftnlen)1);
, &lda);
} else if (packed) {
if (*trace) {
/*
@@ -2760,8 +2740,7 @@ L150:
al__1.aunit = *ntra;
f_rew(&al__1);*/
}
csspr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]
, (ftnlen)1);
csspr_(iorder, uplo, &n, &alpha, &xx[1], &incx, &aa[1]);
}
/* Check if error-exit was taken incorrectly. */
@@ -2945,13 +2924,13 @@ L130:
static logical upper;
static char uplos[1];
static integer ia, ja, ic;
extern /* Subroutine */ void csspr2_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*, ftnlen);
extern /* Subroutine */ void csspr2_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*);
static integer nc, jj, lj, in;
static logical packed;
extern /* Subroutine */ void cssyr2_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*, integer*, ftnlen);
extern /* Subroutine */ void cssyr2_(integer*, char*, integer*, real*, real*, integer*, real*, integer*, real*, integer*);
static integer ix, iy, ns, lx, ly;
static real errmax;
extern logical lseres_(char* type__, char* uplo, integer* m, integer* n, real* aa, real* as, integer* lda, ftnlen ltype_len, ftnlen uplo_len);
extern logical lseres_(char* type__, char* uplo, integer* m, integer* n, real* aa, real* as, integer* lda, ftnlen, ftnlen);
static real transl;
static integer laa, lda;
static real als;
@@ -3131,7 +3110,7 @@ L130:
f_rew(&al__1);*/
}
cssyr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
yy[1], &incy, &aa[1], &lda, (ftnlen)1);
yy[1], &incy, &aa[1], &lda);
} else if (packed) {
if (*trace) {
/*
@@ -3145,7 +3124,7 @@ L130:
f_rew(&al__1);*/
}
csspr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
yy[1], &incy, &aa[1], (ftnlen)1);
yy[1], &incy, &aa[1]);
}
/* Check if error-exit was taken incorrectly. */
@@ -3380,7 +3359,7 @@ L170:
i__2 = *m;
for (i__ = 1; i__ <= i__2; ++i__) {
if (gen || (upper && i__ <= j) || (lower && i__ >= j)) {
if (i__ <= j && (j - i__ <= *ku || i__ >= j && i__ - j <= *kl))
if (((i__ <= j && j - i__ <= *ku) || (i__ >= j && i__ - j <= *kl)))
{
a[i__ + j * a_dim1] = sbeg_(reset) + *transl;
} else {
+15 -33
View File
@@ -21,19 +21,6 @@ typedef float real;
typedef double doublereal;
typedef struct { real r, i; } complex;
typedef struct { doublereal r, i; } doublecomplex;
#ifdef _MSC_VER
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
#else
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
#endif
#define pCf(z) (*_pCf(z))
#define pCd(z) (*_pCd(z))
typedef int logical;
typedef short int shortlogical;
typedef char logical1;
@@ -309,7 +296,7 @@ static logical c_false = FALSE_;
static logical rorder;
static integer layout;
static logical ltestt, tsterr;
extern /* Subroutine */ void cs3chke_(char*, ftnlen);
extern /* Subroutine */ void cs3chke_(char*);
static real alf[7], bet[7];
extern logical lse_(real*, real*, integer*);
static real eps, err;
@@ -522,7 +509,7 @@ L30:
if (i__1 < 2) {
goto L60;
}
for (i__ = 1; i__ <= 9; ++i__) {
for (i__ = 1; i__ <= 6; ++i__) {
if (s_cmp(snamet, snames[i__ - 1] , (ftnlen)12, (ftnlen)12) ==
0) {
goto L50;
@@ -656,7 +643,7 @@ L80:
ftnlen)12);
/* Test error exits. */
if (tsterr) {
cs3chke_(snames[isnum - 1], (ftnlen)12);
cs3chke_(snames[isnum - 1]);
}
/* Test computations. */
infoc_1.infot = 0;
@@ -800,7 +787,7 @@ L230:
extern /* Subroutine */ int smake_(char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*, logical*, real*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ int smmch_(char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, real*, real*, real*, integer*, real*, real*, logical*, integer*, logical*, ftnlen, ftnlen);
static integer ia, ib, ma, mb, na, nb, nc, ik, im, in, ks, ms, ns;
extern /* Subroutine */ void csgemm_(integer*, char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void csgemm_(integer*, char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
static char tranas[1], tranbs[1], transa[1], transb[1];
static real errmax;
extern logical lseres_(char*, char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
@@ -1003,8 +990,7 @@ L230:
}
csgemm_(iorder, transa, transb, &m, &n, &k, &
alpha, &aa[1], &lda, &bb[1], &ldb, &
beta, &cc[1], &ldc, (ftnlen)1, (
ftnlen)1);
beta, &cc[1], &ldc);
/* Check if error-exit was taken incorrectly. */
@@ -1197,7 +1183,7 @@ L130:
static integer ia, ib, na, nc, im, in, ms, ns;
static real errmax;
extern logical lseres_(char*, char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cssymm_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cssymm_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
extern void sprcn2_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, real*, integer*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ int smake_(char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*, logical*, real*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ int smmch_(char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, real*, real*, real*, integer*, real*, real*, logical*, integer*, logical*, ftnlen, ftnlen);
@@ -1378,8 +1364,7 @@ L130:
// f_rew(&al__1);
}
cssymm_(iorder, side, uplo, &m, &n, &alpha, &aa[1]
, &lda, &bb[1], &ldb, &beta, &cc[1], &ldc,
(ftnlen)1, (ftnlen)1);
, &lda, &bb[1], &ldb, &beta, &cc[1], &ldc);
/* Check if error-exit was taken incorrectly. */
@@ -1575,8 +1560,8 @@ L120:
extern /* Subroutine */ int smake_(char*, char*, char*, integer*, integer*, real*, integer*, real*, integer*, logical*, real*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ int smmch_(char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, real*, real*, real*, integer*, real*, real*, logical*, integer*, logical*, ftnlen, ftnlen);
extern logical lseres_(char*, char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cstrmm_(integer*, char*, char*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cstrsm_(integer*, char*, char*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cstrmm_(integer*, char*, char*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*);
extern /* Subroutine */ void cstrsm_(integer*, char*, char*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*);
static integer laa, icd, lbb, lda, ldb, ics;
static real als;
static integer ict, icu;
@@ -1752,8 +1737,7 @@ L120:
}
cstrmm_(iorder, side, uplo, transa, diag,
&m, &n, &alpha, &aa[1], &lda, &bb[
1], &ldb, (ftnlen)1, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
1], &ldb);
} else if (s_cmp(sname + 9, "sm", (ftnlen)2, (
ftnlen)2) == 0) {
if (*trace) {
@@ -1768,8 +1752,7 @@ L120:
}
cstrsm_(iorder, side, uplo, transa, diag,
&m, &n, &alpha, &aa[1], &lda, &bb[
1], &ldb, (ftnlen)1, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
1], &ldb);
}
/* Check if error-exit was taken incorrectly. */
@@ -2028,7 +2011,7 @@ L160:
static real errmax;
extern logical lseres_(char*, char*, integer*, integer*, real*, real*, integer*, ftnlen, ftnlen);
static char transs[1];
extern /* Subroutine */ void cssyrk_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, real*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cssyrk_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, real*, integer*);
static integer laa, lda, lcc, ldc;
static real als;
static integer ict, icu;
@@ -2186,8 +2169,7 @@ L160:
// f_rew(&al__1);
}
cssyrk_(iorder, uplo, trans, &n, &k, &alpha, &aa[
1], &lda, &beta, &cc[1], &ldc, (ftnlen)1,
(ftnlen)1);
1], &lda, &beta, &cc[1], &ldc);
/* Check if error-exit was taken incorrectly. */
@@ -2409,7 +2391,7 @@ L130:
static integer laa, lbb, lda, lcc, ldb, ldc;
static real als;
static integer ict, icu;
extern /* Subroutine */ void cssyr2k_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cssyr2k_(integer*, char*, char*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*);
extern logical lse_(real*, real*, integer*);
extern /* Subroutine */ int smmch_(char*, char*, integer*, integer*, integer*, real*, real*, integer*, real*, integer*, real*, real*, integer*, real*, real*, real*, integer*, real*, real*, logical*, integer*, logical*, ftnlen, ftnlen);
static real err;
@@ -2591,7 +2573,7 @@ L130:
}
cssyr2k_(iorder, uplo, trans, &n, &k, &alpha, &aa[
1], &lda, &bb[1], &ldb, &beta, &cc[1], &
ldc, (ftnlen)1, (ftnlen)1);
ldc);
/* Check if error-exit was taken incorrectly. */
+11 -18
View File
@@ -22,18 +22,10 @@ typedef double doublereal;
typedef struct { real r, i; } complex;
typedef struct { doublereal r, i; } doublecomplex;
#ifdef _MSC_VER
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
#else
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
#endif
#define pCf(z) (*_pCf(z))
#define pCd(z) (*_pCd(z))
typedef int logical;
typedef short int shortlogical;
typedef char logical1;
@@ -380,11 +372,12 @@ static doublereal c_b43 = 1.;
static integer i__;
extern /* Subroutine */ int ctest_(integer*, doublecomplex*, doublecomplex*, doublecomplex*, doublereal*);
static doublecomplex mwpcs[5], mwpct[5];
extern /* Subroutine */ int zscaltest_(integer*, doublecomplex*, doublecomplex*, integer*), itest1_(integer*, integer*), stest1_(doublereal*, doublereal*, doublereal*, doublereal*);
extern /* Subroutine */ void zscaltest_(integer*, doublecomplex*, doublecomplex*, integer*);
extern int itest1_(integer*, integer*), stest1_(doublereal*, doublereal*, doublereal*, doublereal*);
static doublecomplex cx[8];
extern doublereal dznrm2test_(integer*, doublecomplex*, integer*);
static integer np1;
extern /* Subroutine */ int zdscaltest_(integer*, doublereal*, doublecomplex*, integer*);
extern /* Subroutine */ void zdscaltest_(integer*, doublereal*, doublecomplex*, integer*);
extern integer izamaxtest_(integer*, doublecomplex*, integer*);
extern doublereal dzasumtest_(integer*, doublecomplex*, integer*);
static integer len;
@@ -408,7 +401,7 @@ static doublereal c_b43 = 1.;
i__1 = len;
for (i__ = 1; i__ <= i__1; ++i__) {
i__2 = i__ - 1;
i__3 = i__ + (np1 + combla_1.incx * 5 << 3) - 49;
i__3 = i__ + ((np1 + combla_1.incx * 5) << 3) - 49;
cx[i__2].r = cv[i__3].r, cx[i__2].i = cv[i__3].i;
/* L20: */
}
@@ -423,13 +416,13 @@ static doublereal c_b43 = 1.;
} else if (combla_1.icase == 8) {
/* .. ZSCALTEST .. */
zscaltest_(&combla_1.n, &ca, cx, &combla_1.incx);
ctest_(&len, cx, &ctrue5[(np1 + combla_1.incx * 5 << 3) - 48],
&ctrue5[(np1 + combla_1.incx * 5 << 3) - 48], sfac);
ctest_(&len, cx, &ctrue5[((np1 + combla_1.incx * 5) << 3) - 48],
&ctrue5[((np1 + combla_1.incx * 5) << 3) - 48], sfac);
} else if (combla_1.icase == 9) {
/* .. ZDSCALTEST .. */
zdscaltest_(&combla_1.n, &sa, cx, &combla_1.incx);
ctest_(&len, cx, &ctrue6[(np1 + combla_1.incx * 5 << 3) - 48],
&ctrue6[(np1 + combla_1.incx * 5 << 3) - 48], sfac);
ctest_(&len, cx, &ctrue6[((np1 + combla_1.incx * 5) << 3) - 48],
&ctrue6[((np1 + combla_1.incx * 5) << 3) - 48], sfac);
} else if (combla_1.icase == 10) {
/* .. IZAMAXTEST .. */
i__1 = izamaxtest_(&combla_1.n, cx, &combla_1.incx);
@@ -591,11 +584,11 @@ static doublereal c_b43 = 1.;
extern /* Subroutine */ int ctest_(integer*, doublecomplex*, doublecomplex*, doublecomplex*, doublereal*);
static integer ksize;
static doublecomplex ztemp;
extern /* Subroutine */ int zdotctest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*), zcopytest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*);
extern /* Subroutine */ void zdotctest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*), zcopytest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*);
static integer ki;
extern /* Subroutine */ int zdotutest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*), zswaptest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*);
extern /* Subroutine */ void zdotutest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*), zswaptest_(integer*, doublecomplex*, integer*, doublecomplex*, integer*);
static integer kn;
extern /* Subroutine */ int zaxpytest_(integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*);
extern /* Subroutine */ void zaxpytest_(integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*);
static doublecomplex cx[7], cy[7];
static integer mx, my;
+32 -48
View File
@@ -22,17 +22,12 @@ typedef double doublereal;
typedef struct { real r, i; } complex;
typedef struct { doublereal r, i; } doublecomplex;
#ifdef _MSC_VER
static inline _Fcomplex Cf(complex *z) {_Fcomplex zz={z->r , z->i}; return zz;}
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
static inline _Fcomplex * _pCf(complex *z) {return (_Fcomplex*)z;}
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
#else
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex float * _pCf(complex *z) {return (_Complex float*)z;}
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
#endif
#define pCf(z) (*_pCf(z))
#define pCd(z) (*_pCd(z))
typedef int logical;
typedef short int shortlogical;
@@ -322,7 +317,7 @@ static logical c_false = FALSE_;
static logical rorder;
static integer layout;
static logical ltestt, tsterr;
extern /* Subroutine */ void cz2chke_(char*, ftnlen);
extern /* Subroutine */ void cz2chke_(char*);
static doublecomplex alf[7];
static integer inc[7], nkb;
static doublecomplex bet[7];
@@ -713,7 +708,7 @@ L100:
ftnlen)12);
/* Test error exits. */
if (tsterr) {
cz2chke_(snames[isnum - 1], (ftnlen)12);
cz2chke_(snames[isnum - 1]);
}
/* Test computations. */
infoc_1.infot = 0;
@@ -893,9 +888,9 @@ L240:
static integer ia, ib, ic;
static logical banded;
static integer nc, nd, im, in, kl, ml, nk, nl, ku, ix, iy, ms, lx, ly, ns;
extern /* Subroutine */ void czgbmv_(integer*, char*, integer*, integer*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen);
extern /* Subroutine */ void czgbmv_(integer*, char*, integer*, integer*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
static char ctrans[14];
extern /* Subroutine */ void czgemv_(integer*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen);
extern /* Subroutine */ void czgemv_(integer*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
static doublereal errmax;
static doublecomplex transl;
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
@@ -1144,8 +1139,7 @@ L240:
}
czgemv_(iorder, trans, &m, &n, &alpha,
&aa[1], &lda, &xx[1], &incx,
&beta, &yy[1], &incy, (ftnlen)
1);
&beta, &yy[1], &incy);
} else if (banded) {
if (*trace) {
/*
@@ -1160,8 +1154,7 @@ L240:
}
czgbmv_(iorder, trans, &m, &n, &kl, &
ku, &alpha, &aa[1], &lda, &xx[
1], &incx, &beta, &yy[1], &
incy, (ftnlen)1);
1], &incx, &beta, &yy[1], &incy);
}
/* Check if error-exit was taken incorrectly. */
@@ -1349,12 +1342,12 @@ L140:
static integer nc, ik, in;
static logical packed;
static integer nk, ks, ix, iy, ns, lx, ly;
extern /* Subroutine */ void czhbmv_(integer*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen);
extern /* Subroutine */ void czhemv_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen);
extern /* Subroutine */ void czhbmv_(integer*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
extern /* Subroutine */ void czhemv_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
static doublereal errmax;
static doublecomplex transl;
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void czhpmv_(integer*, char*, integer*, doublecomplex*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen);
extern /* Subroutine */ void czhpmv_(integer*, char*, integer*, doublecomplex*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
static integer laa, lda;
static doublecomplex als, bls;
static doublereal err;
@@ -1568,7 +1561,7 @@ L140:
}
czhemv_(iorder, uplo, &n, &alpha, &aa[1],
&lda, &xx[1], &incx, &beta, &yy[1]
, &incy, (ftnlen)1);
, &incy);
} else if (banded) {
if (*trace) {
/*
@@ -1583,7 +1576,7 @@ L140:
}
czhbmv_(iorder, uplo, &n, &k, &alpha, &aa[
1], &lda, &xx[1], &incx, &beta, &
yy[1], &incy, (ftnlen)1);
yy[1], &incy);
} else if (packed) {
if (*trace) {
/*
@@ -1597,8 +1590,7 @@ L140:
f_rew(&al__1);*/
}
czhpmv_(iorder, uplo, &n, &alpha, &aa[1],
&xx[1], &incx, &beta, &yy[1], &
incy, (ftnlen)1);
&xx[1], &incx, &beta, &yy[1], &incy);
}
/* Check if error-exit was taken incorrectly. */
@@ -1798,13 +1790,13 @@ L130:
static doublereal errmax;
static doublecomplex transl;
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cztbmv_(integer*, char*, char*, char*, integer*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cztbmv_(integer*, char*, char*, char*, integer*, integer*, doublecomplex*, integer*, doublecomplex*, integer*);
static char transs[1];
extern /* Subroutine */ void cztbsv_(integer*, char*, char*, char*, integer*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cztpmv_(integer*, char*, char*, char*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cztpsv_(integer*, char*, char*, char*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cztrmv_(integer*, char*, char*, char*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cztrsv_(integer*, char*, char*, char*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cztbsv_(integer*, char*, char*, char*, integer*, integer*, doublecomplex*, integer*, doublecomplex*, integer*);
extern /* Subroutine */ void cztpmv_(integer*, char*, char*, char*, integer*, doublecomplex*, doublecomplex*, integer*);
extern /* Subroutine */ void cztpsv_(integer*, char*, char*, char*, integer*, doublecomplex*, doublecomplex*, integer*);
extern /* Subroutine */ void cztrmv_(integer*, char*, char*, char*, integer*, doublecomplex*, integer*, doublecomplex*, integer*);
extern /* Subroutine */ void cztrsv_(integer*, char*, char*, char*, integer*, doublecomplex*, integer*, doublecomplex*, integer*);
static integer laa, icd, lda, ict, icu;
static doublereal err;
extern logical lze_(doublecomplex*, doublecomplex*, integer*);
@@ -2014,8 +2006,7 @@ L130:
f_rew(&al__1);*/
}
cztrmv_(iorder, uplo, trans, diag, &n, &
aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
aa[1], &lda, &xx[1], &incx);
} else if (banded) {
if (*trace) {
/*
@@ -2029,8 +2020,7 @@ L130:
f_rew(&al__1);*/
}
cztbmv_(iorder, uplo, trans, diag, &n, &k,
&aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
&aa[1], &lda, &xx[1], &incx);
} else if (packed) {
if (*trace) {
/*
@@ -2044,8 +2034,7 @@ L130:
f_rew(&al__1);*/
}
cztpmv_(iorder, uplo, trans, diag, &n, &
aa[1], &xx[1], &incx, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
aa[1], &xx[1], &incx);
}
} else if (s_cmp(sname + 9, "sv", (ftnlen)2, (
ftnlen)2) == 0) {
@@ -2062,8 +2051,7 @@ L130:
f_rew(&al__1);*/
}
cztrsv_(iorder, uplo, trans, diag, &n, &
aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
aa[1], &lda, &xx[1], &incx);
} else if (banded) {
if (*trace) {
/*
@@ -2077,8 +2065,7 @@ L130:
f_rew(&al__1);*/
}
cztbsv_(iorder, uplo, trans, diag, &n, &k,
&aa[1], &lda, &xx[1], &incx, (
ftnlen)1, (ftnlen)1, (ftnlen)1);
&aa[1], &lda, &xx[1], &incx);
} else if (packed) {
if (*trace) {
/*
@@ -2092,8 +2079,7 @@ L130:
f_rew(&al__1);*/
}
cztpsv_(iorder, uplo, trans, diag, &n, &
aa[1], &xx[1], &incx, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
aa[1], &xx[1], &incx);
}
}
@@ -2644,11 +2630,11 @@ L150:
static logical isame[13];
extern /* Subroutine */ int zmake_(char*, char*, char*, integer*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, integer*, integer*, logical*, doublecomplex*, ftnlen, ftnlen, ftnlen);
static integer nargs;
extern /* Subroutine */ void czher_(integer*, char*, integer*, doublereal*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen);
extern /* Subroutine */ void czher_(integer*, char*, integer*, doublereal*, doublecomplex*, integer*, doublecomplex*, integer*);
static logical reset;
static char cuplo[14];
static integer incxs;
extern /* Subroutine */ void czhpr_(integer*, char*, integer*, doublereal*, doublecomplex*, integer*, doublecomplex*, ftnlen);
extern /* Subroutine */ void czhpr_(integer*, char*, integer*, doublereal*, doublecomplex*, integer*, doublecomplex*);
extern /* Subroutine */ int zmvch_(char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublereal*, doublecomplex*, doublereal*, doublereal*, logical*, integer*, logical*, ftnlen);
static logical upper;
static char uplos[1];
@@ -2817,8 +2803,7 @@ L150:
al__1.aunit = *ntra;
f_rew(&al__1);*/
}
czher_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[
1], &lda, (ftnlen)1);
czher_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[1], &lda);
} else if (packed) {
if (*trace) {
/*
@@ -2831,8 +2816,7 @@ L150:
al__1.aunit = *ntra;
f_rew(&al__1);*/
}
czhpr_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[
1], (ftnlen)1);
czhpr_(iorder, uplo, &n, &ralpha, &xx[1], &incx, &aa[1]);
}
/* Check if error-exit was taken incorrectly. */
@@ -3011,8 +2995,8 @@ L130:
extern /* Subroutine */ int zmvch_(char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublereal*, doublecomplex*, doublereal*, doublereal*, logical*, integer*, logical*, ftnlen);
static logical upper;
static char uplos[1];
extern /* Subroutine */ void czher2_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen);
extern /* Subroutine */ void czhpr2_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, ftnlen);
extern /* Subroutine */ void czher2_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, integer*);
extern /* Subroutine */ void czhpr2_(integer*, char*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*);
static integer ia, ja, ic, nc, jj, lj, in;
static logical packed;
static integer ix, iy, ns, lx, ly;
@@ -3208,7 +3192,7 @@ L130:
f_rew(&al__1);*/
}
czher2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
yy[1], &incy, &aa[1], &lda, (ftnlen)1);
yy[1], &incy, &aa[1], &lda);
} else if (packed) {
if (*trace) {
/*
@@ -3222,7 +3206,7 @@ L130:
f_rew(&al__1);*/
}
czhpr2_(iorder, uplo, &n, &alpha, &xx[1], &incx, &
yy[1], &incy, &aa[1], (ftnlen)1);
yy[1], &incy, &aa[1]);
}
/* Check if error-exit was taken incorrectly. */
+20 -27
View File
@@ -25,11 +25,9 @@ typedef struct { doublereal r, i; } doublecomplex;
static inline _Dcomplex Cd(doublecomplex *z) {_Dcomplex zz={z->r , z->i};return zz;}
static inline _Dcomplex * _pCd(doublecomplex *z) {return (_Dcomplex*)z;}
#else
static inline _Complex float Cf(complex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double Cd(doublecomplex *z) {return z->r + z->i*_Complex_I;}
static inline _Complex double * _pCd(doublecomplex *z) {return (_Complex double*)z;}
#endif
#define pCf(z) (*_pCf(z))
#define pCd(z) (*_pCd(z))
typedef int logical;
typedef short int shortlogical;
@@ -314,7 +312,7 @@ static logical c_false = FALSE_;
static logical rorder;
static integer layout;
static logical ltestt, tsterr;
extern /* Subroutine */ int cz3chke_(char*, ftnlen);
extern /* Subroutine */ void cz3chke_(char*);
static doublecomplex alf[7], bet[7];
static doublereal eps, err;
extern logical lze_(doublecomplex*, doublecomplex*, integer*);
@@ -679,7 +677,7 @@ L80:
ftnlen)12);
/* Test error exits. */
if (tsterr) {
cz3chke_(snames[isnum - 1], (ftnlen)12);
cz3chke_(snames[isnum - 1]);
}
/* Test computations. */
infoc_1.infot = 0;
@@ -831,7 +829,7 @@ L230:
static integer ia, ib;
extern /* Subroutine */ int zprcn1_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, integer*, doublecomplex*, integer*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
static integer ma, mb, na, nb, nc, ik, im, in, ks, ms, ns;
extern /* Subroutine */ void czgemm_(integer*, char*, char*, integer*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void czgemm_(integer*, char*, char*, integer*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
static char tranas[1], tranbs[1], transa[1], transb[1];
static doublereal errmax;
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
@@ -1047,8 +1045,7 @@ L230:
}
czgemm_(iorder, transa, transb, &m, &n, &k, &
alpha, &aa[1], &lda, &bb[1], &ldb, &
beta, &cc[1], &ldc, (ftnlen)1, (
ftnlen)1);
beta, &cc[1], &ldc);
/* Check if error-exit was taken incorrectly. */
@@ -1242,10 +1239,10 @@ return 0;
static integer ia, ib;
extern /* Subroutine */ int zprcn2_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, doublecomplex*, integer*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen);
static integer na, nc, im, in, ms, ns;
extern /* Subroutine */ void czhemm_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void czhemm_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
static doublereal errmax;
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void czsymm_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void czsymm_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
static integer laa, lbb, lda, lcc, ldb, ldc, ics;
static doublecomplex als, bls;
static integer icu;
@@ -1438,11 +1435,11 @@ return 0;
if (isconj) {
czhemm_(iorder, side, uplo, &m, &n, &alpha, &
aa[1], &lda, &bb[1], &ldb, &beta, &cc[
1], &ldc, (ftnlen)1, (ftnlen)1);
1], &ldc);
} else {
czsymm_(iorder, side, uplo, &m, &n, &alpha, &
aa[1], &lda, &bb[1], &ldb, &beta, &cc[
1], &ldc, (ftnlen)1, (ftnlen)1);
1], &ldc);
}
/* Check if error-exit was taken incorrectly. */
@@ -1641,8 +1638,8 @@ return 0;
static char tranas[1], transa[1];
static doublereal errmax;
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void cztrmm_(integer*, char*, char*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cztrsm_(integer*, char*, char*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, ftnlen, ftnlen, ftnlen, ftnlen);
extern /* Subroutine */ void cztrmm_(integer*, char*, char*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*);
extern /* Subroutine */ void cztrsm_(integer*, char*, char*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*);
static integer laa, icd, lbb, lda, ldb, ics;
static doublecomplex als;
static integer ict, icu;
@@ -1828,8 +1825,7 @@ return 0;
}
cztrmm_(iorder, side, uplo, transa, diag,
&m, &n, &alpha, &aa[1], &lda, &bb[
1], &ldb, (ftnlen)1, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
1], &ldb);
} else if (s_cmp(sname + 9, "sm", (ftnlen)2, (
ftnlen)2) == 0) {
if (*trace) {
@@ -1846,8 +1842,7 @@ return 0;
}
cztrsm_(iorder, side, uplo, transa, diag,
&m, &n, &alpha, &aa[1], &lda, &bb[
1], &ldb, (ftnlen)1, (ftnlen)1, (
ftnlen)1, (ftnlen)1);
1], &ldb);
}
/* Check if error-exit was taken incorrectly. */
@@ -2119,11 +2114,11 @@ return 0;
extern /* Subroutine */ int zprcn6_(integer*, integer*, char*, integer*, char*, char*, integer*, integer*, doublereal*, integer*, doublereal*, integer*, ftnlen, ftnlen, ftnlen);
static integer ik, in, jj, lj, ks, ns;
static doublereal ralpha;
extern /* Subroutine */ int czherk_(integer*, char*, char*, integer*, integer*, doublereal*, doublecomplex*, integer*, doublereal*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void czherk_(integer*, char*, char*, integer*, integer*, doublereal*, doublecomplex*, integer*, doublereal*, doublecomplex*, integer*);
static doublereal errmax;
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
static char transs[1], transt[1];
extern /* Subroutine */ int czsyrk_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void czsyrk_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
static integer laa, lda, lcc, ldc;
static doublecomplex als;
static integer ict, icu;
@@ -2319,8 +2314,7 @@ return 0;
f_rew(&al__1);*/
}
czherk_(iorder, uplo, trans, &n, &k, &ralpha,
&aa[1], &lda, &rbeta, &cc[1], &ldc, (
ftnlen)1, (ftnlen)1);
&aa[1], &lda, &rbeta, &cc[1], &ldc);
} else {
if (*trace) {
zprcn4_(ntra, &nc, sname, iorder, uplo,
@@ -2334,8 +2328,7 @@ return 0;
f_rew(&al__1);*/
}
czsyrk_(iorder, uplo, trans, &n, &k, &alpha, &
aa[1], &lda, &beta, &cc[1], &ldc, (
ftnlen)1, (ftnlen)1);
aa[1], &lda, &beta, &cc[1], &ldc);
}
/* Check if error-exit was taken incorrectly. */
@@ -2615,11 +2608,11 @@ return 0;
static doublereal errmax;
extern logical lzeres_(char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
static char transs[1], transt[1];
extern /* Subroutine */ int czher2k_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublereal*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void czher2k_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublereal*, doublecomplex*, integer*);
static integer laa, lbb, lda, lcc, ldb, ldc;
static doublecomplex als;
static integer ict, icu;
extern /* Subroutine */ int czsyr2k_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*, ftnlen, ftnlen);
extern /* Subroutine */ void czsyr2k_(integer*, char*, char*, integer*, integer*, doublecomplex*, doublecomplex*, integer*, doublecomplex*, integer*, doublecomplex*, doublecomplex*, integer*);
static doublereal err;
extern logical lze_(doublecomplex*, doublecomplex*, integer*);
@@ -2830,7 +2823,7 @@ return 0;
}
czher2k_(iorder, uplo, trans, &n, &k, &alpha,
&aa[1], &lda, &bb[1], &ldb, &rbeta, &
cc[1], &ldc, (ftnlen)1, (ftnlen)1);
cc[1], &ldc);
} else {
if (*trace) {
zprcn5_(ntra, &nc, sname, iorder, uplo,
@@ -2845,7 +2838,7 @@ return 0;
}
czsyr2k_(iorder, uplo, trans, &n, &k, &alpha,
&aa[1], &lda, &bb[1], &ldb, &beta, &
cc[1], &ldc, (ftnlen)1, (ftnlen)1);
cc[1], &ldc);
}
/* Check if error-exit was taken incorrectly. */
+13 -1
View File
@@ -13,7 +13,9 @@ This page documents those non-standard APIs.
| ?omatcopy | s,d,c,z | out-of-place transposition/copying |
| ?geadd | s,d,c,z | ATLAS-like matrix add `B = &alpha;*A+&beta;*B` |
| ?gemmt | s,d,c,z | `gemm` but only a triangular part updated |
| cblas_?gemm_batch | s,d,c,z,b | `gemm` with several groups of input data
|
| cblas_?gemm_batch_strided | s,d,c,z,b | `gemm` with groups of data stored at fixed offsets in the input arrays
## bfloat16 functionality
@@ -26,6 +28,15 @@ BLAS-like and conversion functions for `bfloat16` (available when OpenBLAS was c
* `float cblas_sbdot` computes the dot product of two bfloat16 arrays
* `void cblas_sbgemv` performs the matrix-vector operations of GEMV with the input matrix and X vector as bfloat16
* `void cblas_sbgemm` performs the matrix-matrix operations of GEMM with both input arrays containing bfloat16
* `void cblas_bgemv` performs the matrix-vector operations of GEMV with the input matrix, X vector and result as bfloat16
* `void cblas_bgemm` performs the matrix-matrix operations of GEMM with both input arrays containing bfloat16 and the output being bfloat16 as well
## half-precision float or fp16 functionality
BLAS-like and conversion functions for `hfloat16` (available when OpenBLAS was compiled with `BUILD_HFLOAT16=1`):
* `void cblas_shgemm` performs the matrix-matrix operations of GEMM with both input arrays containing hfloat16
## Utility functions
@@ -36,4 +47,5 @@ BLAS-like and conversion functions for `bfloat16` (available when OpenBLAS was c
* `char * openblas_get_config()` returns the options OpenBLAS was built with, something like `NO_LAPACKE DYNAMIC_ARCH NO_AFFINITY Haswell`
* `int openblas_set_affinity(int thread_index, size_t cpusetsize, cpu_set_t *cpuset)` sets the CPU affinity mask of the given thread
to the provided cpuset. Only available on Linux, with semantics identical to `pthread_setaffinity_np`.
* `openblas_set_thread_callback_function` overrides the default multithreading backend with the provided argument
+8 -2
View File
@@ -47,7 +47,8 @@ You can find the full list of modifications in Changelog.txt.
The detailed explanation is probably in the original publication authored by Kazushige Goto - Goto, Kazushige; van de Geijn, Robert A; Anatomy of high-performance matrix multiplication. ACM Transactions on Mathematical Software (TOMS). Volume 34 Issue 3, May 2008
While this article is paywalled and too old for preprints to be available on arxiv.org, more recent
publications like https://arxiv.org/pdf/1609.00076 contain at least a brief description of the algorithm.
In practice, the values are derived by experimentation to yield the block sizes that give the highest performance. A general rule of thumb for selecting a starting point seems to be that PxQ is about half the size of L2 cache.
In practice, the values are derived by experimentation to yield the block sizes that give the highest performance. A general rule of thumb for selecting a starting point seems to be that PxQ is about half the size of L2 cache. R needs to be greater than the bigger of P and Q by
at least 64, or bad things will happen with the work splitting in (at least) POTRF.
### <a name="reportbug"></a>How can I report a bug?
@@ -344,7 +345,12 @@ Multithreading support in OpenBLAS requires the use of internal buffers for shar
If you get a message "error while loading shared libraries: libopenblas.so.0: ELF load command address/offset not properly aligned" when starting a program that is (dynamically) linked to OpenBLAS, this is very likely due to a bug in the GNU linker (ld) that is part of the
GNU binutils package. This error was specifically observed on older versions of Ubuntu Linux updated with the (at the time) most recent binutils version 2.38, but an internet search turned up sporadic reports involving various other libraries dating back several years. A bugfix was created by the binutils developers and should be available in later versions of binutils.(See issue 3708 for details)
#### <a name="OpenMP"></a>Using OpenBLAS with OpenMP
### <a name="CallingConvention"></a>The tests work fine, but calling any complex function from my code produces wrong or no results
This is almost certainly a problem with the calling convention used, in particular with the way the computed result is transported back to the caller. By default, OpenBLAS follows the F2C convention of returning the result on the stack rather than as the first argument to the function. So if your code has a prototype like "void cdotu ( complex *res, int n,...)" change it to "complex cdotu (int n,...)". Better yet,
use the CBLAS interface rather than the Fortran one.
### <a name="OpenMP"></a>Using OpenBLAS with OpenMP
OpenMP provides its own locking mechanisms, so when your code makes BLAS/LAPACK calls from inside OpenMP parallel regions it is imperative
that you use an OpenBLAS that is built with USE_OPENMP=1, as otherwise deadlocks might occur. Furthermore, OpenBLAS will automatically restrict itself to using only a single thread when called from an OpenMP parallel region. When it is certain that calls will only occur
+19 -10
View File
@@ -217,8 +217,11 @@ in this section, since the process for each is quite different.
For Visual Studio, you can use CMake to generate Visual Studio solution files;
note that you will need at least CMake 3.11 for linking to work correctly).
Note that you need a Fortran compiler if you plan to build and use the LAPACK
functions included with OpenBLAS. The sections below describe using either
Note that you need a Fortran compiler if you plan to build and use the latest version
of the LAPACK functions included with OpenBLAS. (If you do not have a Fortran compiler
installed, you can build an older version of the LAPACK sources that has been converted
to C - but its performance will likely be slower and accuracy may be poorer too.)
The sections below describe using either
`flang` as an add-on to clang/LLVM or `gfortran` as part of MinGW for this
purpose. If you want to use the Intel Fortran compiler (`ifort` or `ifx`) for
this, be sure to also use the Intel C compiler (`icc` or `icx`) for building
@@ -226,21 +229,22 @@ the C parts, as the ABI imposed by `ifort` is incompatible with MSVC
A fully-optimized OpenBLAS that can be statically or dynamically linked to your
application can currently be built for the 64-bit architecture with the LLVM
compiler infrastructure. We're going to use [Miniconda3](https://docs.anaconda.com/miniconda/)
compiler infrastructure. We're going to use [Miniforge3] the pre-configured
and more versatile alternative to [Miniconda](https://docs.anaconda.com/miniconda/)
to grab all of the tools we need, since some of them are in an experimental
status. Before you begin, you'll need to have Microsoft Visual Studio 2015 or
newer installed.
1. Install Miniconda3 for 64-bit Windows using `winget install --id Anaconda.Miniconda3`,
or easily download from [conda.io](https://docs.conda.io/en/latest/miniconda.html).
2. Open the "Anaconda Command Prompt" now available in the Start Menu, or at `%USERPROFILE%\miniconda3\shell\condabin\conda-hook.ps1`.
1. Install Miniforge for 64-bit Windows with the latest version of the installer Miniforge3-Windows-x86_64.exe
available on [github.com](https://github.com/conda-forge/miniforge/releases/)
2. Open the "Miniforge Command Prompt" now available in the Start Menu, or at `%USERPROFILE%\miniforge3\shell\condabin\conda-hook.ps1`.
3. In that command prompt window, use `cd` to change to the directory where you want to build OpenBLAS.
4. Now install all of the tools we need:
```
conda update -n base conda
conda config --add channels conda-forge
conda install -y cmake flang clangdev perl libflang ninja
conda install -y cmake flang_win-64 clangdev perl libflang ninja
```
(if you want to build with OpenMP support, add `llvm-openmp` and `llvm-openmp-fortran`)
5. Still in the Anaconda Command Prompt window, activate the 64-bit MSVC environment with `vcvarsall x64`.
On Windows 11 with Visual Studio 2022, this would be done by invoking:
@@ -439,6 +443,10 @@ To then use the built OpenBLAS shared library in Visual Studio:
### Windows on Arm
If you want to use a regular x64 Windows build of OpenBLAS with x64 software in the Prism emulator, be sure to use the latest version of Prism, and to check the box
to "Disable floating point optimization" in the Emulation settings. (Right-click on the executable to open "Properties", then on the "Compatibility" tab click on
"Change emulation settings").
A fully functional native OpenBLAS for WoA that can be built as both a static and dynamic library using LLVM toolchain and Visual Studio 2022. Before starting to build, make sure that you have installed Visual Studio 2022 on your ARM device, including the "Desktop Development with C++" component (that contains the cmake tool).
(Note that you can use the free "Visual Studio 2022 Community Edition" for this task. In principle it would be possible to build with VisualStudio alone, but using
the LLVM toolchain enables native compilation of the Fortran sources of LAPACK and of all the optimized assembly files, which VisualStudio cannot handle on its own)
@@ -708,9 +716,10 @@ fully working OpenBLAS for this platform.
Go to the directory where you unpacked OpenBLAS,and enter the following commands:
```bash
CC=/Applications/Xcode_12.4.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang
CC="/Applications/Xcode.app/Contents/Developer/Toolchains/XcodeDefault.xctoolchain/usr/bin/clang"
CFLAGS= -O2 -Wno-macro-redefined -isysroot /Applications/Xcode_12.4.app/Contents/Developer/Platforms/iPhoneOS.platform/Developer/SDKs/iPhoneOS14.4.sdk -arch arm64 -miphoneos-version-min=10.0
SDKROOT="$(xcrun --sdk iphoneos --show-sdk-path)"
CFLAGS="-O2 -Wno-macro-redefined -isysroot $SDKROOT -arch arm64 -miphoneos-version-min=10.0"
make TARGET=ARMV8 DYNAMIC_ARCH=1 NUM_THREADS=32 HOSTCC=clang NOFORTRAN=1
```
+1 -1
View File
@@ -30,7 +30,7 @@ OpenBLAS checks the following environment variables on startup:
cache where it is not reported correctly (in virtual environments)
Deprecated variables still recognized for compatibilty:
Deprecated variables still recognized for compatibility:
* `GOTO_NUM_THREADS`: equivalent to `OPENBLAS_NUM_THREADS`
* `GOTOBLAS_MAIN_FREE`: equivalent to `OPENBLAS_MAIN_FREE`
+12 -4
View File
@@ -59,13 +59,21 @@
#define GEMM_Q 128
#endif
#ifdef GEMM_DIVIDE_RATE
#ifdef DYNAMIC_ARCH
#define DIVIDE_LIMIT gotoblas->divide_limit
#define DIVIDE_RATE gotoblas->divide_rate
#else
#define DIVIDE_LIMIT GEMM_DIVIDE_LIMIT
#define DIVIDE_RATE GEMM_DIVIDE_RATE
#endif
#ifdef GEMM_DIVIDE_LIMIT
#define DIVIDE_LIMIT GEMM_DIVIDE_LIMIT
#endif
//#ifdef GEMM_DIVIDE_RATE
//#define DIVIDE_RATE GEMM_DIVIDE_RATE
//#endif
//#ifdef GEMM_DIVIDE_LIMIT
//#define DIVIDE_LIMIT GEMM_DIVIDE_LIMIT
//#endif
#ifdef THREADED_LEVEL3
#include "level3_thread.c"
+3 -2
View File
@@ -41,6 +41,7 @@
#define CACHE_LINE_SIZE 8
#endif
#define DIVIDE_RATE_MAX 2
#ifndef DIVIDE_RATE
#define DIVIDE_RATE 2
#endif
@@ -93,7 +94,7 @@ typedef struct {
#else
volatile
#endif
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE];
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE_MAX];
} job_t;
@@ -294,7 +295,7 @@ static int inner_thread(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n,
FLOAT *a, *b, *c;
job_t *job = (job_t *)args -> common;
BLASLONG xxx, bufferside;
FLOAT *buffer[DIVIDE_RATE];
FLOAT *buffer[DIVIDE_RATE_MAX];
BLASLONG ls, min_l, jjs, min_jj;
BLASLONG is, min_i, div_n;
+75 -2
View File
@@ -41,6 +41,8 @@
#define CACHE_LINE_SIZE 8
#endif
#define DIVIDE_RATE_MAX 2
#ifndef DIVIDE_RATE
#define DIVIDE_RATE 2
#endif
@@ -69,7 +71,7 @@ _Atomic
#else
volatile
#endif
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE];
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE_MAX];
} job_t;
@@ -133,7 +135,7 @@ _Atomic
static int inner_thread(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLOAT *sb, BLASLONG mypos){
FLOAT *buffer[DIVIDE_RATE];
FLOAT *buffer[DIVIDE_RATE_MAX];
BLASLONG k, lda, ldc;
BLASLONG m_from, m_to, n_from, n_to;
@@ -504,6 +506,33 @@ static int inner_thread(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n,
int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLOAT *sb, BLASLONG mypos){
#ifdef USE_OPENMP
static omp_lock_t level3_lock, critical_section_lock;
static volatile BLASULONG init_lock = 0, omp_lock_initialized = 0,
parallel_section_left = MAX_PARALLEL_NUMBER;
// Lock initialization; Todo : Maybe this part can be moved to blas_init() in blas_server_omp.c
while(omp_lock_initialized == 0)
{
blas_lock(&init_lock);
{
if(omp_lock_initialized == 0)
{
omp_init_lock(&level3_lock);
omp_init_lock(&critical_section_lock);
omp_lock_initialized = 1;
WMB;
}
blas_unlock(&init_lock);
}
}
#elif defined(OS_WINDOWS)
CRITICAL_SECTION level3_lock;
InitializeCriticalSection((PCRITICAL_SECTION)&level3_lock);
#else
static pthread_mutex_t level3_lock = PTHREAD_MUTEX_INITIALIZER;
#endif
blas_arg_t newarg;
#ifndef USE_ALLOC_HEAP
@@ -560,6 +589,30 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
#endif
#endif
#ifdef USE_OPENMP
omp_set_lock(&level3_lock);
omp_set_lock(&critical_section_lock);
parallel_section_left--;
/*
How OpenMP locks works with NUM_PARALLEL
1) parallel_section_left = Number of available concurrent executions of OpenBLAS - Number of currently executing OpenBLAS executions
2) level3_lock is acting like a master lock or barrier which stops OpenBLAS calls when all the parallel_section are currently busy executing other OpenBLAS calls
3) critical_section_lock is used for updating variables shared between threads executing OpenBLAS calls concurrently and for unlocking of master lock whenever required
4) Unlock master lock only when we have not already exhausted all the parallel_sections and allow another thread with a OpenBLAS call to enter
*/
if(parallel_section_left != 0)
omp_unset_lock(&level3_lock);
omp_unset_lock(&critical_section_lock);
#elif defined(OS_WINDOWS)
EnterCriticalSection((PCRITICAL_SECTION)&level3_lock);
#else
pthread_mutex_lock(&level3_lock);
#endif
newarg.m = args -> m;
newarg.n = args -> n;
newarg.k = args -> k;
@@ -706,5 +759,25 @@ int CNAME(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, FLOAT *sa, FLO
free(job);
#endif
#ifdef USE_OPENMP
omp_set_lock(&critical_section_lock);
parallel_section_left++;
/*
Unlock master lock only when all the parallel_sections are already exhausted and one of the thread has completed its OpenBLAS call
otherwise just increment the parallel_section_left
The master lock is only locked when we have exhausted all the parallel_sections, So only unlock it then and otherwise just increment the count
*/
if(parallel_section_left == 1)
omp_unset_lock(&level3_lock);
omp_unset_lock(&critical_section_lock);
#elif defined(OS_WINDOWS)
LeaveCriticalSection((PCRITICAL_SECTION)&level3_lock);
#else
pthread_mutex_unlock(&level3_lock);
#endif
return 0;
}
+11 -6
View File
@@ -41,12 +41,17 @@
#define CACHE_LINE_SIZE 8
#endif
#define DIVIDE_RATE_MAX 2
#ifndef DIVIDE_RATE
#define DIVIDE_RATE 2
#endif
#ifndef GEMM_PREFERED_SIZE
#define GEMM_PREFERED_SIZE 1
#ifdef DYNAMIC_ARCH
#define GEMM_PREFERRED_SIZE gotoblas->preferred_size
#endif
#ifndef GEMM_PREFERRED_SIZE
#define GEMM_PREFERRED_SIZE 1
#endif
//The array of job_t may overflow the stack.
@@ -93,7 +98,7 @@
typedef struct {
volatile
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE];
BLASLONG working[MAX_CPU_NUMBER][CACHE_LINE_SIZE * DIVIDE_RATE_MAX];
} job_t;
@@ -234,7 +239,7 @@ typedef struct {
static int inner_thread(blas_arg_t *args, BLASLONG *range_m, BLASLONG *range_n, IFLOAT *sa, IFLOAT *sb, BLASLONG mypos){
IFLOAT *buffer[DIVIDE_RATE];
IFLOAT *buffer[DIVIDE_RATE_MAX];
BLASLONG k, lda, ldb, ldc;
BLASLONG m_from, m_to, n_from, n_to;
@@ -707,7 +712,7 @@ static int gemm_driver(blas_arg_t *args, BLASLONG *range_m, BLASLONG
while (m > 0){
width = blas_quickdivide(m + nthreads_m - num_parts - 1, nthreads_m - num_parts);
width = round_up(m, width, GEMM_PREFERED_SIZE);
width = round_up(m, width, GEMM_PREFERRED_SIZE);
m -= width;
@@ -758,7 +763,7 @@ static int gemm_driver(blas_arg_t *args, BLASLONG *range_m, BLASLONG
if (width < switch_ratio) {
width = switch_ratio;
}
width = round_up(width_n, width, GEMM_PREFERED_SIZE);
width = round_up(width_n, width, GEMM_PREFERRED_SIZE);
width_n -= width;
if (width_n < 0) {
+1 -1
View File
@@ -27,7 +27,6 @@ if (USE_THREAD)
${BLAS_SERVER}
divtable.c # TODO: Makefile has -UDOUBLE
blas_l1_thread.c
blas_server_callback.c
)
if (NOT NO_AFFINITY)
@@ -42,6 +41,7 @@ set(COMMON_SOURCES
openblas_env.c
openblas_get_num_procs.c
openblas_get_num_threads.c
blas_server_callback.c
)
# these need to have NAME/CNAME set, so use GenerateNamedObjects, but don't use standard name mangling
+2 -2
View File
@@ -1,12 +1,12 @@
TOPDIR = ../..
include ../../Makefile.system
COMMONOBJS = memory.$(SUFFIX) xerbla.$(SUFFIX) c_abs.$(SUFFIX) z_abs.$(SUFFIX) openblas_set_num_threads.$(SUFFIX) openblas_get_num_threads.$(SUFFIX) openblas_get_num_procs.$(SUFFIX) openblas_get_config.$(SUFFIX) openblas_get_parallel.$(SUFFIX) openblas_error_handle.$(SUFFIX) openblas_env.$(SUFFIX)
COMMONOBJS = memory.$(SUFFIX) xerbla.$(SUFFIX) c_abs.$(SUFFIX) z_abs.$(SUFFIX) openblas_set_num_threads.$(SUFFIX) openblas_get_num_threads.$(SUFFIX) openblas_get_num_procs.$(SUFFIX) openblas_get_config.$(SUFFIX) openblas_get_parallel.$(SUFFIX) openblas_error_handle.$(SUFFIX) openblas_env.$(SUFFIX) blas_server_callback.$(SUFFIX)
#COMMONOBJS += slamch.$(SUFFIX) slamc3.$(SUFFIX) dlamch.$(SUFFIX) dlamc3.$(SUFFIX)
ifdef SMP
COMMONOBJS += blas_server.$(SUFFIX) divtable.$(SUFFIX) blasL1thread.$(SUFFIX) blas_server_callback.$(SUFFIX)
COMMONOBJS += blas_server.$(SUFFIX) divtable.$(SUFFIX) blasL1thread.$(SUFFIX)
ifneq ($(NO_AFFINITY), 1)
COMMONOBJS += init.$(SUFFIX)
endif
+50 -16
View File
@@ -38,6 +38,7 @@
/*********************************************************************/
#include "common.h"
#include <strings.h>
#if (defined OS_LINUX || defined OS_ANDROID)
#include <asm/hwcap.h>
#include <sys/auxv.h>
@@ -128,6 +129,18 @@ extern gotoblas_t gotoblas_ARMV9SME;
#else
#define gotoblas_ARMV9SME gotoblas_ARMV8
#endif
#ifdef DYN_VORTEX
extern gotoblas_t gotoblas_VORTEX;
#elif defined(DYN_NEOVERSEN1)
#define gotoblas_VORTEX gotoblas_NEOVERSEN1
#else
#define gotoblas_VORTEX gotoblas_ARMV8
#endif
#ifdef DYN_VORTEXM4
extern gotoblas_t gotoblas_VORTEXM4;
#else
#define gotoblas_VORTEXM4 gotoblas_ARMV8
#endif
#ifdef DYN_CORTEXA55
extern gotoblas_t gotoblas_CORTEXA55;
#else
@@ -138,7 +151,7 @@ extern gotoblas_t gotoblas_A64FX;
#else
#define gotoblas_A64FX gotoblas_ARMV8
#endif
#else
#else //not a user-specified dynamic_list
extern gotoblas_t gotoblas_CORTEXA53;
#define gotoblas_CORTEXA55 gotoblas_CORTEXA53
extern gotoblas_t gotoblas_CORTEXA57;
@@ -150,22 +163,32 @@ extern gotoblas_t gotoblas_THUNDERX2T99;
extern gotoblas_t gotoblas_TSV110;
extern gotoblas_t gotoblas_EMAG8180;
extern gotoblas_t gotoblas_NEOVERSEN1;
#define gotoblas_VORTEX gotoblas_NEOVERSEN1
#ifndef NO_SVE
extern gotoblas_t gotoblas_NEOVERSEV1;
extern gotoblas_t gotoblas_NEOVERSEN2;
extern gotoblas_t gotoblas_ARMV8SVE;
extern gotoblas_t gotoblas_A64FX;
#ifndef NO_SME
extern gotoblas_t gotoblas_ARMV9SME;
#else
#define gotoblas_ARMV9SME gotoblas_ARMV8SVE
#endif
#else
#define gotoblas_NEOVERSEV1 gotoblas_ARMV8
#define gotoblas_NEOVERSEN2 gotoblas_ARMV8
#define gotoblas_ARMV8SVE gotoblas_ARMV8
#define gotoblas_A64FX gotoblas_ARMV8
#define gotoblas_ARMV9SME gotoblas_ARMV8
#endif
#ifndef NO_SME
extern gotoblas_t gotoblas_ARMV9SME;
#if defined (__clang__) && defined(OS_DARWIN)
extern gotoblas_t gotoblas_VORTEXM4;
#else
#define gotoblas_VORTEXM4 gotoblas_NEOVERSEN1
#endif
#else
#ifndef NO_SVE
#define gotoblas_ARMV9SME gotoblas_ARMV8SVE
#else
#define gotoblas_ARMV9SME gotoblas_NEOVERSEN1
#endif
#define gotoblas_VORTEXM4 gotoblas_NEOVERSEN1
#endif
extern gotoblas_t gotoblas_THUNDERX3T110;
@@ -176,7 +199,7 @@ extern void openblas_warning(int verbose, const char * msg);
#define FALLBACK_VERBOSE 1
#define NEOVERSEN1_FALLBACK "OpenBLAS : Your OS does not support SVE instructions. OpenBLAS is using Neoverse N1 kernels as a fallback, which may give poorer performance.\n"
#define NUM_CORETYPES 19
#define NUM_CORETYPES 21
/*
* In case asm/hwcap.h is outdated on the build system, make sure
@@ -216,6 +239,8 @@ static char *corename[] = {
"armv8sve",
"a64fx",
"armv9sme",
"vortex",
"vortexm4",
"unknown"
};
@@ -239,6 +264,8 @@ char *gotoblas_corename(void) {
if (gotoblas == &gotoblas_ARMV8SVE) return corename[16];
if (gotoblas == &gotoblas_A64FX) return corename[17];
if (gotoblas == &gotoblas_ARMV9SME) return corename[18];
if (gotoblas == &gotoblas_VORTEX) return corename[19];
if (gotoblas == &gotoblas_VORTEXM4) return corename[20];
return corename[NUM_CORETYPES];
}
@@ -277,6 +304,8 @@ static gotoblas_t *force_coretype(char *coretype) {
case 16: return (&gotoblas_ARMV8SVE);
case 17: return (&gotoblas_A64FX);
case 18: return (&gotoblas_ARMV9SME);
case 19: return (&gotoblas_VORTEX);
case 20: return (&gotoblas_VORTEXM4);
}
snprintf(message, 128, "Core not found: %s\n", coretype);
openblas_warning(1, message);
@@ -288,12 +317,12 @@ static gotoblas_t *get_coretype(void) {
char coremsg[128];
#if defined (OS_DARWIN)
//future #if !defined(NO_SME)
// if (support_sme1()) {
// return &gotoblas_ARMV9SME;
// }
// #endif
return &gotoblas_NEOVERSEN1;
#if !defined(NO_SME)
if (support_sme1()) {
return &gotoblas_VORTEXM4;
}
#endif
return &gotoblas_VORTEX;
#endif
#if (!defined OS_LINUX && !defined OS_ANDROID)
@@ -378,6 +407,8 @@ static gotoblas_t *get_coretype(void) {
case 0xd08: // Cortex A72
return &gotoblas_CORTEXA72;
case 0xd09: // Cortex A73
case 0xd0a: // Cortex A75
case 0xd0b: // Cortex A76
return &gotoblas_CORTEXA73;
case 0xd0c: // Neoverse N1
return &gotoblas_NEOVERSEN1;
@@ -395,6 +426,9 @@ static gotoblas_t *get_coretype(void) {
}else
return &gotoblas_NEOVERSEV1;
case 0xd4f:
case 0xd83:
case 0xd85:
case 0xd87:
if (!(getauxval(AT_HWCAP) & HWCAP_SVE)) {
openblas_warning(FALLBACK_VERBOSE, NEOVERSEN1_FALLBACK);
return &gotoblas_NEOVERSEN1;
@@ -463,8 +497,8 @@ static gotoblas_t *get_coretype(void) {
}
break;
case 0x61: // Apple
//future if (support_sme1()) return &gotoblas_ARMV9SME;
return &gotoblas_NEOVERSEN1;
if (support_sme1()) return &gotoblas_VORTEXM4;
return &gotoblas_VORTEX;
break;
default:
snprintf(coremsg, 128, "Unknown CPU model - implementer %x part %x\n",implementer,part);
+1 -1
View File
@@ -99,7 +99,7 @@ struct riscv_hwprobe {
#define RISCV_HWPROBE_IMA_V (1 << 2)
#define RISCV_HWPROBE_EXT_ZFH (1 << 27)
#define RISCV_HWPROBE_EXT_ZVFH (1 << 30)
#define RISCV_HWPROBE_EXT_ZVFBFWMA (1 << 54)
#define RISCV_HWPROBE_EXT_ZVFBFWMA (1ULL << 54)
#ifndef NR_riscv_hwprobe
#ifndef NR_arch_specific_syscall
+20 -7
View File
@@ -1317,7 +1317,11 @@ UNLOCK_COMMAND(&alloc_lock);
error:
printf("OpenBLAS : Program will terminate because you tried to allocate too many TLS memory regions.\n");
printf("This library was built to support a maximum of %d threads - either rebuild OpenBLAS\n", NUM_BUFFERS);
printf("with a larger NUM_THREADS value or set the environment variable OPENBLAS_NUM_THREADS to\n");
#ifdef USE_OPENMP
printf("with a larger NUM_THREADS value or set the environment variable OMP_NUM_THREADS to\n");
#else
printf("with a larger NUM_THREADS value or set the environment variable OPENBLAS_NUM_THREADS to\n");
#endif
printf("a sufficiently small number. This error typically occurs when the software that relies on\n");
printf("OpenBLAS calls BLAS functions from many threads in parallel, or when your computer has more\n");
printf("cpu cores than what OpenBLAS was configured to handle.\n");
@@ -1601,7 +1605,7 @@ void DESTRUCTOR gotoblas_quit(void) {
}
#if defined(_MSC_VER) && !defined(__clang__)
BOOL APIENTRY DllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved)
BOOL APIENTRY OpenBLASDllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved)
{
switch (ul_reason_for_call)
{
@@ -1650,10 +1654,10 @@ static int on_process_term(void)
#endif
#ifdef _WIN64
static const PIMAGE_TLS_CALLBACK dll_callback(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = DllMain;
static const PIMAGE_TLS_CALLBACK dll_callback(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = OpenBLASDllMain;
#pragma const_seg()
#else
static void (APIENTRY *dll_callback)(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = DllMain;
static void (APIENTRY *dll_callback)(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = OpenBLASDllMain;
#pragma data_seg()
#endif
@@ -3039,8 +3043,13 @@ void *blas_memory_alloc(int procpos){
#endif
if (memory_overflowed) goto terminate;
fprintf(stderr,"OpenBLAS warning: precompiled NUM_THREADS exceeded, adding auxiliary array for thread metadata.\n");
fprintf(stderr,"Note that your application may still crash, if it is calling OpenBLAS from multiple threads in parallel\n");
fprintf(stderr,"To avoid this warning, please rebuild your copy of OpenBLAS with a larger NUM_THREADS setting\n");
#ifdef USE_OPENMP
fprintf(stderr,"or set the environment variable OMP_NUM_THREADS to %d or lower\n", MAX_CPU_NUMBER);
#else
fprintf(stderr,"or set the environment variable OPENBLAS_NUM_THREADS to %d or lower\n", MAX_CPU_NUMBER);
#endif
memory_overflowed=1;
MB;
new_release_info = (struct release_t*) malloc(NEW_BUFFERS * sizeof(struct release_t));
@@ -3142,7 +3151,11 @@ terminate:
#endif
printf("OpenBLAS : Program is Terminated. Because you tried to allocate too many memory regions.\n");
printf("This library was built to support a maximum of %d threads - either rebuild OpenBLAS\n", NUM_BUFFERS);
printf("with a larger NUM_THREADS value or set the environment variable OPENBLAS_NUM_THREADS to\n");
#ifdef USE_OPENMP
printf("with a larger NUM_THREADS value or set the environment variable OMP_NUM_THREADS to\n");
#else
printf("with a larger NUM_THREADS value or set the environment variable OPENBLAS_NUM_THREADS to\n");
#endif
printf("a sufficiently small number. This error typically occurs when the software that relies on\n");
printf("OpenBLAS calls BLAS functions from many threads in parallel, or when your computer has more\n");
printf("cpu cores than what OpenBLAS was configured to handle.\n");
@@ -3473,7 +3486,7 @@ void DESTRUCTOR gotoblas_quit(void) {
}
#if defined(_MSC_VER) && !defined(__clang__)
BOOL APIENTRY DllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved)
BOOL APIENTRY OpenBLASDllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved)
{
switch (ul_reason_for_call)
{
@@ -3517,7 +3530,7 @@ static int on_process_term(void)
#else
#pragma data_seg(".CRT$XLB")
#endif
static void (APIENTRY *dll_callback)(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = DllMain;
static void (APIENTRY *dll_callback)(HINSTANCE h, DWORD ul_reason_for_call, PVOID pv) = OpenBLASDllMain;
#ifdef _WIN64
#pragma const_seg()
#else
+4
View File
@@ -162,11 +162,15 @@ ifeq ($(F_COMPILER), INTEL)
else
ifeq ($(F_COMPILER), FLANG)
$(FC) $(FFLAGS) $(LDFLAGS) -fno-fortran-main -Mnomain -all_load -headerpad_max_install_names -install_name "$(CURDIR)/../$(INTERNALNAME)" -dynamiclib -o ../$(LIBDYNNAME) $< -Wl,-exported_symbols_list,osx.def $(FEXTRALIB)
else
ifeq ($(F_COMPILER), FLANGNEW)
$(FC) $(FFLAGS) $(LDFLAGS) -Wl,-all_load -Wl,-headerpad_max_install_names -Wl,-install_name,"$(CURDIR)/../$(INTERNALNAME)" -Wl,-dylib -o ../$(LIBDYNNAME) $< -Wl,-exported_symbols_list,osx.def $(FEXTRALIB)
else
$(FC) $(FFLAGS) $(LDFLAGS) -all_load -headerpad_max_install_names -install_name "$(CURDIR)/../$(INTERNALNAME)" -dynamiclib -o ../$(LIBDYNNAME) $< -Wl,-exported_symbols_list,osx.def $(FEXTRALIB)
endif
endif
endif
endif
dllinit.$(SUFFIX) : dllinit.c
$(CC) $(CFLAGS) -c -o $(@F) -s $<
+1
View File
@@ -181,6 +181,7 @@ misc_no_underscore_objs="
goto_set_num_threads
openblas_get_config
openblas_get_corename
openblas_set_threads_callback_function
"
misc_underscore_objs=""
+1
View File
@@ -177,6 +177,7 @@
goto_set_num_threads,
openblas_get_config,
openblas_get_corename,
openblas_set_threads_callback_function,
);
@misc_underscore_objs = (
+5 -5
View File
@@ -92,7 +92,7 @@ else
vendor=FLANG
openmp='-fopenmp'
;;
*GNU*|*GCC*)
*GCC*)
v="${data#*GCC: *\) }"
v="${v%%\"*}"
@@ -343,13 +343,13 @@ linker_a=""
if [ -n "$link" ]; then
link=`echo "$link" | sed 's/\-Y[[:space:]]P\,/\-Y/g'`
link=`echo " $link" | sed 's/ \-Y[[:space:]]P\,/ \-Y/g'`
link=`echo "$link" | sed 's/\-R[[:space:]]*/\-rpath\%/g'`
link=`echo "$link" | sed 's/ \-R[[:space:]]*/ \-rpath\%/g'`
link=`echo "$link" | sed 's/\-rpath[[:space:]]+/\-rpath\%/g'`
link=`echo "$link" | sed 's/ \-rpath[[:space:]]+/ \-rpath\%/g'`
link=`echo "$link" | sed 's/\-rpath-link[[:space:]]+/\-rpath-link\%/g'`
link=`echo "$link" | sed 's/ \-rpath-link[[:space:]]+/ \-rpath-link\%/g'`
flags=`echo "$link" | tr "',\n" " "`
# remove leading and trailing quotes from each flag.
+40
View File
@@ -1232,6 +1232,20 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#else
#endif
#ifdef FORCE_WASM128_GENERIC
#define FORCE
#define ARCHITECTURE "WASM"
#define SUBARCHITECTURE "WASM128_GENERIC"
#define SUBDIRNAME "wasm"
#define ARCHCONFIG "-DWASM128_GENERIC " \
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=32 " \
"-DL2_SIZE=1048576 -DL2_LINESIZE=32 " \
"-DDTB_DEFAULT_ENTRIES=128 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=4 "
#define LIBNAME "wasm128"
#define CORENAME "WASM128_GENERIC"
#else
#endif
#ifdef FORCE_CORTEXA15
#define FORCE
#define ARCHITECTURE "ARM"
@@ -1654,6 +1668,28 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#define CORENAME "VORTEX"
#endif
#ifdef FORCE_VORTEXM4
#define FORCE
#define ARCHITECTURE "ARM64"
#define SUBARCHITECTURE "VORTEXM4"
#define SUBDIRNAME "arm64"
#ifdef __clang__
#define ARCHCONFIG "-DVORTEXM4 " \
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=32 " \
"-DHAVE_VFPV4 -DHAVE_VFPV3 -DHAVE_VFP -DHAVE_NEON -DHAVE_SME -DARMV8"
#else
#define ARCHCONFIG "-DVORTEX " \
"-DL1_DATA_SIZE=32768 -DL1_DATA_LINESIZE=64 " \
"-DL2_SIZE=262144 -DL2_LINESIZE=64 " \
"-DDTB_DEFAULT_ENTRIES=64 -DDTB_SIZE=4096 -DL2_ASSOCIATIVE=32 " \
"-DHAVE_VFPV4 -DHAVE_VFPV3 -DHAVE_VFP -DHAVE_NEON -DARMV8"
#endif
#define LIBNAME "vortexm4"
#define CORENAME "VORTEXM4"
#endif
#ifdef FORCE_A64FX
#define ARMV8
#define FORCE
@@ -1927,6 +1963,10 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#define OPENBLAS_SUPPORTED
#endif
#ifdef __wasm__
#define OPENBLAS_SUPPORTED
#endif
#ifndef OPENBLAS_SUPPORTED
#error "This arch/CPU is not supported by OpenBLAS."
#endif
+2 -2
View File
@@ -530,8 +530,8 @@ ifneq ($(NO_LAPACK), 1)
SBLASOBJS += $(SLAPACKOBJS)
DBLASOBJS += $(DLAPACKOBJS)
#QBLASOBJS += $(QLAPACKOBJS)
CBLASOBJS += $(CLAPACKOBJS)
ZBLASOBJS += $(ZLAPACKOBJS)
CBLASOBJS += $(CLAPACKOBJS) slaed3.$(SUFFIX)
ZBLASOBJS += $(ZLAPACKOBJS) dlaed3.$(SUFFIX)
#XBLASOBJS += $(XLAPACKOBJS)
endif
+36 -25
View File
@@ -184,11 +184,11 @@ static int init_amxtile_permission() {
}
#endif
#ifdef SMP
#ifdef DYNAMIC_ARCH
extern char* gotoblas_corename(void);
#endif
#ifdef SMP
#if defined(DYNAMIC_ARCH) || defined(NEOVERSEV1)
static inline int get_gemm_optimal_nthreads_neoversev1(double MNK, int ncpu) {
return
@@ -266,6 +266,7 @@ void NAME(char *TRANSA, char *TRANSB,
int transa, transb, nrowa, nrowb;
blasint info;
int order = -1;
char transA, transB;
IFLOAT *buffer;
@@ -424,30 +425,6 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_TRANSPOSE TransA, enum CBLAS_TRANS
PRINT_DEBUG_CNAME;
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
#if defined(ARCH_x86) && (defined(USE_SGEMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
#if defined(DYNAMIC_ARCH)
if (support_avx512() )
#endif
if (beta == 0 && alpha == 1.0 && order == CblasRowMajor && TransA == CblasNoTrans && TransB == CblasNoTrans && SGEMM_DIRECT_PERFORMANT(m,n,k)) {
SGEMM_DIRECT(m, n, k, a, lda, b, ldb, c, ldc);
return;
}
#endif
#if defined(ARCH_ARM64) && (defined(USE_SGEMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
#if defined(DYNAMIC_ARCH)
if (support_sme1())
#endif
if (beta == 0 && alpha == 1.0 && order == CblasRowMajor && TransA == CblasNoTrans && TransB == CblasNoTrans) {
SGEMM_DIRECT(m, n, k, a, lda, b, ldb, c, ldc);
return;
}else if (order == CblasRowMajor && TransA == CblasNoTrans && TransB == CblasNoTrans) {
SGEMM_DIRECT_ALPHA_BETA(m, n, k, alpha, a, lda, b, ldb, beta, c, ldc);
return;
}
#endif
#endif
#ifndef COMPLEX
args.alpha = (void *)&alpha;
args.beta = (void *)&beta;
@@ -564,6 +541,40 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_TRANSPOSE TransA, enum CBLAS_TRANS
return;
}
if ((args.m == 0) || (args.n == 0)) return;
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
#if defined(ARCH_x86) && (defined(USE_SGEMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
#if defined(DYNAMIC_ARCH)
if (support_avx512() )
#endif
if (order == CblasRowMajor && beta == 0 && alpha == 1.0 && TransA == CblasNoTrans && TransB == CblasNoTrans && SGEMM_DIRECT_PERFORMANT(m,n,k)) {
SGEMM_DIRECT(m, n, k, a, lda, b, ldb, c, ldc);
return;
}
#endif
#if defined(ARCH_ARM64) && (defined(USE_SGEMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
#if defined(DYNAMIC_ARCH)
if (strcmp(gotoblas_corename(), "armv9sme") == 0
#if defined(__clang__)
|| strcmp(gotoblas_corename(), "vortexm4") == 0
#endif
)
// if (support_sme1())
#endif
if (order == CblasRowMajor && m==lda && n ==ldb && k==ldc && beta == 0 && alpha == 1.0 && TransA == CblasNoTrans && TransB == CblasNoTrans&& SGEMM_DIRECT_PERFORMANT(m,n,k)) {
SGEMM_DIRECT(m, n, k, a, lda, b, ldb, c, ldc);
return;
}
else
if (order == CblasRowMajor && m==lda && n==ldb && k==ldc && TransA == CblasNoTrans && TransB == CblasNoTrans&& SGEMM_DIRECT_PERFORMANT(m,n,k)) {
SGEMM_DIRECT_ALPHA_BETA(m, n, k, alpha, a, lda, b, ldb, beta, c, ldc);
return;
}
#endif
#endif
#endif
#if defined(__linux__) && defined(__x86_64__) && defined(BFLOAT16)
+9 -5
View File
@@ -1,4 +1,5 @@
/*********************************************************************/
/* Copyright 2025 The OpenBLAS Project */
/* Copyright 2009, 2010 The University of Texas at Austin. */
/* All rights reserved. */
/* */
@@ -81,9 +82,12 @@ static inline int get_gemv_optimal_nthreads_neoversev1(BLASLONG MN, int ncpu) {
: (MN < 1050625L) ? MIN(ncpu, 40)
: ncpu;
#else
return (MN < 25600L) ? 1
return
(MN < 25600L) ? 1
: (MN < 63001L) ? MIN(ncpu, 4)
: (MN < 459684L) ? MIN(ncpu, 16)
: (MN < 202500L) ? MIN(ncpu, 8)
: (MN < 806404L) ? MIN(ncpu, 16)
: (MN < 1638400L) ? MIN(ncpu, 32)
: ncpu;
#endif
}
@@ -93,9 +97,9 @@ static inline int get_gemv_optimal_nthreads_neoversev1(BLASLONG MN, int ncpu) {
static inline int get_gemv_optimal_nthreads_neoversev2(BLASLONG MN, int ncpu) {
return
MN < 24964L ? 1
: MN < 65536L ? MIN(ncpu, 8)
: MN < 262144L ? MIN(ncpu, 32)
: MN < 1638400L ? MIN(ncpu, 64)
: MN < 145924L ? MIN(ncpu, 8)
: MN < 692224L ? MIN(ncpu, 16)
: MN < 1638400L ? MIN(ncpu, 32)
: ncpu;
}
#endif
+8 -8
View File
@@ -99,7 +99,7 @@ int NAME(blasint *N, blasint *NRHS, FLOAT *a, blasint *ldA, blasint *ipiv,
*Info = 0;
if (args.m == 0 || args.n == 0) return 0;
if (args.m == 0) return 0;
IDEBUG_START;
@@ -117,20 +117,20 @@ int NAME(blasint *N, blasint *NRHS, FLOAT *a, blasint *ldA, blasint *ipiv,
#if defined(_WIN64) && defined(_M_ARM64)
#ifdef COMPLEX
if (args.m * args.n <= 300)
if (args.m * args.m <= 300)
#else
if (args.m * args.n <= 500)
if (args.m * args.m <= 500)
#endif
args.nthreads = 1;
else if (args.m * args.n <= 1000)
else if (args.m * args.m <= 1000)
args.nthreads = 4;
else
args.nthreads = num_cpu_avail(4);
#else
#ifndef DOUBLE
if (args.m * args.n < 40000)
if (args.m * args.m < 40000)
#else
if (args.m * args.n < 10000)
if (args.m * args.m < 10000)
#endif
args.nthreads = 1;
else
@@ -143,7 +143,7 @@ int NAME(blasint *N, blasint *NRHS, FLOAT *a, blasint *ldA, blasint *ipiv,
args.n = *N;
info = GETRF_SINGLE(&args, NULL, NULL, sa, sb, 0);
if (info == 0){
if (info == 0 && *NRHS >0){
args.n = *NRHS;
GETRS_N_SINGLE(&args, NULL, NULL, sa, sb, 0);
}
@@ -154,7 +154,7 @@ int NAME(blasint *N, blasint *NRHS, FLOAT *a, blasint *ldA, blasint *ipiv,
args.n = *N;
info = GETRF_PARALLEL(&args, NULL, NULL, sa, sb, 0);
if (info == 0){
if (info == 0 && *NRHS > 0){
args.n = *NRHS;
GETRS_N_PARALLEL(&args, NULL, NULL, sa, sb, 0);
}
+1 -1
View File
@@ -73,7 +73,7 @@ void CNAME(blasint n, FLOAT alpha, FLOAT *x, blasint incx){
float alpha_float;
SBF16TOS_K(1, &alpha, 1, &alpha_float, 1);
#else
float alpha_float = alpha;
FLOAT alpha_float = alpha;
#endif
if (alpha_float == ONE) return;
+8 -1
View File
@@ -97,6 +97,9 @@
#define GEMM_MULTITHREAD_THRESHOLD 4
#endif
#ifdef DYNAMIC_ARCH
extern char* gotoblas_corename(void);
#endif
#ifdef SMP
#ifndef COMPLEX
@@ -374,7 +377,11 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_SIDE Side, enum CBLAS_UPLO Uplo,
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
#if defined(ARCH_ARM64) && (defined(USE_SSYMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
#if defined(DYNAMIC_ARCH)
if (support_sme1())
if (strcmp(gotoblas_corename(), "armv9sme") == 0
#if defined(__clang__)
|| strcmp(gotoblas_corename(), "vortexm4") == 0
#endif
)
#endif
if (args.m == 0 || args.n == 0) return;
if (order == CblasRowMajor && m == lda && n == ldb && n == ldc)
+38 -1
View File
@@ -345,9 +345,46 @@ void CNAME(enum CBLAS_ORDER order, enum CBLAS_UPLO Uplo, enum CBLAS_TRANSPOSE Tr
return;
}
if (args.n == 0) return;
#ifdef DYNAMIC_ARCH
extern char* gotoblas_corename(void);
#endif
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
#if defined(ARCH_ARM64) && (defined(USE_SSYR2K_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
#if defined(DYNAMIC_ARCH)
if (strcmp(gotoblas_corename(), "armv9sme") == 0
#if defined(__clang__)
|| strcmp(gotoblas_corename(), "vortexm4") == 0
#endif
)
#endif
if (order == CblasRowMajor && n == ldc) {
if (Trans == CblasNoTrans && k == lda && k == ldb) {
if (Uplo == CblasUpper) {
SSYR2K_DIRECT_ALPHA_BETA_UN(n, k, alpha, a, lda, b, ldb, beta, c, ldc);
return;
}else if (Uplo == CblasLower) {
SSYR2K_DIRECT_ALPHA_BETA_LN(n, k, alpha, a, lda, b, ldb, beta, c, ldc);
return;
}
}
else if (Trans == CblasTrans && n == lda && n ==ldb) {
if (Uplo == CblasUpper) {
SSYR2K_DIRECT_ALPHA_BETA_UT(n, k, alpha, a, lda, b, ldb, beta, c, ldc);
return;
}else if (Uplo == CblasLower) {
SSYR2K_DIRECT_ALPHA_BETA_LT(n, k, alpha, a, lda, b, ldb, beta, c, ldc);
return;
}
}
}
#endif
#endif
#endif
if (args.n == 0) return;
IDEBUG_START;
+12 -3
View File
@@ -338,12 +338,22 @@ double NNK;
BLASFUNC(xerbla)(ERROR_NAME, &info, sizeof(ERROR_NAME));
return;
}
if (args.n == 0) return;
#ifdef DYNAMIC_ARCH
extern char* gotoblas_corename(void);
#endif
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
#if defined(ARCH_ARM64) && (defined(USE_SSYRK_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
#if defined(DYNAMIC_ARCH)
if (support_sme1())
if (strcmp(gotoblas_corename(), "armv9sme") == 0
#if defined(__clang__)
|| strcmp(gotoblas_corename(), "vortexm4") == 0
#endif
)
#endif
if (args.n == 0) return;
if (order == CblasRowMajor && n == ldc) {
if (Trans == CblasNoTrans && k == lda) {
(Uplo == CblasUpper ? SSYRK_DIRECT_ALPHA_BETA_UN : SSYRK_DIRECT_ALPHA_BETA_LN)(n, k, alpha, a, lda, beta, c, ldc);
@@ -358,7 +368,6 @@ double NNK;
#endif
if (args.n == 0) return;
IDEBUG_START;
+9 -1
View File
@@ -87,6 +87,10 @@
#define SMP_FACTOR 128
#endif
#ifdef DYNAMIC_ARCH
extern char* gotoblas_corename(void);
#endif
static int (*trsm[])(blas_arg_t *, BLASLONG *, BLASLONG *, FLOAT *, FLOAT *, BLASLONG) = {
#ifndef TRMM
TRSM_LNUU, TRSM_LNUN, TRSM_LNLU, TRSM_LNLN,
@@ -358,7 +362,11 @@ void CNAME(enum CBLAS_ORDER order,
#if !defined(COMPLEX) && !defined(DOUBLE) && !defined(BFLOAT16) && !defined(HFLOAT16)
#if defined(ARCH_ARM64) && (defined(USE_STRMM_KERNEL_DIRECT)||defined(DYNAMIC_ARCH))
#if defined(DYNAMIC_ARCH)
if (support_sme1())
if (strcmp(gotoblas_corename(), "armv9sme") == 0
#if defined(__clang__)
|| strcmp(gotoblas_corename(), "vortexm4") == 0
#endif
)
#endif
if (args.m == 0 || args.n == 0) return;
if (order == CblasRowMajor && Diag == CblasNonUnit && Side == CblasLeft && m == lda && n == ldb) {
+21 -5
View File
@@ -48,7 +48,7 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
endif ()
if (${ADD_COMMONOBJS})
if (X86)
if (X86 AND NOT EMSCRIPTEN)
if (NOT "${CMAKE_C_COMPILER_ID}" STREQUAL "MSVC")
GenerateNamedObjects("${KERNELDIR}/cpuid.S" "" "" false "" "" true)
else()
@@ -235,7 +235,7 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
# Makefile.L3
set(USE_TRMM false)
string(TOUPPER ${TARGET_CORE} UC_TARGET_CORE)
if (ARM OR ARM64 OR RISCV64 OR (UC_TARGET_CORE MATCHES LONGSOON3B) OR (UC_TARGET_CORE MATCHES GENERIC) OR (UC_TARGET_CORE MATCHES HASWELL) OR (UC_TARGET_CORE MATCHES ZEN) OR (UC_TARGET_CORE MATCHES SKYLAKEX) OR (UC_TARGET_CORE MATCHES COOPERLAKE) OR (UC_TARGET_CORE MATCHES SAPPHIRERAPIDS))
if (ARM OR ARM64 OR RISCV64 OR WASM OR (UC_TARGET_CORE MATCHES LONGSOON3B) OR (UC_TARGET_CORE MATCHES GENERIC) OR (UC_TARGET_CORE MATCHES HASWELL) OR (UC_TARGET_CORE MATCHES ZEN) OR (UC_TARGET_CORE MATCHES SKYLAKEX) OR (UC_TARGET_CORE MATCHES COOPERLAKE) OR (UC_TARGET_CORE MATCHES SAPPHIRERAPIDS))
set(USE_TRMM true)
endif ()
if (ZARCH OR (UC_TARGET_CORE MATCHES POWER8) OR (UC_TARGET_CORE MATCHES POWER9) OR (UC_TARGET_CORE MATCHES POWER10))
@@ -249,6 +249,10 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
if (ARM64)
set(USE_DIRECT_SSYRK true)
endif()
set(USE_DIRECT_SSYR2K false)
if (ARM64)
set(USE_DIRECT_SSYR2K true)
endif()
set(USE_DIRECT_SGEMM false)
if (X86_64 OR ARM64)
set(USE_DIRECT_SGEMM true)
@@ -257,7 +261,7 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
if (ARM64)
set(USE_DIRECT_SSYMM true)
endif()
if (UC_TARGET_CORE MATCHES ARMV9SME)
if (UC_TARGET_CORE MATCHES ARMV9SME OR UC_TARGET_CORE MATCHES VORTEXM4)
set (HAVE_SME true)
endif ()
@@ -270,14 +274,16 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTKERNEL}" "" "gemm_direct" false "" "" false SINGLE)
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTPERFORMANT}" "" "gemm_direct_performant" false "" "" false SINGLE)
elseif (ARM64)
set (SGEMMDIRECTPERFORMANT sgemm_direct_performant.c)
set (SGEMMDIRECTKERNEL sgemm_direct_arm64_sme1.c)
set (SGEMMDIRECTKERNEL_ALPHA_BETA sgemm_direct_alpha_beta_arm64_sme1.c)
set (SGEMMDIRECTSMEKERNEL sgemm_direct_sme1.S)
set (SGEMMDIRECTSMEKERNEL sgemm_direct_sme1_2VLx2VL.S)
set (SGEMMDIRECTPREKERNEL sgemm_direct_sme1_preprocess.S)
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTPERFORMANT}" "" "gemm_direct_performant" false "" "" false SINGLE)
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTKERNEL}" "" "gemm_direct" false "" "" false SINGLE)
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTKERNEL_ALPHA_BETA}" "" "gemm_direct_alpha_beta" false "" "" false SINGLE)
if (HAVE_SME)
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTSMEKERNEL}" "" "gemm_direct_sme1" false "" "" false SINGLE)
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTSMEKERNEL}" "" "gemm_direct_sme1_2VLx2VL" false "" "" false SINGLE)
GenerateNamedObjects("${KERNELDIR}/${SGEMMDIRECTPREKERNEL}" "" "gemm_direct_sme1_preprocess" false "" "" false SINGLE)
endif ()
endif ()
@@ -311,6 +317,16 @@ function (build_core TARGET_CORE KDIR TSUFFIX KERNEL_DEFINITIONS)
endif ()
endif()
if (USE_DIRECT_SSYR2K)
if (ARM64)
set (SSYR2KDIRECTKERNEL_ALPHA_BETA ssyr2k_direct_alpha_beta_arm64_sme1.c)
GenerateNamedObjects("${KERNELDIR}/${SSYR2KDIRECTKERNEL_ALPHA_BETA}" "" "syr2k_direct_alpha_betaUN" false "" "" false SINGLE)
GenerateNamedObjects("${KERNELDIR}/${SSYR2KDIRECTKERNEL_ALPHA_BETA}" "" "syr2k_direct_alpha_betaUT" false "" "" false SINGLE)
GenerateNamedObjects("${KERNELDIR}/${SSYR2KDIRECTKERNEL_ALPHA_BETA}" "" "syr2k_direct_alpha_betaLN" false "" "" false SINGLE)
GenerateNamedObjects("${KERNELDIR}/${SSYR2KDIRECTKERNEL_ALPHA_BETA}" "" "syr2k_direct_alpha_betaLT" false "" "" false SINGLE)
endif ()
endif()
foreach (float_type SINGLE DOUBLE)
string(SUBSTRING ${float_type} 0 1 float_char)
GenerateNamedObjects("${KERNELDIR}/${${float_char}GEMMKERNEL}" "" "gemm_kernel" false "" "" false ${float_type})
+23 -1
View File
@@ -27,7 +27,29 @@ endif
ifdef TARGET_CORE
ifeq ($(TARGET_CORE), ARMV9SME)
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE) -DHAVE_SME -march=armv9-a+sve2+sme
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE) -march=armv9-a+sve2+sme
ifdef OS_WINDOWS
ifeq ($(C_COMPILER), CLANG)
override CFLAGS += --aarch64-stack-hazard-size=0
endif
endif
endif
ifeq ($(TARGET_CORE), VORTEXM4)
ifeq ($(C_COMPILER), GCC)
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE) -UHAVE_SME -march=armv8.4-a
else
ifeq ($(APPLECLANG),1)
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE) -march=armv8.4-a+sme
else
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE) -march=armv8.4-a+sme
override LDFLAGS += -lclang_rt_builtins-aarch64
endif
ifdef OS_WINDOWS
ifeq ($(C_COMPILER), CLANG)
override CFLAGS += --aarch64-stack-hazard-size=0
endif
endif
endif
endif
ifeq ($(TARGET_CORE), SAPPHIRERAPIDS)
override CFLAGS += -DBUILD_KERNEL -DTABLE_NAME=gotoblas_$(TARGET_CORE)
+67 -26
View File
@@ -53,14 +53,27 @@ ifeq ($(ARCH), arm64)
USE_TRMM = 1
USE_DIRECT_SGEMM = 1
USE_DIRECT_SSYMM = 1
USE_DIRECT_STRMM = 1
USE_DIRECT_SSYRK = 1
USE_DIRECT_SSYR2K = 1
USE_DIRECT_STRMM = 1
ifeq ($(CORE), ARMV9SME)
USE_SME = 1
endif
ifeq ($(CORE), VORTEXM4)
ifneq ($(C_COMPILER), GCC)
USE_SME = 1
endif
endif
endif
ifeq ($(ARCH), riscv64)
USE_TRMM = 1
endif
ifeq ($(ARCH), wasm)
USE_TRMM = 1
endif
ifneq ($(DYNAMIC_ARCH), 1)
ifeq ($(TARGET), GENERIC)
USE_TRMM = 1
@@ -131,11 +144,9 @@ SGEMMDIRECTKERNEL = sgemm_direct_skylakex.c
SGEMMDIRECTPERFORMANT = sgemm_direct_performant.c
endif
ifeq ($(ARCH), arm64)
ifeq ($(TARGET_CORE), ARMV9SME)
HAVE_SME = 1
endif
SGEMMDIRECTKERNEL = sgemm_direct_arm64_sme1.c
SGEMMDIRECTKERNEL_ALPHA_BETA = sgemm_direct_alpha_beta_arm64_sme1.c
SGEMMDIRECTPERFORMANT = sgemm_direct_performant.c
endif
endif
endif
@@ -143,9 +154,6 @@ endif
ifdef USE_DIRECT_SSYMM
ifndef SSYMMDIRECTKERNEL_ALPHA_BETA
ifeq ($(ARCH), arm64)
ifeq ($(TARGET_CORE), ARMV9SME)
HAVE_SME = 1
endif
SSYMMDIRECTKERNEL_ALPHA_BETA = ssymm_direct_alpha_beta_arm64_sme1.c
endif
endif
@@ -154,9 +162,6 @@ endif
ifdef USE_DIRECT_STRMM
ifndef STRMMDIRECTKERNEL
ifeq ($(ARCH), arm64)
ifeq ($(TARGET_CORE), ARMV9SME)
HAVE_SME = 1
endif
STRMMDIRECTKERNEL = strmm_direct_arm64_sme1.c
endif
endif
@@ -165,10 +170,18 @@ endif
ifdef USE_DIRECT_SSYRK
ifndef SSYRKDIRECTKERNEL_ALPHA_BETA
ifeq ($(ARCH), arm64)
SSYRKDIRECTKERNEL_ALPHA_BETA = ssyrk_direct_alpha_beta_arm64_sme1.c
endif
endif
endif
ifdef USE_DIRECT_SSYR2K
ifndef SSYR2KDIRECTKERNEL_ALPHA_BETA
ifeq ($(ARCH), arm64)
ifeq ($(TARGET_CORE), ARMV9SME)
HAVE_SME = 1
endif
SSYRKDIRECTKERNEL_ALPHA_BETA = ssyrk_direct_alpha_beta_arm64_sme1.c
SSYR2KDIRECTKERNEL_ALPHA_BETA = ssyr2k_direct_alpha_beta_arm64_sme1.c
endif
endif
endif
@@ -245,11 +258,12 @@ SKERNELOBJS += \
endif
ifeq ($(ARCH), arm64)
SKERNELOBJS += \
sgemm_direct_performant$(TSUFFIX).$(SUFFIX) \
sgemm_direct$(TSUFFIX).$(SUFFIX) \
sgemm_direct_alpha_beta$(TSUFFIX).$(SUFFIX)
ifdef HAVE_SME
ifdef USE_SME
SKERNELOBJS += \
sgemm_direct_sme1$(TSUFFIX).$(SUFFIX) \
sgemm_direct_sme1_2VLx2VL$(TSUFFIX).$(SUFFIX) \
sgemm_direct_sme1_preprocess$(TSUFFIX).$(SUFFIX)
endif
endif
@@ -275,8 +289,18 @@ endif
ifdef USE_DIRECT_SSYRK
ifeq ($(ARCH), arm64)
SKERNELOBJS += \
ssyrk_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) ssyrk_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) \
ssyrk_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) ssyrk_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX)
ssyrk_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) ssyrk_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) \
ssyrk_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) ssyrk_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX)
endif
endif
ifdef USE_DIRECT_SSYR2K
ifeq ($(ARCH), arm64)
SKERNELOBJS += \
ssyr2k_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) ssyr2k_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) \
ssyr2k_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) ssyr2k_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) \
ssyr2k_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) ssyr2k_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) \
ssyr2k_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX) ssyr2k_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX)
endif
endif
@@ -1029,13 +1053,15 @@ $(KDIR)sgemm_direct$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMMDIRECTKERNEL)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
endif
ifeq ($(ARCH), arm64)
$(KDIR)sgemm_direct_performant$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMMDIRECTPERFORMANT)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
$(KDIR)sgemm_direct$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMMDIRECTKERNEL)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
$(KDIR)sgemm_direct_alpha_beta$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SGEMMDIRECTKERNEL_ALPHA_BETA)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX $< -o $@
ifdef HAVE_SME
$(KDIR)sgemm_direct_sme1$(TSUFFIX).$(SUFFIX) :
$(CC) $(CFLAGS) -c $(KERNELDIR)/sgemm_direct_sme1.S -UDOUBLE -UCOMPLEX -o $@
ifdef USE_SME
$(KDIR)sgemm_direct_sme1_2VLx2VL$(TSUFFIX).$(SUFFIX) :
$(CC) $(CFLAGS) -c $(KERNELDIR)/sgemm_direct_sme1_2VLx2VL.S -UDOUBLE -UCOMPLEX -o $@
$(KDIR)sgemm_direct_sme1_preprocess$(TSUFFIX).$(SUFFIX) :
$(CC) $(CFLAGS) -c $(KERNELDIR)/sgemm_direct_sme1_preprocess.S -UDOUBLE -UCOMPLEX -o $@
endif
@@ -1051,6 +1077,22 @@ $(KDIR)ssymm_direct_alpha_betaLL$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYMMDIREC
endif
endif
ifdef USE_DIRECT_SSYRK
ifeq ($(ARCH), arm64)
$(KDIR)ssyrk_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DUPPER -UTRANSA $< -o $@
$(KDIR)ssyrk_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DUPPER -DTRANSA $< -o $@
$(KDIR)ssyrk_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -UUPPER -UTRANSA $< -o $@
$(KDIR)ssyrk_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -UUPPER -DTRANSA $< -o $@
endif
endif
ifeq ($(BUILD_BFLOAT16), 1)
$(KDIR)bgemm_kernel$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(BGEMMKERNEL)
$(CC) $(CFLAGS) -c -DBFLOAT16 -DBGEMM -UDOUBLE -UCOMPLEX $< -o $@
@@ -1177,19 +1219,18 @@ $(KDIR)xgemm_kernel_r$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(XGEMMKERNEL) $(XGEMMD
$(KDIR)xgemm_kernel_b$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(XGEMMKERNEL) $(XGEMMDEPEND)
$(CC) $(CFLAGS) -c -DXDOUBLE -DCOMPLEX -DCC $< -o $@
ifdef USE_DIRECT_SSYRK
ifdef USE_DIRECT_SSYR2K
ifeq ($(ARCH), arm64)
$(KDIR)ssyrk_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
$(KDIR)ssyr2k_direct_alpha_betaUN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYR2KDIRECTKERNEL_ALPHA_BETA)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DUPPER -UTRANSA $< -o $@
$(KDIR)ssyrk_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
$(KDIR)ssyr2k_direct_alpha_betaUT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYR2KDIRECTKERNEL_ALPHA_BETA)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -DUPPER -DTRANSA $< -o $@
$(KDIR)ssyrk_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
$(KDIR)ssyr2k_direct_alpha_betaLN$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYR2KDIRECTKERNEL_ALPHA_BETA)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -UUPPER -UTRANSA $< -o $@
$(KDIR)ssyrk_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYRKDIRECTKERNEL_ALPHA_BETA)
$(KDIR)ssyr2k_direct_alpha_betaLT$(TSUFFIX).$(SUFFIX) : $(KERNELDIR)/$(SSYR2KDIRECTKERNEL_ALPHA_BETA)
$(CC) $(CFLAGS) -c -UDOUBLE -UCOMPLEX -UUPPER -DTRANSA $< -o $@
endif
endif
+6 -3
View File
@@ -1,5 +1,5 @@
/***************************************************************************
Copyright (c) 2013, The OpenBLAS Project
Copyright (c) 2013-2026, The OpenBLAS Project
All rights reserved.
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the following conditions are
@@ -50,8 +50,11 @@ FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x, FLOAT *y, BLASLONG inc_y)
while(i < n)
{
dot += y[iy] * x[ix] ;
#if defined(DSDOT)
dot += (double)y[iy] * (double)x[ix] ;
#else
dot += y[iy] * x[ix];
#endif
ix += inc_x ;
iy += inc_y ;
i++ ;
+1 -1
View File
@@ -42,7 +42,7 @@ FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x)
n *= inc_x;
if (inc_x == 1)
{
#if V_SIMD && (!defined(DOUBLE) || (defined(DOUBLE) && V_SIMD_F64 && V_SIMD > 128))
#if V_SIMD && (!defined(DOUBLE) || (defined(DOUBLE) && V_SIMD_F64 && (V_SIMD > 128 || defined(ARCH_WASM))))
#ifdef DOUBLE
const int vstep = v_nlanes_f64;
const int unrollx4 = n & (-vstep * 4);
+5 -10
View File
@@ -80,6 +80,11 @@ DASUMKERNEL = dasum_thunderx2t99.c
CASUMKERNEL = casum_thunderx2t99.c
ZASUMKERNEL = zasum_thunderx2t99.c
SSUMKERNEL = ssum_thunderx2t99.c
DSUMKERNEL = dsum_thunderx2t99.c
CSUMKERNEL = csum_thunderx2t99.c
ZSUMKERNEL = zsum_thunderx2t99.c
SCOPYKERNEL = copy_thunderx2t99.c
DCOPYKERNEL = copy_thunderx2t99.c
CCOPYKERNEL = copy_thunderx2t99.c
@@ -102,18 +107,8 @@ ZNRM2KERNEL = znrm2.S
DDOTKERNEL = dot.c
SDOTKERNEL = dot.c
ifeq ($(OSNAME), WINNT)
ifeq ($(C_COMPILER), CLANG)
CDOTKERNEL = zdot.S
ZDOTKERNEL = zdot.S
else
CDOTKERNEL = zdot_thunderx2t99.c
ZDOTKERNEL = zdot_thunderx2t99.c
endif
else
CDOTKERNEL = zdot_thunderx2t99.c
ZDOTKERNEL = zdot_thunderx2t99.c
endif
DSDOTKERNEL = dot.S
DGEMM_BETA = dgemm_beta.S
+27 -4
View File
@@ -191,25 +191,48 @@ ZGEMMOTCOPYOBJ = zgemm_otcopy$(TSUFFIX).$(SUFFIX)
ifeq ($(BUILD_BFLOAT16), 1)
BGEMM_BETA = bgemm_beta_neon.c
BGEMMKERNEL = sbgemm_kernel_$(BGEMM_UNROLL_M)x$(BGEMM_UNROLL_N)_neoversen2.c
ifneq ($(BGEMM_UNROLL_M), $(BGEMM_UNROLL_N))
BGEMMINCOPY = sbgemm_ncopy_$(BGEMM_UNROLL_M)_neoversen2.c
BGEMMITCOPY = sbgemm_tcopy_$(BGEMM_UNROLL_M)_neoversen2.c
BGEMMONCOPY = sbgemm_ncopy_$(BGEMM_UNROLL_N)_neoversen2.c
BGEMMOTCOPY = sbgemm_tcopy_$(BGEMM_UNROLL_N)_neoversen2.c
BGEMMINCOPYOBJ = bgemm_incopy$(TSUFFIX).$(SUFFIX)
BGEMMITCOPYOBJ = bgemm_itcopy$(TSUFFIX).$(SUFFIX)
endif
BGEMMONCOPY = sbgemm_ncopy_$(BGEMM_UNROLL_N)_neoversen2.c
BGEMMOTCOPY = sbgemm_tcopy_$(BGEMM_UNROLL_N)_neoversen2.c
BGEMMONCOPYOBJ = bgemm_oncopy$(TSUFFIX).$(SUFFIX)
BGEMMOTCOPYOBJ = bgemm_otcopy$(TSUFFIX).$(SUFFIX)
BGEMVTKERNEL = sbgemv_t_bfdot.c
BGEMVNKERNEL = bgemv_n_sve_v3x4.c
ifeq ($(BUILD_HFLOAT16), 1)
SHGEMMKERNEL = shgemm_kernel_$(SHGEMM_UNROLL_M)x$(SHGEMM_UNROLL_N)_neoversen2.c
SHGEMMINCOPY = shgemm_ncopy_$(SHGEMM_UNROLL_M)_neoversen2.c
SHGEMMITCOPY = shgemm_tcopy_$(SHGEMM_UNROLL_M)_neoversen2.c
ifneq ($(SHGEMM_UNROLL_M), $(SHGEMM_UNROLL_N))
SHGEMMINCOPY = ../generic/gemm_ncopy_$(SHGEMM_UNROLL_M).c
SHGEMMITCOPY = ../generic/gemm_tcopy_$(SHGEMM_UNROLL_M).c
endif
SHGEMMONCOPY = shgemm_ncopy_$(SHGEMM_UNROLL_N)_neoversen2.c
SHGEMMOTCOPY = shgemm_tcopy_$(SHGEMM_UNROLL_N)_neoversen2.c
SHGEMMINCOPYOBJ = shgemm_incopy$(TSUFFIX).$(SUFFIX)
SHGEMMITCOPYOBJ = shgemm_itcopy$(TSUFFIX).$(SUFFIX)
SHGEMMONCOPYOBJ = shgemm_oncopy$(TSUFFIX).$(SUFFIX)
SHGEMMOTCOPYOBJ = shgemm_otcopy$(TSUFFIX).$(SUFFIX)
ifndef SHGEMM_BETA
SHGEMM_BETA = sbgemm_beta_neoversen2.c
endif
endif
SBGEMM_BETA = sbgemm_beta_neoversen2.c
SBGEMMKERNEL = sbgemm_kernel_$(SBGEMM_UNROLL_M)x$(SBGEMM_UNROLL_N)_neoversen2.c
ifneq ($(SBGEMM_UNROLL_M), $(SBGEMM_UNROLL_N))
SBGEMMINCOPY = sbgemm_ncopy_$(SBGEMM_UNROLL_M)_neoversen2.c
SBGEMMITCOPY = sbgemm_tcopy_$(SBGEMM_UNROLL_M)_neoversen2.c
SBGEMMONCOPY = sbgemm_ncopy_$(SBGEMM_UNROLL_N)_neoversen2.c
SBGEMMOTCOPY = sbgemm_tcopy_$(SBGEMM_UNROLL_N)_neoversen2.c
SBGEMMINCOPYOBJ = sbgemm_incopy$(TSUFFIX).$(SUFFIX)
SBGEMMITCOPYOBJ = sbgemm_itcopy$(TSUFFIX).$(SUFFIX)
endif
SBGEMMONCOPY = sbgemm_ncopy_$(SBGEMM_UNROLL_N)_neoversen2.c
SBGEMMOTCOPY = sbgemm_tcopy_$(SBGEMM_UNROLL_N)_neoversen2.c
SBGEMMONCOPYOBJ = sbgemm_oncopy$(TSUFFIX).$(SUFFIX)
SBGEMMOTCOPYOBJ = sbgemm_otcopy$(TSUFFIX).$(SUFFIX)
SBGEMVTKERNEL = sbgemv_t_bfdot.c
+1
View File
@@ -0,0 +1 @@
include $(KERNELDIR)/KERNEL.NEOVERSEN1
+4 -1
View File
@@ -262,7 +262,10 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
static RETURN_TYPE dot_kernel_asimd(BLASLONG n, FLOAT *x, BLASLONG inc_x, FLOAT *y, BLASLONG inc_y)
{
RETURN_TYPE dot = 0.0;
#ifndef DOUBLE
volatile
#endif
RETURN_TYPE dot = 0.0;
BLASLONG j = 0;
__asm__ __volatile__ (
+244
View File
@@ -0,0 +1,244 @@
/***************************************************************************
Copyright (c) 2017, The OpenBLAS Project
All rights reserved.
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the following conditions are
met:
1. Redistributions of source code must retain the above copyright
notice, this list of conditions and the following disclaimer.
2. Redistributions in binary form must reproduce the above copyright
notice, this list of conditions and the following disclaimer in
the documentation and/or other materials provided with the
distribution.
3. Neither the name of the OpenBLAS project nor the names of
its contributors may be used to endorse or promote products
derived from this software without specific prior written permission.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*****************************************************************************/
#include "common.h"
#include <arm_neon.h>
#define N "x0" /* vector length */
#define X "x1" /* "X" vector address */
#define INC_X "x2" /* "X" stride */
#define J "x5" /* loop variable */
#define REG0 "xzr"
#define SUMF "d0"
#define TMPF "d1"
/******************************************************************************/
#define KERNEL_F1 \
"ldr "TMPF", ["X"] \n" \
"add "X", "X", #8 \n" \
"fadd "SUMF", "SUMF", "TMPF" \n"
#define KERNEL_F32 \
"ldr q16, ["X"] \n" \
"ldr q17, ["X", #16] \n" \
"ldr q18, ["X", #32] \n" \
"ldr q19, ["X", #48] \n" \
"ldp q20, q21, ["X", #64] \n" \
"ldp q22, q23, ["X", #96] \n" \
"ldp q24, q25, ["X", #128] \n" \
"ldp q26, q27, ["X", #160] \n" \
"fadd v16.2d, v16.2d, v17.2d \n" \
"fadd v18.2d, v18.2d, v19.2d \n" \
"ldp q28, q29, ["X", #192] \n" \
"ldp q30, q31, ["X", #224] \n" \
"add "X", "X", #256 \n" \
"fadd v20.2d, v20.2d, v21.2d \n" \
"fadd v22.2d, v22.2d, v23.2d \n" \
"PRFM PLDL1KEEP, ["X", #1024] \n" \
"PRFM PLDL1KEEP, ["X", #1024+64] \n" \
"fadd v24.2d, v24.2d, v25.2d \n" \
"fadd v26.2d, v26.2d, v27.2d \n" \
"fadd v28.2d, v28.2d, v29.2d \n" \
"fadd v30.2d, v30.2d, v31.2d \n" \
"fadd v0.2d, v0.2d, v16.2d \n" \
"fadd v1.2d, v1.2d, v18.2d \n" \
"fadd v2.2d, v2.2d, v20.2d \n" \
"fadd v3.2d, v3.2d, v22.2d \n" \
"PRFM PLDL1KEEP, ["X", #1024+128] \n" \
"PRFM PLDL1KEEP, ["X", #1024+192] \n" \
"fadd v4.2d, v4.2d, v24.2d \n" \
"fadd v5.2d, v5.2d, v26.2d \n" \
"fadd v6.2d, v6.2d, v28.2d \n" \
"fadd v7.2d, v7.2d, v30.2d \n"
#define KERNEL_F32_FINALIZE \
"fadd v0.2d, v0.2d, v1.2d \n" \
"fadd v2.2d, v2.2d, v3.2d \n" \
"fadd v4.2d, v4.2d, v5.2d \n" \
"fadd v6.2d, v6.2d, v7.2d \n" \
"fadd v0.2d, v0.2d, v2.2d \n" \
"fadd v4.2d, v4.2d, v6.2d \n" \
"fadd v0.2d, v0.2d, v4.2d \n" \
"faddp "SUMF", v0.2d \n"
#define INIT_S \
"lsl "INC_X", "INC_X", #3 \n"
#define KERNEL_S1 \
"ldr "TMPF", ["X"] \n" \
"add "X", "X", "INC_X" \n" \
"fadd "SUMF", "SUMF", "TMPF" \n"
#if defined(SMP)
extern int blas_level1_thread_with_return_value(int mode, BLASLONG m, BLASLONG n,
BLASLONG k, void *alpha, void *a, BLASLONG lda, void *b, BLASLONG ldb,
void *c, BLASLONG ldc, int (*function)(), int nthreads);
#endif
static FLOAT dsum_compute(BLASLONG n, FLOAT *x, BLASLONG inc_x)
{
FLOAT dsum = 0.0 ;
if ( n < 0 ) return(dsum);
__asm__ __volatile__ (
" mov "N", %[N_] \n"
" mov "X", %[X_] \n"
" mov "INC_X", %[INCX_] \n"
" fmov "SUMF", "REG0" \n"
" fmov d1, "REG0" \n"
" fmov d2, "REG0" \n"
" fmov d3, "REG0" \n"
" fmov d4, "REG0" \n"
" fmov d5, "REG0" \n"
" fmov d6, "REG0" \n"
" fmov d7, "REG0" \n"
" cmp "N", xzr \n"
" ble 9f //dsum_kernel_L999 \n"
" cmp "INC_X", xzr \n"
" ble 9f //dsum_kernel_L999 \n"
" cmp "INC_X", #1 \n"
" bne 5f //dsum_kernel_S_BEGIN \n"
"1: //dsum_kernel_F_BEGIN: \n"
" asr "J", "N", #5 \n"
" cmp "J", xzr \n"
" beq 3f //dsum_kernel_F1 \n"
#if !(defined(__clang__) && defined(OS_WINDOWS))
".align 5 \n"
#endif
"2: //dsum_kernel_F32: \n"
" "KERNEL_F32" \n"
" subs "J", "J", #1 \n"
" bne 2b //dsum_kernel_F32 \n"
" "KERNEL_F32_FINALIZE" \n"
"3: //dsum_kernel_F1: \n"
" ands "J", "N", #31 \n"
" ble 9f //dsum_kernel_L999 \n"
"4: //dsum_kernel_F10: \n"
" "KERNEL_F1" \n"
" subs "J", "J", #1 \n"
" bne 4b //dsum_kernel_F10 \n"
" b 9f //dsum_kernel_L999 \n"
"5: //dsum_kernel_S_BEGIN: \n"
" "INIT_S" \n"
" asr "J", "N", #2 \n"
" cmp "J", xzr \n"
" ble 7f //dsum_kernel_S1 \n"
"6: //dsum_kernel_S4: \n"
" "KERNEL_S1" \n"
" "KERNEL_S1" \n"
" "KERNEL_S1" \n"
" "KERNEL_S1" \n"
" subs "J", "J", #1 \n"
" bne 6b //dsum_kernel_S4 \n"
"7: //dsum_kernel_S1: \n"
" ands "J", "N", #3 \n"
" ble 9f //dsum_kernel_L999 \n"
"8: //dsum_kernel_S10: \n"
" "KERNEL_S1" \n"
" subs "J", "J", #1 \n"
" bne 8b //dsum_kernel_S10 \n"
"9: //dsum_kernel_L999: \n"
" fmov %[DSUM_], "SUMF" \n"
: [DSUM_] "=r" (dsum) //%0
: [N_] "r" (n), //%1
[X_] "r" (x), //%2
[INCX_] "r" (inc_x) //%3
: "cc",
"memory",
"x0", "x1", "x2", "x3", "x4", "x5",
"d0", "d1", "d2", "d3", "d4", "d5", "d6", "d7"
);
return dsum;
}
#if defined(SMP)
static int dsum_thread_function(BLASLONG n, BLASLONG dummy0,
BLASLONG dummy1, FLOAT dummy2, FLOAT *x, BLASLONG inc_x, FLOAT *y,
BLASLONG inc_y, FLOAT *result, BLASLONG dummy3)
{
*result = dsum_compute(n, x, inc_x);
return 0;
}
#endif
FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x)
{
#if defined(SMP)
int nthreads;
FLOAT dummy_alpha;
#endif
FLOAT dsum = 0.0;
#if defined(SMP)
if (inc_x == 0 || n <= 10000)
nthreads = 1;
else
nthreads = num_cpu_avail(1);
if (nthreads == 1) {
dsum = dsum_compute(n, x, inc_x);
} else {
int mode, i;
char result[MAX_CPU_NUMBER * sizeof(double) * 2];
FLOAT *ptr;
mode = BLAS_DOUBLE;
blas_level1_thread_with_return_value(mode, n, 0, 0, &dummy_alpha,
x, inc_x, NULL, 0, result, 0,
( void *)dsum_thread_function, nthreads);
ptr = (FLOAT *)result;
for (i = 0; i < nthreads; i++) {
dsum = dsum + (*ptr);
ptr = (FLOAT *)(((char *)ptr) + sizeof(double) * 2);
}
}
#else
dsum = dsum_compute(n, x, inc_x);
#endif
return dsum;
}
+3
View File
@@ -155,7 +155,10 @@ static double nrm2_compute(BLASLONG n, FLOAT *x, BLASLONG inc_x)
" cmp "J", xzr \n"
" beq .Lnrm2_kernel_F1 \n"
/* https://github.com/llvm/llvm-project/issues/149547 */
#if !(defined(__clang__) && defined(OS_WINDOWS))
" .align 5 \n"
#endif
".Lnrm2_kernel_F: \n"
" "KERNEL_F" \n"
" subs "J", "J", #1 \n"
+10 -37
View File
@@ -35,16 +35,13 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#define I x3
#if !defined(DOUBLE)
#define SSQ s0
#define SCALE s1
#define REGZERO s5
#define REGONE s6
#else
#define SSQF s0
#endif
#define SSQ d0
#define SCALE d1
#define REGZERO d5
#define REGONE d6
#endif
/*******************************************************************************
* Macro definitions
@@ -53,22 +50,10 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
.macro KERNEL_F1
#if !defined(DOUBLE)
ldr s4, [X], #4
fcmp s4, REGZERO
beq 2f /* KERNEL_F1_NEXT_\@ */
fabs s4, s4
fcmp SCALE, s4
bge 1f /* KERNEL_F1_SCALE_GE_X_\@ */
fdiv s2, SCALE, s4
fmul s2, s2, s2
fmul s3, SSQ, s2
fadd SSQ, REGONE, s3
fmov SCALE, s4
b 2f /* KERNEL_F1_NEXT_\@ */
1: /* KERNEL_F1_SCALE_GE_X_\@: */
fdiv s2, s4, SCALE
fmla SSQ, s2, v2.s[0]
fcvt d4, s4
#else
ldr d4, [X], #8
#endif
fcmp d4, REGZERO
beq 2f /* KERNEL_F1_NEXT_\@ */
fabs d4, d4
@@ -83,29 +68,16 @@ USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
1: /* KERNEL_F1_SCALE_GE_X_\@: */
fdiv d2, d4, SCALE
fmla SSQ, d2, v2.d[0]
#endif
2: /* KERNEL_F1_NEXT_\@: */
.endm
.macro KERNEL_S1
#if !defined(DOUBLE)
ldr s4, [X]
fcmp s4, REGZERO
beq KERNEL_S1_NEXT
fabs s4, s4
fcmp SCALE, s4
bge KERNEL_S1_SCALE_GE_X
fdiv s2, SCALE, s4
fmul s2, s2, s2
fmul s3, SSQ, s2
fadd SSQ, REGONE, s3
fmov SCALE, s4
b KERNEL_S1_NEXT
KERNEL_S1_SCALE_GE_X:
fdiv s2, s4, SCALE
fmla SSQ, s2, v2.s[0]
fcvt d4, s4
#else
ldr d4, [X]
#endif
fcmp d4, REGZERO
beq KERNEL_S1_NEXT
fabs d4, d4
@@ -120,7 +92,6 @@ KERNEL_S1_SCALE_GE_X:
KERNEL_S1_SCALE_GE_X:
fdiv d2, d4, SCALE
fmla SSQ, d2, v2.d[0]
#endif
KERNEL_S1_NEXT:
add X, X, INC_X
.endm
@@ -218,7 +189,9 @@ KERNEL_S1_NEXT:
.Lnrm2_kernel_L999:
fsqrt SSQ, SSQ
fmul SSQ, SCALE, SSQ
#if !defined(DOUBLE)
fcvt SSQF, SSQ
#endif
ret
EPILOGUE
@@ -0,0 +1,56 @@
/***************************************************************************
* Copyright (c) 2026 The OpenBLAS Project
* All rights reserved.
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions are
* met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in
* the documentation and/or other materials provided with the
* distribution.
* 3. Neither the name of the OpenBLAS project nor the names of
* its contributors may be used to endorse or promote products
* derived from this software without specific prior written permission.
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
* POSSIBILITY OF SUCH DAMAGE.
* *****************************************************************************/
#include <arm_sve.h>
#include <arm_neon.h>
#include "common.h"
#define ALPHA_ONE
#include "sbgemm_kernel_8x8_neoversen2_impl.c"
#undef ALPHA_ONE
#undef UPDATE_C
#include "sbgemm_kernel_8x8_neoversen2_impl.c"
int CNAME(BLASLONG m, BLASLONG n, BLASLONG k, FLOAT alpha, IFLOAT *A, IFLOAT *B,
FLOAT *C, BLASLONG ldc) {
#ifdef BGEMM
bfloat16_t alpha_bf16;
memcpy(&alpha_bf16, &alpha, sizeof(bfloat16_t));
float alpha_f32 = vcvtah_f32_bf16(alpha_bf16);
#else
float alpha_f32 = alpha;
#endif
if (alpha_f32 == 1.0f)
return gemm_kernel_neoversen2_alpha_one(m, n, k, alpha, A, B, C, ldc);
else
return gemm_kernel_neoversen2_alpha(m, n, k, alpha, A, B, C, ldc);
return 0;
}
@@ -0,0 +1,763 @@
/***************************************************************************
* Copyright (c) 2022,2026 The OpenBLAS Project
* All rights reserved.
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions are
* met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in
* the documentation and/or other materials provided with the
* distribution.
* 3. Neither the name of the OpenBLAS project nor the names of
* its contributors may be used to endorse or promote products
* derived from this software without specific prior written permission.
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
* POSSIBILITY OF SUCH DAMAGE.
* *****************************************************************************/
#include <arm_sve.h>
#include <arm_neon.h>
#include "common.h"
#define INIT_C(M, N) mc##M##N = svdup_f32(0);
#define MATMUL(M, N) mc##M##N = svbfmmla(mc##M##N, ma##M, mb##N);
#define INIT_C_8x4 \
do { \
INIT_C(0, 0); \
INIT_C(0, 1); \
INIT_C(1, 0); \
INIT_C(1, 1); \
INIT_C(2, 0); \
INIT_C(2, 1); \
INIT_C(3, 0); \
INIT_C(3, 1); \
} while (0);
#define INIT_C_8x8 \
do { \
INIT_C(0, 0); \
INIT_C(0, 1); \
INIT_C(0, 2); \
INIT_C(0, 3); \
INIT_C(1, 0); \
INIT_C(1, 1); \
INIT_C(1, 2); \
INIT_C(1, 3); \
INIT_C(2, 0); \
INIT_C(2, 1); \
INIT_C(2, 2); \
INIT_C(2, 3); \
INIT_C(3, 0); \
INIT_C(3, 1); \
INIT_C(3, 2); \
INIT_C(3, 3); \
} while (0);
#ifdef BGEMM
#ifdef ALPHA_ONE
#define UPDATE_C(PG16, PG32, PTR, SRC) \
do { \
tmp16 = svld1_bf16((PG16), (PTR)); \
tmp32 = svreinterpret_f32(svzip1_bf16(zeros, tmp16)); \
tmp32 = svadd_z((PG32), SRC, tmp32); \
tmp16 = svcvt_bf16_f32_z((PG32), tmp32); \
tmp16 = svuzp1_bf16(tmp16, tmp16); \
svst1_bf16((PG16), (PTR), tmp16); \
} while (0)
#else
#define UPDATE_C(PG16, PG32, PTR, SRC) \
do { \
tmp16 = svld1_bf16((PG16), (PTR)); \
tmp32 = svreinterpret_f32(svzip1_bf16(zeros, tmp16)); \
tmp32 = svmad_z((PG32), svalpha, SRC, tmp32); \
tmp16 = svcvt_bf16_f32_z((PG32), tmp32); \
tmp16 = svuzp1_bf16(tmp16, tmp16); \
svst1_bf16((PG16), (PTR), tmp16); \
} while (0)
#endif
#else
#ifdef ALPHA_ONE
#define UPDATE_C(PG16, PG32, PTR, SRC) \
do { \
tmp32 = svld1_f32((PG32), (PTR)); \
tmp32 = svadd_z((PG32), SRC, tmp32); \
svst1_f32((PG32), (PTR), tmp32); \
} while (0);
#else
#define UPDATE_C(PG16, PG32, PTR, SRC) \
do { \
tmp32 = svld1_f32((PG32), (PTR)); \
tmp32 = svmad_z((PG32), svalpha, SRC, tmp32); \
svst1_f32((PG32), (PTR), tmp32); \
} while (0);
#endif
#endif
#ifdef BGEMM
#define OUTPUT_FLOAT bfloat16_t
#else
#define OUTPUT_FLOAT float
#endif
#ifdef ALPHA_ONE
static int gemm_kernel_neoversen2_alpha_one(BLASLONG m, BLASLONG n, BLASLONG k, FLOAT alpha, IFLOAT * A, IFLOAT * B, FLOAT * C, BLASLONG ldc)
#else
static int gemm_kernel_neoversen2_alpha(BLASLONG m, BLASLONG n, BLASLONG k, FLOAT alpha, IFLOAT * A, IFLOAT * B, FLOAT * C, BLASLONG ldc)
#endif
{
BLASLONG pad_k = (k + 3) & ~3;
svbfloat16_t ma0, ma1, ma2, ma3, mb0, mb1, mb2, mb3;
svfloat32_t mc00, mc01, mc02, mc03;
svfloat32_t mc10, mc11, mc12, mc13;
svfloat32_t mc20, mc21, mc22, mc23;
svfloat32_t mc30, mc31, mc32, mc33;
svfloat32_t vc0, vc1, vc2, vc3, vc4, vc5, vc6, vc7;
svfloat32_t vc8, vc9, vc10, vc11, vc12, vc13, vc14, vc15;
#ifndef ALPHA_ONE
#ifdef BGEMM
bfloat16_t alpha_bf16;
memcpy(&alpha_bf16, &alpha, sizeof(bfloat16_t));
svfloat32_t svalpha = svdup_f32(vcvtah_f32_bf16(alpha_bf16));
#else
svfloat32_t svalpha = svdup_f32(alpha);
#endif
#endif
svbool_t pg32_first_4 = svdupq_b32(1, 1, 1, 1);
svbool_t pg32_first_2 = svdupq_b32(1, 1, 0, 0);
svbool_t pg32_first_1 = svdupq_b32(1, 0, 0, 0);
svbool_t pg16_first_8 = svdupq_b16(1, 1, 1, 1, 1, 1, 1, 1);
svbool_t pg16_first_4 = svdupq_b16(1, 1, 1, 1, 0, 0, 0, 0);
#ifdef BGEMM
svbool_t pg16_first_2 = svdupq_b16(1, 1, 0, 0, 0, 0, 0, 0);
svbool_t pg16_first_1 = svdupq_b16(1, 0, 0, 0, 0, 0, 0, 0);
svbfloat16_t zeros = svdup_n_bf16(vcvth_bf16_f32(0.0));
#endif
bfloat16_t *ptr_a = (bfloat16_t *)A;
bfloat16_t *ptr_b = (bfloat16_t *)B;
OUTPUT_FLOAT *ptr_c = (OUTPUT_FLOAT*)C;
bfloat16_t *ptr_a0;
bfloat16_t *ptr_b0;
OUTPUT_FLOAT *ptr_c0, *ptr_c1, *ptr_c2, *ptr_c3;
OUTPUT_FLOAT *ptr_c4, *ptr_c5, *ptr_c6, *ptr_c7;
svfloat32_t tmp32;
#ifdef BGEMM
svbfloat16_t tmp16;
#endif
for (BLASLONG j = 0; j < n / 8; j++) {
ptr_c0 = ptr_c;
ptr_c1 = ptr_c0 + ldc;
ptr_c2 = ptr_c1 + ldc;
ptr_c3 = ptr_c2 + ldc;
ptr_c4 = ptr_c3 + ldc;
ptr_c5 = ptr_c4 + ldc;
ptr_c6 = ptr_c5 + ldc;
ptr_c7 = ptr_c6 + ldc;
ptr_c += 8 * ldc;
ptr_a = (bfloat16_t *)A;
for (BLASLONG i = 0; i < m / 8; i++) {
ptr_a0 = ptr_a;
ptr_a += 8 * pad_k;
ptr_b0 = ptr_b;
INIT_C_8x8;
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
ma2 = svld1_bf16(pg16_first_8, ptr_a0 + 16);
ma3 = svld1_bf16(pg16_first_8, ptr_a0 + 24);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
mb2 = svld1_bf16(pg16_first_8, ptr_b0 + 16);
mb3 = svld1_bf16(pg16_first_8, ptr_b0 + 24);
MATMUL(0, 0); MATMUL(0, 1); MATMUL(0, 2); MATMUL(0, 3);
MATMUL(1, 0); MATMUL(1, 1); MATMUL(1, 2); MATMUL(1, 3);
MATMUL(2, 0); MATMUL(2, 1); MATMUL(2, 2); MATMUL(2, 3);
MATMUL(3, 0); MATMUL(3, 1); MATMUL(3, 2); MATMUL(3, 3);
ptr_a0 += 32;
ptr_b0 += 32;
}
vc0 = svuzp1(mc00, mc10);
vc1 = svuzp1(mc20, mc30);
vc2 = svuzp2(mc00, mc10);
vc3 = svuzp2(mc20, mc30);
vc4 = svuzp1(mc01, mc11);
vc5 = svuzp1(mc21, mc31);
vc6 = svuzp2(mc01, mc11);
vc7 = svuzp2(mc21, mc31);
vc8 = svuzp1(mc02, mc12);
vc9 = svuzp1(mc22, mc32);
vc10 = svuzp2(mc02, mc12);
vc11 = svuzp2(mc22, mc32);
vc12 = svuzp1(mc03, mc13);
vc13 = svuzp1(mc23, mc33);
vc14 = svuzp2(mc03, mc13);
vc15 = svuzp2(mc23, mc33);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0 + 4, vc1);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc2);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1 + 4, vc3);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2, vc4);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2 + 4, vc5);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3, vc6);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3 + 4, vc7);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c4, vc8);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c4 + 4, vc9);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c5, vc10);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c5 + 4, vc11);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c6, vc12);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c6 + 4, vc13);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c7, vc14);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c7 + 4, vc15);
ptr_c0 += 8;
ptr_c1 += 8;
ptr_c2 += 8;
ptr_c3 += 8;
ptr_c4 += 8;
ptr_c5 += 8;
ptr_c6 += 8;
ptr_c7 += 8;
}
if (m & 4) {
ptr_a0 = ptr_a;
ptr_a += 4 * pad_k;
ptr_b0 = ptr_b;
INIT_C(0, 0); INIT_C(0, 1); INIT_C(0, 2); INIT_C(0, 3);
INIT_C(1, 0); INIT_C(1, 1); INIT_C(1, 2); INIT_C(1, 3);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
mb2 = svld1_bf16(pg16_first_8, ptr_b0 + 16);
mb3 = svld1_bf16(pg16_first_8, ptr_b0 + 24);
MATMUL(0, 0); MATMUL(0, 1); MATMUL(0, 2); MATMUL(0, 3);
MATMUL(1, 0); MATMUL(1, 1); MATMUL(1, 2); MATMUL(1, 3);
ptr_a0 += 16;
ptr_b0 += 32;
}
vc0 = svuzp1(mc00, mc10);
vc1 = svuzp2(mc00, mc10);
vc2 = svuzp1(mc01, mc11);
vc3 = svuzp2(mc01, mc11);
vc4 = svuzp1(mc02, mc12);
vc5 = svuzp2(mc02, mc12);
vc6 = svuzp1(mc03, mc13);
vc7 = svuzp2(mc03, mc13);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc1);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2, vc2);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3, vc3);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c4, vc4);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c5, vc5);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c6, vc6);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c7, vc7);
ptr_c0 += 4;
ptr_c1 += 4;
ptr_c2 += 4;
ptr_c3 += 4;
ptr_c4 += 4;
ptr_c5 += 4;
ptr_c6 += 4;
ptr_c7 += 4;
}
if (m & 2) {
ptr_a0 = ptr_a;
ptr_a += 2 * pad_k;
ptr_b0 = ptr_b;
INIT_C(0, 0); INIT_C(0, 1); INIT_C(0, 2); INIT_C(0, 3);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
mb2 = svld1_bf16(pg16_first_8, ptr_b0 + 16);
mb3 = svld1_bf16(pg16_first_8, ptr_b0 + 24);
MATMUL(0, 0); MATMUL(0, 1); MATMUL(0, 2); MATMUL(0, 3);
ptr_a0 += 8;
ptr_b0 += 32;
}
vc0 = svuzp1(mc00, mc00);
vc1 = svuzp2(mc00, mc00);
vc2 = svuzp1(mc01, mc01);
vc3 = svuzp2(mc01, mc01);
vc4 = svuzp1(mc02, mc02);
vc5 = svuzp2(mc02, mc02);
vc6 = svuzp1(mc03, mc03);
vc7 = svuzp2(mc03, mc03);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c0, vc0);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c1, vc1);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c2, vc2);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c3, vc3);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c4, vc4);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c5, vc5);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c6, vc6);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c7, vc7);
ptr_c0 += 2;
ptr_c1 += 2;
ptr_c2 += 2;
ptr_c3 += 2;
ptr_c4 += 2;
ptr_c5 += 2;
ptr_c6 += 2;
ptr_c7 += 2;
}
if (m & 1) {
ptr_a0 = ptr_a;
ptr_b0 = ptr_b;
INIT_C(0, 0); INIT_C(0, 1); INIT_C(0, 2); INIT_C(0, 3);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_4, ptr_a0);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
mb2 = svld1_bf16(pg16_first_8, ptr_b0 + 16);
mb3 = svld1_bf16(pg16_first_8, ptr_b0 + 24);
MATMUL(0, 0); MATMUL(0, 1); MATMUL(0, 2); MATMUL(0, 3);
ptr_a0 += 4;
ptr_b0 += 32;
}
vc1 = svuzp2(mc00, mc00);
vc3 = svuzp2(mc01, mc01);
vc5 = svuzp2(mc02, mc02);
vc7 = svuzp2(mc03, mc03);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c0, mc00);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c1, vc1);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c2, mc01);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c3, vc3);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c4, mc02);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c5, vc5);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c6, mc03);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c7, vc7);
}
ptr_b += 8 * pad_k;
}
if (n & 4) {
ptr_c0 = ptr_c;
ptr_c1 = ptr_c0 + ldc;
ptr_c2 = ptr_c1 + ldc;
ptr_c3 = ptr_c2 + ldc;
ptr_c += 4 * ldc;
ptr_a = (bfloat16_t *)A;
for (BLASLONG i = 0; i < m / 8; i++) {
ptr_a0 = ptr_a;
ptr_a += 8 * pad_k;
ptr_b0 = ptr_b;
INIT_C_8x4;
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
ma2 = svld1_bf16(pg16_first_8, ptr_a0 + 16);
ma3 = svld1_bf16(pg16_first_8, ptr_a0 + 24);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
MATMUL(0, 0); MATMUL(0, 1);
MATMUL(1, 0); MATMUL(1, 1);
MATMUL(2, 0); MATMUL(2, 1);
MATMUL(3, 0); MATMUL(3, 1);
ptr_a0 += 32;
ptr_b0 += 16;
}
vc0 = svuzp1(mc00, mc10);
vc1 = svuzp1(mc20, mc30);
vc2 = svuzp2(mc00, mc10);
vc3 = svuzp2(mc20, mc30);
vc4 = svuzp1(mc01, mc11);
vc5 = svuzp1(mc21, mc31);
vc6 = svuzp2(mc01, mc11);
vc7 = svuzp2(mc21, mc31);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0+4, vc1);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc2);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1+4, vc3);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2, vc4);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2+4, vc5);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3, vc6);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3+4, vc7);
ptr_c0 += 8;
ptr_c1 += 8;
ptr_c2 += 8;
ptr_c3 += 8;
}
if (m & 4) {
ptr_a0 = ptr_a;
ptr_a += 4 * pad_k;
ptr_b0 = ptr_b;
INIT_C(0, 0); INIT_C(0, 1);
INIT_C(1, 0); INIT_C(1, 1);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
MATMUL(0, 0); MATMUL(0, 1);
MATMUL(1, 0); MATMUL(1, 1);
ptr_a0 += 16;
ptr_b0 += 16;
}
vc0 = svuzp1(mc00, mc10);
vc1 = svuzp2(mc00, mc10);
vc2 = svuzp1(mc01, mc11);
vc3 = svuzp2(mc01, mc11);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc1);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c2, vc2);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c3, vc3);
ptr_c0 += 4;
ptr_c1 += 4;
ptr_c2 += 4;
ptr_c3 += 4;
}
if (m & 2) {
ptr_a0 = ptr_a;
ptr_a += 2 * pad_k;
ptr_b0 = ptr_b;
INIT_C(0, 0); INIT_C(0, 1);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
MATMUL(0, 0); MATMUL(0, 1);
ptr_a0 += 8;
ptr_b0 += 16;
}
vc0 = svuzp1(mc00, mc00);
vc1 = svuzp2(mc00, mc00);
vc2 = svuzp1(mc01, mc01);
vc3 = svuzp2(mc01, mc01);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c0, vc0);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c1, vc1);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c2, vc2);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c3, vc3);
ptr_c0 += 2;
ptr_c1 += 2;
ptr_c2 += 2;
ptr_c3 += 2;
}
if (m & 1) {
ptr_a0 = ptr_a;
ptr_b0 = ptr_b;
INIT_C(0, 0); INIT_C(0, 1);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_4, ptr_a0);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
mb1 = svld1_bf16(pg16_first_8, ptr_b0 + 8);
MATMUL(0, 0); MATMUL(0, 1);
ptr_a0 += 4;
ptr_b0 += 16;
}
vc1 = svuzp2(mc00, mc00);
vc3 = svuzp2(mc01, mc01);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c0, mc00);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c1, vc1);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c2, mc01);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c3, vc3);
}
ptr_b += 4 * pad_k;
}
if (n & 2) {
ptr_c0 = ptr_c;
ptr_c1 = ptr_c0 + ldc;
ptr_c += 2 * ldc;
ptr_a = (bfloat16_t *)A;
for (BLASLONG i = 0; i < m / 8; i++) {
ptr_a0 = ptr_a;
ptr_a += 8 * pad_k;
ptr_b0 = ptr_b;
INIT_C(0, 0);
INIT_C(1, 0);
INIT_C(2, 0);
INIT_C(3, 0);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
ma2 = svld1_bf16(pg16_first_8, ptr_a0 + 16);
ma3 = svld1_bf16(pg16_first_8, ptr_a0 + 24);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
MATMUL(0, 0);
MATMUL(1, 0);
MATMUL(2, 0);
MATMUL(3, 0);
ptr_a0 += 32;
ptr_b0 += 8;
}
vc0 = svuzp1(mc00, mc10);
vc1 = svuzp1(mc20, mc30);
vc2 = svuzp2(mc00, mc10);
vc3 = svuzp2(mc20, mc30);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0 + 4, vc1);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc2);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1 + 4, vc3);
ptr_c0 += 8;
ptr_c1 += 8;
}
if (m & 4) {
ptr_a0 = ptr_a;
ptr_a += 4 * pad_k;
ptr_b0 = ptr_b;
INIT_C(0, 0);
INIT_C(1, 0);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
MATMUL(0, 0);
MATMUL(1, 0);
ptr_a0 += 16;
ptr_b0 += 8;
}
vc0 = svuzp1(mc00, mc10);
vc1 = svuzp2(mc00, mc10);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c1, vc1);
ptr_c0 += 4;
ptr_c1 += 4;
}
if (m & 2) {
ptr_a0 = ptr_a;
ptr_a += 2 * pad_k;
ptr_b0 = ptr_b;
INIT_C(0, 0);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
MATMUL(0, 0);
ptr_a0 += 8;
ptr_b0 += 8;
}
vc0 = svuzp1(mc00, mc00);
vc1 = svuzp2(mc00, mc00);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c0, vc0);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c1, vc1);
ptr_c0 += 2;
ptr_c1 += 2;
}
if (m & 1) {
ptr_a0 = ptr_a;
ptr_b0 = ptr_b;
INIT_C(0, 0);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_4, ptr_a0);
mb0 = svld1_bf16(pg16_first_8, ptr_b0);
MATMUL(0, 0);
ptr_a0 += 4;
ptr_b0 += 8;
}
vc1 = svuzp2(mc00, mc00);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c0, mc00);
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c1, vc1);
}
ptr_b += 2 * pad_k;
}
if (n & 1) {
ptr_c0 = ptr_c;
ptr_a = (bfloat16_t *)A;
for (BLASLONG i = 0; i < m / 8; i++) {
ptr_a0 = ptr_a;
ptr_a += 8 * pad_k;
ptr_b0 = ptr_b;
INIT_C(0, 0);
INIT_C(1, 0);
INIT_C(2, 0);
INIT_C(3, 0);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
ma2 = svld1_bf16(pg16_first_8, ptr_a0 + 16);
ma3 = svld1_bf16(pg16_first_8, ptr_a0 + 24);
mb0 = svld1_bf16(pg16_first_4, ptr_b0);
MATMUL(0, 0);
MATMUL(1, 0);
MATMUL(2, 0);
MATMUL(3, 0);
ptr_a0 += 32;
ptr_b0 += 4;
}
vc0 = svuzp1(mc00, mc10);
vc1 = svuzp1(mc20, mc30);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0 + 4, vc1);
ptr_c0 += 8;
}
if (m & 4) {
ptr_a0 = ptr_a;
ptr_a += 4 * pad_k;
ptr_b0 = ptr_b;
INIT_C(0, 0);
INIT_C(1, 0);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
ma1 = svld1_bf16(pg16_first_8, ptr_a0 + 8);
mb0 = svld1_bf16(pg16_first_4, ptr_b0);
MATMUL(0, 0);
MATMUL(1, 0);
ptr_a0 += 16;
ptr_b0 += 4;
}
vc0 = svuzp1(mc00, mc10);
UPDATE_C(pg16_first_4, pg32_first_4, ptr_c0, vc0);
ptr_c0 += 4;
}
if (m & 2) {
ptr_a0 = ptr_a;
ptr_a += 2 * pad_k;
ptr_b0 = ptr_b;
INIT_C(0, 0);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_8, ptr_a0);
mb0 = svld1_bf16(pg16_first_4, ptr_b0);
MATMUL(0, 0);
ptr_a0 += 8;
ptr_b0 += 4;
}
vc0 = svuzp1(mc00, mc00);
UPDATE_C(pg16_first_2, pg32_first_2, ptr_c0, vc0);
ptr_c0 += 2;
}
if (m & 1) {
ptr_a0 = ptr_a;
ptr_b0 = ptr_b;
INIT_C(0, 0);
for (BLASLONG p = 0; p < pad_k; p += 4) {
ma0 = svld1_bf16(pg16_first_4, ptr_a0);
mb0 = svld1_bf16(pg16_first_4, ptr_b0);
MATMUL(0, 0);
ptr_a0 += 4;
ptr_b0 += 4;
}
UPDATE_C(pg16_first_1, pg32_first_1, ptr_c0, mc00);
}
}
return 0;
}
+3
View File
@@ -238,7 +238,10 @@ static double nrm2_compute(BLASLONG n, FLOAT *x, BLASLONG inc_x)
" cmp "J", xzr \n"
" beq 5f //nrm2_kernel_S_BEGIN \n"
/* https://github.com/llvm/llvm-project/issues/149547 */
#if !(defined(__clang__) && defined(OS_WINDOWS))
" .align 5 \n"
#endif
"2: //nrm2_kernel_F: \n"
" "KERNEL_F" \n"
" subs "J", "J", #1 \n"
@@ -7,17 +7,30 @@
#include <stdlib.h>
#include <inttypes.h>
#include <math.h>
#include "sme_abi.h"
#if defined(DYNAMIC_ARCH)
#define COMBINE(a,b) a ## b
#define COMBINE2(a,b) COMBINE(a,b)
#define SME1_PREPROCESS_BASE sgemm_direct_sme1_preprocess
#define SME1_PREPROCESS COMBINE2(SME1_PREPROCESS_BASE,TS)
#define SME1_KERNEL2X2_BASE sgemm_direct_alpha_beta_sme1_2VLx2VL
#define SME1_KERNEL2X2 COMBINE2(SME1_KERNEL2X2_BASE,TS)
#else
#define SME1_PREPROCESS sgemm_direct_sme1_preprocess
#define SME1_KERNEL2X2 sgemm_direct_alpha_beta_sme1_2VLx2VL
#endif
/* Function prototypes */
extern void SME1_PREPROCESS(uint64_t nbr, uint64_t nbc,\
const float * restrict a, float * a_mod);
#if defined(HAVE_SME)
#include "sme_abi.h"
#if defined(__ARM_FEATURE_SME) && defined(__clang__) && __clang_major__ >= 16
#include <arm_sme.h>
#endif
/* Function prototypes */
extern void sgemm_direct_sme1_preprocess(uint64_t nbr, uint64_t nbc,\
const float * restrict a, float * a_mod) __asm__("sgemm_direct_sme1_preprocess");
/* Function Definitions */
static uint64_t sve_cntw() {
uint64_t cnt;
@@ -99,10 +112,11 @@ kernel_2x2(const float *A, const float *B, float *C, size_t shared_dim,
svst1_hor_za32(/*tile*/2, /*slice*/i, pg_c_0, &C[i * ldc]);
svst1_hor_za32(/*tile*/3, /*slice*/i, pg_c_1, &C[i * ldc + svl]);
}
return;
}
__arm_new("za") __arm_locally_streaming
void sgemm_direct_alpha_beta_sme1_2VLx2VL(uint64_t m, uint64_t k, uint64_t n, const float* alpha,\
void SME1_KERNEL2X2(uint64_t m, uint64_t k, uint64_t n, const float* alpha,\
const float *ba, const float *restrict bb, const float* beta,\
float *restrict C) {
@@ -125,6 +139,7 @@ void sgemm_direct_alpha_beta_sme1_2VLx2VL(uint64_t m, uint64_t k, uint64_t n, co
// Block over row dimension of C
for (; row_idx < num_rows; row_idx += row_batch) {
row_batch = MIN(row_batch, num_rows - row_idx);
uint64_t col_idx = 0;
uint64_t col_batch = 2*svl;
@@ -141,9 +156,9 @@ void sgemm_direct_alpha_beta_sme1_2VLx2VL(uint64_t m, uint64_t k, uint64_t n, co
}
#else
void sgemm_direct_alpha_beta_sme1_2VLx2VL(uint64_t m, uint64_t k, uint64_t n, const float* alpha,\
void SME1_KERNEL2X2(uint64_t m, uint64_t k, uint64_t n, const float* alpha,\
const float *ba, const float *restrict bb, const float* beta,\
float *restrict C){}
float *restrict C){fprintf(stderr,"empty sgemm_alpha_beta2x2 should never get called!!!\n");}
#endif
/*void sgemm_kernel_direct (BLASLONG M, BLASLONG N, BLASLONG K,\
@@ -166,7 +181,7 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float alpha, float * __restrict
* of reading directly from vector (z) registers.
* */
asm volatile("" : : :"p0", "p1", "p2", "p3", "p4", "p5", "p6", "p7",
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15",
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15", "d8", "d9", "d10", "d11", "d12", "d13", "d14", "d15",
"z0", "z1", "z2", "z3", "z4", "z5", "z6", "z7",
"z8", "z9", "z10", "z11", "z12", "z13", "z14", "z15",
"z16", "z17", "z18", "z19", "z20", "z21", "z22", "z23",
@@ -175,17 +190,19 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float alpha, float * __restrict
/* Pre-process the left matrix to make it suitable for
matrix sum of outer-product calculation
*/
sgemm_direct_sme1_preprocess(M, K, A, A_mod);
SME1_PREPROCESS(M, K, A, A_mod);
asm volatile("" : : :"p0", "p1", "p2", "p3", "p4", "p5", "p6", "p7",
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15",
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15","d8", "d9", "d10", "d11", "d12", "d13", "d14", "d15",
"z0", "z1", "z2", "z3", "z4", "z5", "z6", "z7",
"z8", "z9", "z10", "z11", "z12", "z13", "z14", "z15",
"z16", "z17", "z18", "z19", "z20", "z21", "z22", "z23",
"z24", "z25", "z26", "z27", "z28", "z29", "z30", "z31");
/* Calculate C = alpha*A*B + beta*C */
sgemm_direct_alpha_beta_sme1_2VLx2VL(M, K, N, &alpha, A_mod, B, &beta, R);
SME1_KERNEL2X2(M, K, N, &alpha, A_mod, B, &beta, R);
free(A_mod);
}
@@ -194,6 +211,7 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float alpha, float * __restrict
void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float alpha, float * __restrict A,\
BLASLONG strideA, float * __restrict B, BLASLONG strideB ,\
float beta, float * __restrict R, BLASLONG strideR){}
float beta, float * __restrict R, BLASLONG strideR){fprintf(stderr,"empty sgemm_direct_alpha_beta should not be called!!!\n");}
#endif
+33 -13
View File
@@ -7,18 +7,29 @@
#include <stdlib.h>
#include <inttypes.h>
#include <math.h>
#if defined(DYNAMIC_ARCH)
#define COMBINE(a,b) a ## b
#define COMBINE2(a,b) COMBINE(a,b)
#define SME1_PREPROCESS_BASE sgemm_direct_sme1_preprocess
#define SME1_PREPROCESS COMBINE2(SME1_PREPROCESS_BASE,TS)
#define SME1_DIRECT2X2_BASE sgemm_direct_sme1_2VLx2VL
#define SME1_DIRECT2X2 COMBINE2(SME1_DIRECT2X2_BASE,TS)
#else
#define SME1_PREPROCESS sgemm_direct_sme1_preprocess
#define SME1_DIRECT2X2 sgemm_direct_sme1_2VLx2VL
#endif
#if defined(HAVE_SME)
/* Function prototypes */
extern void sgemm_direct_sme1_preprocess(uint64_t nbr, uint64_t nbc,\
const float * restrict a, float * a_mod) __asm__("sgemm_direct_sme1_preprocess");
extern void sgemm_direct_sme1_2VLx2VL(uint64_t m, uint64_t k, uint64_t n,\
extern void SME1_PREPROCESS(uint64_t nbr, uint64_t nbc,\
const float * restrict a, float * a_mod) ;
extern void SME1_DIRECT2X2(uint64_t m, uint64_t k, uint64_t n,\
const float * matLeft,\
const float * restrict matRight,\
const float * restrict matResult) __asm__("sgemm_direct_sme1_2VLx2VL");
const float * restrict matResult) ;
/* Function Definitions */
uint64_t sve_cntw() {
static uint64_t sve_cntw() {
uint64_t cnt;
asm volatile(
"rdsvl %[res], #1\n"
@@ -39,7 +50,6 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float * __restrict A,\
uint64_t m_mod, vl_elms;
vl_elms = sve_cntw();
m_mod = ceil((double)M/(double)vl_elms) * vl_elms;
float *A_mod = (float *) malloc(m_mod*K*sizeof(float));
@@ -48,7 +58,7 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float * __restrict A,\
* of reading directly from vector (z) registers.
* */
asm volatile("" : : :"p0", "p1", "p2", "p3", "p4", "p5", "p6", "p7",
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15",
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15", "d8", "d9", "d10", "d11", "d12", "d13", "d14", "d15",
"z0", "z1", "z2", "z3", "z4", "z5", "z6", "z7",
"z8", "z9", "z10", "z11", "z12", "z13", "z14", "z15",
"z16", "z17", "z18", "z19", "z20", "z21", "z22", "z23",
@@ -57,13 +67,13 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float * __restrict A,\
/* Pre-process the left matrix to make it suitable for
matrix sum of outer-product calculation
*/
sgemm_direct_sme1_preprocess(M, K, A, A_mod);
SME1_PREPROCESS(M, K, A, A_mod);
/* Calculate C = A*B */
sgemm_direct_sme1_2VLx2VL(M, K, N, A_mod, B, R);
SME1_DIRECT2X2(M, K, N, A_mod, B, R);
asm volatile("" : : :"p0", "p1", "p2", "p3", "p4", "p5", "p6", "p7",
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15",
"p8", "p9", "p10", "p11", "p12", "p13", "p14", "p15", "d8", "d9", "d10", "d11", "d12", "d13", "d14", "d15",
"z0", "z1", "z2", "z3", "z4", "z5", "z6", "z7",
"z8", "z9", "z10", "z11", "z12", "z13", "z14", "z15",
"z16", "z17", "z18", "z19", "z20", "z21", "z22", "z23",
@@ -75,6 +85,16 @@ void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float * __restrict A,\
void CNAME (BLASLONG M, BLASLONG N, BLASLONG K, float * __restrict A,\
BLASLONG strideA, float * __restrict B, BLASLONG strideB ,\
float * __restrict R, BLASLONG strideR){}
float * __restrict R, BLASLONG strideR){
fprintf(stderr,"EMPTY sgemm_kernel_direct should never be called \n");
}
void SME1_DIRECT2X2( uint64_t M , uint64_t K, uint64_t N,\
const float * restrict A_base,\
const float * restrict B_base,\
const float * restrict C_base){};
void SME1_PREPROCESS(uint64_t nbr, uint64_t nbc,\
const float * restrict a, float * a_mod){};
#endif
+15
View File
@@ -0,0 +1,15 @@
#include "common.h"
/* helper for the direct sgemm code adapted from Arjan van der Ven's x86_64 version */
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K)
{
if (M<3) return 0;
unsigned long long mnk = M * N * K;
/* benchmark performance on M4 peaks around 512 and crosses the graph of the NEON SGEMM at about 3100 */
if (mnk >= 3100L * 3100L * 3100L)
return 0;
return 1;
}
@@ -35,16 +35,17 @@
#define K_exit x15 //Exit condition for K loop
#define M_cntr x16 //M loop counter
#define C1 x17 //Constant1: N*(SVLs+1);SVLs-No. of 32-bit elements
#define C2 x18 //Constant2: N + SVLs
#define C3 x19 //Constant3: K*SVLs + SVLs
#define C4 x20 //Constant4: SVLs-2
#define C5 x21 //Constant5: K*SVLs
#define C6 x22 //Constant6: N*SVLs
#define C2 x19 //Constant2: N + SVLs
#define C3 x20 //Constant3: K*SVLs + SVLs
#define C4 x21 //Constant4: SVLs-2
#define C5 x22 //Constant5: K*SVLs
#define C6 x23 //Constant6: N*SVLs
.text
.global sgemm_direct_sme1_2VLx2VL
.global ASMNAME
sgemm_direct_sme1_2VLx2VL:
ASMNAME:
//sgemm_direct_sme1_2VLx2VL:
stp x19, x20, [sp, #-48]!
stp x21, x22, [sp, #16]
@@ -61,7 +62,7 @@
add C2, N, C4 //N + SVLs
add C3, C5, C4 //K*SVLs + SVLs
whilelt p2.s, M_cntr, M //Tile 0,1 predicate (M dimension)
sub w20, w20, #2 //SVLs-2
sub w21, w21, #2 //SVLs-2
.M_Loop:
incw M_cntr
@@ -198,7 +199,7 @@ process_K_less_than_equal_2:
st1w {za1h.s[w13, #0]}, p5, [Cptr1]
st1w {za2h.s[w13, #0]}, p6, [Cptr0, C6, lsl #2]
st1w {za3h.s[w13, #0]}, p7, [Cptr1, C6, lsl #2]
cmp w13, w20
cmp w13, w21
b.mi .Loop_store_ZA
psel p4, p0, p2.s[w13, 1]
psel p5, p1, p2.s[w13, 1]
@@ -211,12 +212,12 @@ process_K_less_than_equal_2:
addvl Cptr, Cptr, #2
addvl Bptr, Bptr, #1
whilelt p0.b, Bptr, N_exit //1st Tile predicate (N dimension)
b.first .N_Loop
b.mi .N_Loop
add A_base, A_base, C5, lsl #3 //A_base += 2*K*SVLs FP32 elements
add C_base, C_base, C6, lsl #3 //C_base += 2*N*SVLs FP32 elements
incw M_cntr
whilelt p2.s, M_cntr, M //1st Tile predicate (M dimension)
b.first .M_Loop
b.mi .M_Loop
smstop
+4 -4
View File
@@ -37,9 +37,9 @@
#define C6 x15 //Constant6: 3*ncol
.text
.global sgemm_direct_sme1_preprocess
.global ASMNAME //sgemm_direct_sme1_preprocess
sgemm_direct_sme1_preprocess:
ASMNAME: //sgemm_direct_sme1_preprocess:
stp x19, x20, [sp, #-48]!
stp x21, x22, [sp, #16]
@@ -114,14 +114,14 @@
addvl mat_ptr0, mat_ptr0, #1 //mat_ptr0 += SVLb
whilelt p8.b, mat_ptr0, inner_loop_exit
b.first .Loop_process
b.mi .Loop_process
add mat_mod, mat_mod, C3, lsl #2 //mat_mod+=SVLs*nbc FP32 elements
add mat, mat, C3, lsl #2 //mat+=SVLs*nbc FP32 elements
incw outer_loop_cntr
whilelt p0.s, outer_loop_cntr, nrow
b.first .M_Loop
b.mi .M_Loop
smstop
+887
View File
@@ -0,0 +1,887 @@
/***************************************************************************
* Copyright (c) 2026 The OpenBLAS Project
* All rights reserved.
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions are
* met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in
* the documentation and/or other materials provided with the
* distribution.
* 3. Neither the name of the OpenBLAS project nor the names of
* its contributors may be used to endorse or promote products
* derived from this software without specific prior written permission.
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
* POSSIBILITY OF SUCH DAMAGE.
* *****************************************************************************/
#include <arm_neon.h>
#include "common.h"
static inline void kernel_8x8(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
float32x4_t c0_low = vdupq_n_f32(0.0f);
float32x4_t c0_high = vdupq_n_f32(0.0f);
float32x4_t c1_low = vdupq_n_f32(0.0f);
float32x4_t c1_high = vdupq_n_f32(0.0f);
float32x4_t c2_low = vdupq_n_f32(0.0f);
float32x4_t c2_high = vdupq_n_f32(0.0f);
float32x4_t c3_low = vdupq_n_f32(0.0f);
float32x4_t c3_high = vdupq_n_f32(0.0f);
float32x4_t c4_low = vdupq_n_f32(0.0f);
float32x4_t c4_high = vdupq_n_f32(0.0f);
float32x4_t c5_low = vdupq_n_f32(0.0f);
float32x4_t c5_high = vdupq_n_f32(0.0f);
float32x4_t c6_low = vdupq_n_f32(0.0f);
float32x4_t c6_high = vdupq_n_f32(0.0f);
float32x4_t c7_low = vdupq_n_f32(0.0f);
float32x4_t c7_high = vdupq_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float16x8_t a_f16 = vld1q_f16(A);
float32x4_t a_low = vcvt_f32_f16(vget_low_f16(a_f16));
float32x4_t a_high = vcvt_f32_f16(vget_high_f16(a_f16));
float16x8_t b_f16 = vld1q_f16(B);
float32x4_t b_low = vcvt_f32_f16(vget_low_f16(b_f16));
float32x4_t b_high = vcvt_f32_f16(vget_high_f16(b_f16));
float32_t b0_lane0 = vgetq_lane_f32(b_low, 0);
c0_low = vfmaq_n_f32(c0_low, a_low, b0_lane0);
c0_high = vfmaq_n_f32(c0_high, a_high, b0_lane0);
float32_t b0_lane1 = vgetq_lane_f32(b_low, 1);
c1_low = vfmaq_n_f32(c1_low, a_low, b0_lane1);
c1_high = vfmaq_n_f32(c1_high, a_high, b0_lane1);
float32_t b0_lane2 = vgetq_lane_f32(b_low, 2);
c2_low = vfmaq_n_f32(c2_low, a_low, b0_lane2);
c2_high = vfmaq_n_f32(c2_high, a_high, b0_lane2);
float32_t b0_lane3 = vgetq_lane_f32(b_low, 3);
c3_low = vfmaq_n_f32(c3_low, a_low, b0_lane3);
c3_high = vfmaq_n_f32(c3_high, a_high, b0_lane3);
float32_t b1_lane0 = vgetq_lane_f32(b_high, 0);
c4_low = vfmaq_n_f32(c4_low, a_low, b1_lane0);
c4_high = vfmaq_n_f32(c4_high, a_high, b1_lane0);
float32_t b1_lane1 = vgetq_lane_f32(b_high, 1);
c5_low = vfmaq_n_f32(c5_low, a_low, b1_lane1);
c5_high = vfmaq_n_f32(c5_high, a_high, b1_lane1);
float32_t b1_lane2 = vgetq_lane_f32(b_high, 2);
c6_low = vfmaq_n_f32(c6_low, a_low, b1_lane2);
c6_high = vfmaq_n_f32(c6_high, a_high, b1_lane2);
float32_t b1_lane3 = vgetq_lane_f32(b_high, 3);
c7_low = vfmaq_n_f32(c7_low, a_low, b1_lane3);
c7_high = vfmaq_n_f32(c7_high, a_high, b1_lane3);
A += 8;
B += 8;
}
FLOAT *col_0 = C + 0 * ldc;
FLOAT *col_1 = C + 1 * ldc;
FLOAT *col_2 = C + 2 * ldc;
FLOAT *col_3 = C + 3 * ldc;
FLOAT *col_4 = C + 4 * ldc;
FLOAT *col_5 = C + 5 * ldc;
FLOAT *col_6 = C + 6 * ldc;
FLOAT *col_7 = C + 7 * ldc;
float32x4_t t0_l = vld1q_f32(col_0);
float32x4_t t0_h = vld1q_f32(col_0 + 4);
t0_l = vaddq_f32(t0_l, vmulq_n_f32(c0_low, alpha));
t0_h = vaddq_f32(t0_h, vmulq_n_f32(c0_high, alpha));
vst1q_f32(col_0, t0_l);
vst1q_f32(col_0 + 4, t0_h);
float32x4_t t1_l = vld1q_f32(col_1);
float32x4_t t1_h = vld1q_f32(col_1 + 4);
t1_l = vaddq_f32(t1_l, vmulq_n_f32(c1_low, alpha));
t1_h = vaddq_f32(t1_h, vmulq_n_f32(c1_high, alpha));
vst1q_f32(col_1, t1_l);
vst1q_f32(col_1 + 4, t1_h);
float32x4_t t2_l = vld1q_f32(col_2);
float32x4_t t2_h = vld1q_f32(col_2 + 4);
t2_l = vaddq_f32(t2_l, vmulq_n_f32(c2_low, alpha));
t2_h = vaddq_f32(t2_h, vmulq_n_f32(c2_high, alpha));
vst1q_f32(col_2, t2_l);
vst1q_f32(col_2 + 4, t2_h);
float32x4_t t3_l = vld1q_f32(col_3);
float32x4_t t3_h = vld1q_f32(col_3 + 4);
t3_l = vaddq_f32(t3_l, vmulq_n_f32(c3_low, alpha));
t3_h = vaddq_f32(t3_h, vmulq_n_f32(c3_high, alpha));
vst1q_f32(col_3, t3_l);
vst1q_f32(col_3 + 4, t3_h);
float32x4_t t4_l = vld1q_f32(col_4);
float32x4_t t4_h = vld1q_f32(col_4 + 4);
t4_l = vaddq_f32(t4_l, vmulq_n_f32(c4_low, alpha));
t4_h = vaddq_f32(t4_h, vmulq_n_f32(c4_high, alpha));
vst1q_f32(col_4, t4_l);
vst1q_f32(col_4 + 4, t4_h);
float32x4_t t5_l = vld1q_f32(col_5);
float32x4_t t5_h = vld1q_f32(col_5 + 4);
t5_l = vaddq_f32(t5_l, vmulq_n_f32(c5_low, alpha));
t5_h = vaddq_f32(t5_h, vmulq_n_f32(c5_high, alpha));
vst1q_f32(col_5, t5_l);
vst1q_f32(col_5 + 4, t5_h);
float32x4_t t6_l = vld1q_f32(col_6);
float32x4_t t6_h = vld1q_f32(col_6 + 4);
t6_l = vaddq_f32(t6_l, vmulq_n_f32(c6_low, alpha));
t6_h = vaddq_f32(t6_h, vmulq_n_f32(c6_high, alpha));
vst1q_f32(col_6, t6_l);
vst1q_f32(col_6 + 4, t6_h);
float32x4_t t7_l = vld1q_f32(col_7);
float32x4_t t7_h = vld1q_f32(col_7 + 4);
t7_l = vaddq_f32(t7_l, vmulq_n_f32(c7_low, alpha));
t7_h = vaddq_f32(t7_h, vmulq_n_f32(c7_high, alpha));
vst1q_f32(col_7, t7_l);
vst1q_f32(col_7 + 4, t7_h);
}
static inline void kernel_4x8(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
float32x4_t c0 = vdupq_n_f32(0.0f);
float32x4_t c1 = vdupq_n_f32(0.0f);
float32x4_t c2 = vdupq_n_f32(0.0f);
float32x4_t c3 = vdupq_n_f32(0.0f);
float32x4_t c4 = vdupq_n_f32(0.0f);
float32x4_t c5 = vdupq_n_f32(0.0f);
float32x4_t c6 = vdupq_n_f32(0.0f);
float32x4_t c7 = vdupq_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float32x4_t a_f16 = vcvt_f32_f16(vld1_f16(A));
float16x8_t b_f16 = vld1q_f16(B);
float32x4_t b_low = vcvt_f32_f16(vget_low_f16(b_f16));
float32x4_t b_high = vcvt_f32_f16(vget_high_f16(b_f16));
float32_t b0_lane0 = vgetq_lane_f32(b_low, 0);
c0 = vfmaq_n_f32(c0, a_f16, b0_lane0);
float32_t b0_lane1 = vgetq_lane_f32(b_low, 1);
c1 = vfmaq_n_f32(c1, a_f16, b0_lane1);
float32_t b0_lane2 = vgetq_lane_f32(b_low, 2);
c2 = vfmaq_n_f32(c2, a_f16, b0_lane2);
float32_t b0_lane3 = vgetq_lane_f32(b_low, 3);
c3 = vfmaq_n_f32(c3, a_f16, b0_lane3);
float32_t b1_lane0 = vgetq_lane_f32(b_high, 0);
c4 = vfmaq_n_f32(c4, a_f16, b1_lane0);
float32_t b1_lane1 = vgetq_lane_f32(b_high, 1);
c5 = vfmaq_n_f32(c5, a_f16, b1_lane1);
float32_t b1_lane2 = vgetq_lane_f32(b_high, 2);
c6 = vfmaq_n_f32(c6, a_f16, b1_lane2);
float32_t b1_lane3 = vgetq_lane_f32(b_high, 3);
c7 = vfmaq_n_f32(c7, a_f16, b1_lane3);
A += 4;
B += 8;
}
FLOAT *col_0 = C + 0 * ldc;
FLOAT *col_1 = C + 1 * ldc;
FLOAT *col_2 = C + 2 * ldc;
FLOAT *col_3 = C + 3 * ldc;
FLOAT *col_4 = C + 4 * ldc;
FLOAT *col_5 = C + 5 * ldc;
FLOAT *col_6 = C + 6 * ldc;
FLOAT *col_7 = C + 7 * ldc;
float32x4_t t0 = vld1q_f32(col_0);
t0 = vaddq_f32(t0, vmulq_n_f32(c0, alpha));
vst1q_f32(col_0, t0);
float32x4_t t1 = vld1q_f32(col_1);
t1 = vaddq_f32(t1, vmulq_n_f32(c1, alpha));
vst1q_f32(col_1, t1);
float32x4_t t2 = vld1q_f32(col_2);
t2 = vaddq_f32(t2, vmulq_n_f32(c2, alpha));
vst1q_f32(col_2, t2);
float32x4_t t3 = vld1q_f32(col_3);
t3 = vaddq_f32(t3, vmulq_n_f32(c3, alpha));
vst1q_f32(col_3, t3);
float32x4_t t4 = vld1q_f32(col_4);
t4 = vaddq_f32(t4, vmulq_n_f32(c4, alpha));
vst1q_f32(col_4, t4);
float32x4_t t5 = vld1q_f32(col_5);
t5 = vaddq_f32(t5, vmulq_n_f32(c5, alpha));
vst1q_f32(col_5, t5);
float32x4_t t6 = vld1q_f32(col_6);
t6 = vaddq_f32(t6, vmulq_n_f32(c6, alpha));
vst1q_f32(col_6, t6);
float32x4_t t7 = vld1q_f32(col_7);
t7 = vaddq_f32(t7, vmulq_n_f32(c7, alpha));
vst1q_f32(col_7, t7);
}
static inline void kernel_2x8(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
float32x2_t c0 = vdup_n_f32(0.0f);
float32x2_t c1 = vdup_n_f32(0.0f);
float32x2_t c2 = vdup_n_f32(0.0f);
float32x2_t c3 = vdup_n_f32(0.0f);
float32x2_t c4 = vdup_n_f32(0.0f);
float32x2_t c5 = vdup_n_f32(0.0f);
float32x2_t c6 = vdup_n_f32(0.0f);
float32x2_t c7 = vdup_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
float32x2_t a_low = vget_low_f32(a_f32);
float16x8_t b_f16 = vld1q_f16(B);
float32x4_t b_low = vcvt_f32_f16(vget_low_f16(b_f16));
float32x4_t b_high = vcvt_f32_f16(vget_high_f16(b_f16));
float32_t b0_lane0 = vgetq_lane_f32(b_low, 0);
c0 = vfma_n_f32(c0, a_low, b0_lane0);
float32_t b0_lane1 = vgetq_lane_f32(b_low, 1);
c1 = vfma_n_f32(c1, a_low, b0_lane1);
float32_t b0_lane2 = vgetq_lane_f32(b_low, 2);
c2 = vfma_n_f32(c2, a_low, b0_lane2);
float32_t b0_lane3 = vgetq_lane_f32(b_low, 3);
c3 = vfma_n_f32(c3, a_low, b0_lane3);
float32_t b1_lane0 = vgetq_lane_f32(b_high, 0);
c4 = vfma_n_f32(c4, a_low, b1_lane0);
float32_t b1_lane1 = vgetq_lane_f32(b_high, 1);
c5 = vfma_n_f32(c5, a_low, b1_lane1);
float32_t b1_lane2 = vgetq_lane_f32(b_high, 2);
c6 = vfma_n_f32(c6, a_low, b1_lane2);
float32_t b1_lane3 = vgetq_lane_f32(b_high, 3);
c7 = vfma_n_f32(c7, a_low, b1_lane3);
A += 2;
B += 8;
}
FLOAT *col_0 = C + 0 * ldc;
FLOAT *col_1 = C + 1 * ldc;
FLOAT *col_2 = C + 2 * ldc;
FLOAT *col_3 = C + 3 * ldc;
FLOAT *col_4 = C + 4 * ldc;
FLOAT *col_5 = C + 5 * ldc;
FLOAT *col_6 = C + 6 * ldc;
FLOAT *col_7 = C + 7 * ldc;
float32x2_t t0 = vld1_f32(col_0);
t0 = vadd_f32(t0, vmul_n_f32(c0, alpha));
vst1_f32(col_0, t0);
float32x2_t t1 = vld1_f32(col_1);
t1 = vadd_f32(t1, vmul_n_f32(c1, alpha));
vst1_f32(col_1, t1);
float32x2_t t2 = vld1_f32(col_2);
t2 = vadd_f32(t2, vmul_n_f32(c2, alpha));
vst1_f32(col_2, t2);
float32x2_t t3 = vld1_f32(col_3);
t3 = vadd_f32(t3, vmul_n_f32(c3, alpha));
vst1_f32(col_3, t3);
float32x2_t t4 = vld1_f32(col_4);
t4 = vadd_f32(t4, vmul_n_f32(c4, alpha));
vst1_f32(col_4, t4);
float32x2_t t5 = vld1_f32(col_5);
t5 = vadd_f32(t5, vmul_n_f32(c5, alpha));
vst1_f32(col_5, t5);
float32x2_t t6 = vld1_f32(col_6);
t6 = vadd_f32(t6, vmul_n_f32(c6, alpha));
vst1_f32(col_6, t6);
float32x2_t t7 = vld1_f32(col_7);
t7 = vadd_f32(t7, vmul_n_f32(c7, alpha));
vst1_f32(col_7, t7);
}
static inline void kernel_1x8(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
FLOAT c0 = 0, c1 = 0, c2 = 0, c3 = 0, c4 = 0, c5 = 0, c6 = 0, c7 = 0;
for (BLASLONG k = 0; k < K; ++k) {
FLOAT a = A[0];
c0 += a * B[0];
c1 += a * B[1];
c2 += a * B[2];
c3 += a * B[3];
c4 += a * B[4];
c5 += a * B[5];
c6 += a * B[6];
c7 += a * B[7];
A += 1;
B += 8;
}
C[0 * ldc] += alpha * c0;
C[1 * ldc] += alpha * c1;
C[2 * ldc] += alpha * c2;
C[3 * ldc] += alpha * c3;
C[4 * ldc] += alpha * c4;
C[5 * ldc] += alpha * c5;
C[6 * ldc] += alpha * c6;
C[7 * ldc] += alpha * c7;
}
static inline void kernel_8x4(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
float32x4_t c0_low = vdupq_n_f32(0.0f);
float32x4_t c0_high = vdupq_n_f32(0.0f);
float32x4_t c1_low = vdupq_n_f32(0.0f);
float32x4_t c1_high = vdupq_n_f32(0.0f);
float32x4_t c2_low = vdupq_n_f32(0.0f);
float32x4_t c2_high = vdupq_n_f32(0.0f);
float32x4_t c3_low = vdupq_n_f32(0.0f);
float32x4_t c3_high = vdupq_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float16x8_t a_f16 = vld1q_f16(A);
float32x4_t a_low = vcvt_f32_f16(vget_low_f16(a_f16));
float32x4_t a_high = vcvt_f32_f16(vget_high_f16(a_f16));
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
c0_low = vfmaq_n_f32(c0_low, a_low, b0_lane0);
c0_high = vfmaq_n_f32(c0_high, a_high, b0_lane0);
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
c1_low = vfmaq_n_f32(c1_low, a_low, b0_lane1);
c1_high = vfmaq_n_f32(c1_high, a_high, b0_lane1);
float32_t b0_lane2 = vgetq_lane_f32(b_f32, 2);
c2_low = vfmaq_n_f32(c2_low, a_low, b0_lane2);
c2_high = vfmaq_n_f32(c2_high, a_high, b0_lane2);
float32_t b0_lane3 = vgetq_lane_f32(b_f32, 3);
c3_low = vfmaq_n_f32(c3_low, a_low, b0_lane3);
c3_high = vfmaq_n_f32(c3_high, a_high, b0_lane3);
A += 8;
B += 4;
}
FLOAT *col_0 = C + 0 * ldc;
FLOAT *col_1 = C + 1 * ldc;
FLOAT *col_2 = C + 2 * ldc;
FLOAT *col_3 = C + 3 * ldc;
float32x4_t t0_l = vld1q_f32(col_0);
float32x4_t t0_h = vld1q_f32(col_0 + 4);
t0_l = vaddq_f32(t0_l, vmulq_n_f32(c0_low, alpha));
t0_h = vaddq_f32(t0_h, vmulq_n_f32(c0_high, alpha));
vst1q_f32(col_0, t0_l);
vst1q_f32(col_0 + 4, t0_h);
float32x4_t t1_l = vld1q_f32(col_1);
float32x4_t t1_h = vld1q_f32(col_1 + 4);
t1_l = vaddq_f32(t1_l, vmulq_n_f32(c1_low, alpha));
t1_h = vaddq_f32(t1_h, vmulq_n_f32(c1_high, alpha));
vst1q_f32(col_1, t1_l);
vst1q_f32(col_1 + 4, t1_h);
float32x4_t t2_l = vld1q_f32(col_2);
float32x4_t t2_h = vld1q_f32(col_2 + 4);
t2_l = vaddq_f32(t2_l, vmulq_n_f32(c2_low, alpha));
t2_h = vaddq_f32(t2_h, vmulq_n_f32(c2_high, alpha));
vst1q_f32(col_2, t2_l);
vst1q_f32(col_2 + 4, t2_h);
float32x4_t t3_l = vld1q_f32(col_3);
float32x4_t t3_h = vld1q_f32(col_3 + 4);
t3_l = vaddq_f32(t3_l, vmulq_n_f32(c3_low, alpha));
t3_h = vaddq_f32(t3_h, vmulq_n_f32(c3_high, alpha));
vst1q_f32(col_3, t3_l);
vst1q_f32(col_3 + 4, t3_h);
}
static inline void kernel_4x4(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
float32x4_t c0 = vdupq_n_f32(0.0f);
float32x4_t c1 = vdupq_n_f32(0.0f);
float32x4_t c2 = vdupq_n_f32(0.0f);
float32x4_t c3 = vdupq_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
c0 = vfmaq_n_f32(c0, a_f32, b0_lane0);
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
c1 = vfmaq_n_f32(c1, a_f32, b0_lane1);
float32_t b0_lane2 = vgetq_lane_f32(b_f32, 2);
c2 = vfmaq_n_f32(c2, a_f32, b0_lane2);
float32_t b0_lane3 = vgetq_lane_f32(b_f32, 3);
c3 = vfmaq_n_f32(c3, a_f32, b0_lane3);
A += 4;
B += 4;
}
FLOAT *col_0 = C + 0 * ldc;
FLOAT *col_1 = C + 1 * ldc;
FLOAT *col_2 = C + 2 * ldc;
FLOAT *col_3 = C + 3 * ldc;
float32x4_t t0 = vld1q_f32(col_0);
t0 = vaddq_f32(t0, vmulq_n_f32(c0, alpha));
vst1q_f32(col_0, t0);
float32x4_t t1 = vld1q_f32(col_1);
t1 = vaddq_f32(t1, vmulq_n_f32(c1, alpha));
vst1q_f32(col_1, t1);
float32x4_t t2 = vld1q_f32(col_2);
t2 = vaddq_f32(t2, vmulq_n_f32(c2, alpha));
vst1q_f32(col_2, t2);
float32x4_t t3 = vld1q_f32(col_3);
t3 = vaddq_f32(t3, vmulq_n_f32(c3, alpha));
vst1q_f32(col_3, t3);
}
static inline void kernel_2x4(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
float32x2_t c0 = vdup_n_f32(0.0f);
float32x2_t c1 = vdup_n_f32(0.0f);
float32x2_t c2 = vdup_n_f32(0.0f);
float32x2_t c3 = vdup_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
float32x2_t a_low = vget_low_f32(a_f32);
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
c0 = vfma_n_f32(c0, a_low, b0_lane0);
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
c1 = vfma_n_f32(c1, a_low, b0_lane1);
float32_t b0_lane2 = vgetq_lane_f32(b_f32, 2);
c2 = vfma_n_f32(c2, a_low, b0_lane2);
float32_t b0_lane3 = vgetq_lane_f32(b_f32, 3);
c3 = vfma_n_f32(c3, a_low, b0_lane3);
A += 2;
B += 4;
}
FLOAT *col_0 = C + 0 * ldc;
FLOAT *col_1 = C + 1 * ldc;
FLOAT *col_2 = C + 2 * ldc;
FLOAT *col_3 = C + 3 * ldc;
float32x2_t t0 = vld1_f32(col_0);
t0 = vadd_f32(t0, vmul_n_f32(c0, alpha));
vst1_f32(col_0, t0);
float32x2_t t1 = vld1_f32(col_1);
t1 = vadd_f32(t1, vmul_n_f32(c1, alpha));
vst1_f32(col_1, t1);
float32x2_t t2 = vld1_f32(col_2);
t2 = vadd_f32(t2, vmul_n_f32(c2, alpha));
vst1_f32(col_2, t2);
float32x2_t t3 = vld1_f32(col_3);
t3 = vadd_f32(t3, vmul_n_f32(c3, alpha));
vst1_f32(col_3, t3);
}
static inline void kernel_1x4(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
FLOAT c0 = 0, c1 = 0, c2 = 0, c3 = 0;
for (BLASLONG k = 0; k < K; ++k) {
FLOAT a = A[0];
c0 += a * B[0];
c1 += a * B[1];
c2 += a * B[2];
c3 += a * B[3];
A += 1;
B += 4;
}
C[0 * ldc] += alpha * c0;
C[1 * ldc] += alpha * c1;
C[2 * ldc] += alpha * c2;
C[3 * ldc] += alpha * c3;
}
static inline void kernel_8x2(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
float32x4_t c0_low = vdupq_n_f32(0.0f);
float32x4_t c0_high = vdupq_n_f32(0.0f);
float32x4_t c1_low = vdupq_n_f32(0.0f);
float32x4_t c1_high = vdupq_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float16x8_t a_f16 = vld1q_f16(A);
float32x4_t a_low = vcvt_f32_f16(vget_low_f16(a_f16));
float32x4_t a_high = vcvt_f32_f16(vget_high_f16(a_f16));
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
c0_low = vfmaq_n_f32(c0_low, a_low, b0_lane0);
c0_high = vfmaq_n_f32(c0_high, a_high, b0_lane0);
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
c1_low = vfmaq_n_f32(c1_low, a_low, b0_lane1);
c1_high = vfmaq_n_f32(c1_high, a_high, b0_lane1);
A += 8;
B += 2;
}
FLOAT *col_0 = C + 0 * ldc;
FLOAT *col_1 = C + 1 * ldc;
float32x4_t t0_l = vld1q_f32(col_0);
float32x4_t t0_h = vld1q_f32(col_0 + 4);
t0_l = vaddq_f32(t0_l, vmulq_n_f32(c0_low, alpha));
t0_h = vaddq_f32(t0_h, vmulq_n_f32(c0_high, alpha));
vst1q_f32(col_0, t0_l);
vst1q_f32(col_0 + 4, t0_h);
float32x4_t t1_l = vld1q_f32(col_1);
float32x4_t t1_h = vld1q_f32(col_1 + 4);
t1_l = vaddq_f32(t1_l, vmulq_n_f32(c1_low, alpha));
t1_h = vaddq_f32(t1_h, vmulq_n_f32(c1_high, alpha));
vst1q_f32(col_1, t1_l);
vst1q_f32(col_1 + 4, t1_h);
}
static inline void kernel_4x2(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
float32x4_t c0 = vdupq_n_f32(0.0f);
float32x4_t c1 = vdupq_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
c0 = vfmaq_n_f32(c0, a_f32, b0_lane0);
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
c1 = vfmaq_n_f32(c1, a_f32, b0_lane1);
A += 4;
B += 2;
}
FLOAT *col_0 = C + 0 * ldc;
FLOAT *col_1 = C + 1 * ldc;
float32x4_t t0 = vld1q_f32(col_0);
t0 = vaddq_f32(t0, vmulq_n_f32(c0, alpha));
vst1q_f32(col_0, t0);
float32x4_t t1 = vld1q_f32(col_1);
t1 = vaddq_f32(t1, vmulq_n_f32(c1, alpha));
vst1q_f32(col_1, t1);
}
static inline void kernel_2x2(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
float32x2_t c0 = vdup_n_f32(0.0f);
float32x2_t c1 = vdup_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
float32x2_t a_low = vget_low_f32(a_f32);
float32x4_t b_f32 = vcvt_f32_f16(vld1_f16(B));
float32_t b0_lane0 = vgetq_lane_f32(b_f32, 0);
c0 = vfma_n_f32(c0, a_low, b0_lane0);
float32_t b0_lane1 = vgetq_lane_f32(b_f32, 1);
c1 = vfma_n_f32(c1, a_low, b0_lane1);
;
A += 2;
B += 2;
}
FLOAT *col_0 = C + 0 * ldc;
FLOAT *col_1 = C + 1 * ldc;
float32x2_t t0 = vld1_f32(col_0);
t0 = vadd_f32(t0, vmul_n_f32(c0, alpha));
vst1_f32(col_0, t0);
float32x2_t t1 = vld1_f32(col_1);
t1 = vadd_f32(t1, vmul_n_f32(c1, alpha));
vst1_f32(col_1, t1);
}
static inline void kernel_1x2(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, BLASLONG ldc, FLOAT alpha) {
FLOAT c0 = 0, c1 = 0;
for (BLASLONG k = 0; k < K; ++k) {
FLOAT a = A[0];
c0 += a * B[0];
c1 += a * B[1];
A += 1;
B += 2;
}
C[0 * ldc] += alpha * c0;
C[1 * ldc] += alpha * c1;
}
static inline void kernel_8x1(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, FLOAT alpha) {
float32x4_t c0_low = vdupq_n_f32(0.0f);
float32x4_t c0_high = vdupq_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float16x8_t a_f16 = vld1q_f16(A);
float32x4_t a_low = vcvt_f32_f16(vget_low_f16(a_f16));
float32x4_t a_high = vcvt_f32_f16(vget_high_f16(a_f16));
float b_scalar = (float)B[0];
c0_low = vfmaq_n_f32(c0_low, a_low, b_scalar);
c0_high = vfmaq_n_f32(c0_high, a_high, b_scalar);
A += 8;
B += 1;
}
FLOAT *col_0 = C;
float32x4_t t0_l = vld1q_f32(col_0);
float32x4_t t0_h = vld1q_f32(col_0 + 4);
t0_l = vaddq_f32(t0_l, vmulq_n_f32(c0_low, alpha));
t0_h = vaddq_f32(t0_h, vmulq_n_f32(c0_high, alpha));
vst1q_f32(col_0, t0_l);
vst1q_f32(col_0 + 4, t0_h);
}
static inline void kernel_4x1(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, FLOAT alpha) {
float32x4_t c0 = vdupq_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
float b_scalar = (float)B[0];
c0 = vfmaq_n_f32(c0, a_f32, b_scalar);
A += 4;
B += 1;
}
FLOAT *col_0 = C;
float32x4_t t0 = vld1q_f32(col_0);
t0 = vaddq_f32(t0, vmulq_n_f32(c0, alpha));
vst1q_f32(col_0, t0);
}
static inline void kernel_2x1(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, FLOAT alpha) {
float32x2_t c0 = vdup_n_f32(0.0f);
for (BLASLONG k = 0; k < K; ++k) {
float32x4_t a_f32 = vcvt_f32_f16(vld1_f16(A));
float32x2_t a_low = vget_low_f32(a_f32);
float b_scalar = (float)B[0];
c0 = vfma_n_f32(c0, a_low, b_scalar);
A += 2;
B += 1;
}
FLOAT *col_0 = C;
float32x2_t t0 = vld1_f32(col_0);
t0 = vadd_f32(t0, vmul_n_f32(c0, alpha));
vst1_f32(col_0, t0);
}
static inline void kernel_1x1(BLASLONG K, const float16_t *A, const float16_t *B, FLOAT *C, FLOAT alpha) {
FLOAT sum = 0.0f;
for (BLASLONG k = 0; k < K; ++k) {
sum += A[0] * B[0];
A += 1;
B += 1;
}
C[0] += alpha * sum;
}
int CNAME(BLASLONG M, BLASLONG N, BLASLONG K, FLOAT alpha, IFLOAT *A, IFLOAT *B, FLOAT *C, BLASLONG ldc) {
float16_t *A_base = (float16_t *)A;
float16_t *B_base = (float16_t *)B;
FLOAT *Ccol = C;
BLASLONG m_rem1, m_rem2, m_rem3, m_rem4;
while (N >= 8) {
const float16_t *Aptr = A_base;
const float16_t *Bptr = B_base;
FLOAT *Crow = Ccol;
m_rem1 = M;
while (m_rem1 >= 8) {
kernel_8x8(K, Aptr, Bptr, Crow, ldc, alpha);
Aptr += K * 8;
Crow += 8;
m_rem1 -= 8;
}
if (m_rem1 >= 4) {
kernel_4x8(K, Aptr, Bptr, Crow, ldc, alpha);
Aptr += K * 4;
Crow += 4;
m_rem1 -= 4;
}
if (m_rem1 >= 2) {
kernel_2x8(K, Aptr, Bptr, Crow, ldc, alpha);
Aptr += K * 2;
Crow += 2;
m_rem1 -= 2;
}
if (m_rem1 >= 1) {
kernel_1x8(K, Aptr, Bptr, Crow, ldc, alpha);
}
B_base += K * 8;
Ccol += ldc * 8;
N -= 8;
}
if (N >= 4) {
const float16_t *Aptr = A_base;
const float16_t *Bptr = B_base;
FLOAT *Crow = Ccol;
m_rem2 = M;
while (m_rem2 >= 8) {
kernel_8x4(K, Aptr, Bptr, Crow, ldc, alpha);
Aptr += K * 8;
Crow += 8;
m_rem2 -= 8;
}
if (m_rem2 >= 4) {
kernel_4x4(K, Aptr, Bptr, Crow, ldc, alpha);
Aptr += K * 4;
Crow += 4;
m_rem2 -= 4;
}
if (m_rem2 >= 2) {
kernel_2x4(K, Aptr, Bptr, Crow, ldc, alpha);
Aptr += K * 2;
Crow += 2;
m_rem2 -= 2;
}
if (m_rem2 >= 1) {
kernel_1x4(K, Aptr, Bptr, Crow, ldc, alpha);
}
B_base += K * 4;
Ccol += ldc * 4;
N -= 4;
}
if (N >= 2) {
const float16_t *Aptr = A_base;
const float16_t *Bptr = B_base;
FLOAT *Crow = Ccol;
m_rem3 = M;
while (m_rem3 >= 8) {
kernel_8x2(K, Aptr, Bptr, Crow, ldc, alpha);
Aptr += K * 8;
Crow += 8;
m_rem3 -= 8;
}
if (m_rem3 >= 4) {
kernel_4x2(K, Aptr, Bptr, Crow, ldc, alpha);
Aptr += K * 4;
Crow += 4;
m_rem3 -= 4;
}
if (m_rem3 >= 2) {
kernel_2x2(K, Aptr, Bptr, Crow, ldc, alpha);
Aptr += K * 2;
Crow += 2;
m_rem3 -= 2;
}
if (m_rem3 >= 1) {
kernel_1x2(K, Aptr, Bptr, Crow, ldc, alpha);
}
B_base += K * 2;
Ccol += ldc * 2;
N -= 2;
}
if (N >= 1) {
const float16_t *Aptr = A_base;
const float16_t *Bptr = B_base;
FLOAT *Crow = Ccol;
m_rem4 = M;
while (m_rem4 >= 8) {
kernel_8x1(K, Aptr, Bptr, Crow, alpha);
Aptr += K * 8;
Crow += 8;
m_rem4 -= 8;
}
if (m_rem4 >= 4) {
kernel_4x1(K, Aptr, Bptr, Crow, alpha);
Aptr += K * 4;
Crow += 4;
m_rem4 -= 4;
}
if (m_rem4 >= 2) {
kernel_2x1(K, Aptr, Bptr, Crow, alpha);
Aptr += K * 2;
Crow += 2;
m_rem4 -= 2;
}
if (m_rem4 >= 1) {
kernel_1x1(K, Aptr, Bptr, Crow, alpha);
}
}
return 0;
}
+258
View File
@@ -0,0 +1,258 @@
/***************************************************************************
* Copyright (c) 2026, The OpenBLAS Project
* All rights reserved.
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions are
* met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in
* the documentation and/or other materials provided with the
* distribution.
* 3. Neither the name of the OpenBLAS project nor the names of
* its contributors may be used to endorse or promote products
* derived from this software without specific prior written permission.
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
* POSSIBILITY OF SUCH DAMAGE.
* *****************************************************************************/
#include <arm_neon.h>
#include "common.h"
static inline void transpose8x8(float16x8_t *rows, float16x8_t *cols) {
float64x2_t b0 = vtrn1q_f64(vreinterpretq_f64_f16(rows[0]), vreinterpretq_f64_f16(rows[4]));
float64x2_t b1 = vtrn1q_f64(vreinterpretq_f64_f16(rows[1]), vreinterpretq_f64_f16(rows[5]));
float64x2_t b2 = vtrn1q_f64(vreinterpretq_f64_f16(rows[2]), vreinterpretq_f64_f16(rows[6]));
float64x2_t b3 = vtrn1q_f64(vreinterpretq_f64_f16(rows[3]), vreinterpretq_f64_f16(rows[7]));
float64x2_t b4 = vtrn2q_f64(vreinterpretq_f64_f16(rows[0]), vreinterpretq_f64_f16(rows[4]));
float64x2_t b5 = vtrn2q_f64(vreinterpretq_f64_f16(rows[1]), vreinterpretq_f64_f16(rows[5]));
float64x2_t b6 = vtrn2q_f64(vreinterpretq_f64_f16(rows[2]), vreinterpretq_f64_f16(rows[6]));
float64x2_t b7 = vtrn2q_f64(vreinterpretq_f64_f16(rows[3]), vreinterpretq_f64_f16(rows[7]));
float32x4_t c0 = vtrn1q_f32(vreinterpretq_f32_f64(b0), vreinterpretq_f32_f64(b2));
float32x4_t c1 = vtrn1q_f32(vreinterpretq_f32_f64(b1), vreinterpretq_f32_f64(b3));
float32x4_t c2 = vtrn2q_f32(vreinterpretq_f32_f64(b0), vreinterpretq_f32_f64(b2));
float32x4_t c3 = vtrn2q_f32(vreinterpretq_f32_f64(b1), vreinterpretq_f32_f64(b3));
float32x4_t c4 = vtrn1q_f32(vreinterpretq_f32_f64(b4), vreinterpretq_f32_f64(b6));
float32x4_t c5 = vtrn1q_f32(vreinterpretq_f32_f64(b5), vreinterpretq_f32_f64(b7));
float32x4_t c6 = vtrn2q_f32(vreinterpretq_f32_f64(b4), vreinterpretq_f32_f64(b6));
float32x4_t c7 = vtrn2q_f32(vreinterpretq_f32_f64(b5), vreinterpretq_f32_f64(b7));
float16x8_t d0 = vtrn1q_f16(vreinterpretq_f16_f32(c0), vreinterpretq_f16_f32(c1));
float16x8_t d1 = vtrn2q_f16(vreinterpretq_f16_f32(c0), vreinterpretq_f16_f32(c1));
float16x8_t d2 = vtrn1q_f16(vreinterpretq_f16_f32(c2), vreinterpretq_f16_f32(c3));
float16x8_t d3 = vtrn2q_f16(vreinterpretq_f16_f32(c2), vreinterpretq_f16_f32(c3));
float16x8_t d4 = vtrn1q_f16(vreinterpretq_f16_f32(c4), vreinterpretq_f16_f32(c5));
float16x8_t d5 = vtrn2q_f16(vreinterpretq_f16_f32(c4), vreinterpretq_f16_f32(c5));
float16x8_t d6 = vtrn1q_f16(vreinterpretq_f16_f32(c6), vreinterpretq_f16_f32(c7));
float16x8_t d7 = vtrn2q_f16(vreinterpretq_f16_f32(c6), vreinterpretq_f16_f32(c7));
cols[0] = d0;
cols[1] = d1;
cols[2] = d2;
cols[3] = d3;
cols[4] = d4;
cols[5] = d5;
cols[6] = d6;
cols[7] = d7;
}
static inline void transpose_4x4(float16x4_t *rows, float16x4_t *cols) {
float16x8_t t0 = vcombine_f16(rows[0], vdup_n_f16(0.0f));
float16x8_t t1 = vcombine_f16(rows[1], vdup_n_f16(0.0f));
float16x8_t t2 = vcombine_f16(rows[2], vdup_n_f16(0.0f));
float16x8_t t3 = vcombine_f16(rows[3], vdup_n_f16(0.0f));
float16x8_t t02 = vzip1q_f16(t0, t2);
float16x8_t t13 = vzip1q_f16(t1, t3);
float16x8x2_t t0123 = vzipq_f16(t02, t13);
cols[0] = vget_low_f16(t0123.val[0]);
cols[1] = vget_high_f16(t0123.val[0]);
cols[2] = vget_low_f16(t0123.val[1]);
cols[3] = vget_high_f16(t0123.val[1]);
}
int CNAME(BLASLONG m, BLASLONG n, IFLOAT *a, BLASLONG lda, IFLOAT *b) {
BLASLONG i, j;
IFLOAT *a_offset = a;
IFLOAT *b_offset = b;
float16x8_t v0, v1, v2, v3, v4, v5, v6, v7;
float16x4_t v8, v9, v10, v11;
BLASLONG n8 = n >> 3;
for (j = 0; j < n8; j++) {
IFLOAT *a0 = a_offset;
IFLOAT *a1 = a0 + lda;
IFLOAT *a2 = a1 + lda;
IFLOAT *a3 = a2 + lda;
IFLOAT *a4 = a3 + lda;
IFLOAT *a5 = a4 + lda;
IFLOAT *a6 = a5 + lda;
IFLOAT *a7 = a6 + lda;
a_offset += 8 * lda;
BLASLONG m8 = m >> 3;
for (i = 0; i < m8; i++) {
v0 = vld1q_f16((float16_t *)a0);
v1 = vld1q_f16((float16_t *)a1);
v2 = vld1q_f16((float16_t *)a2);
v3 = vld1q_f16((float16_t *)a3);
v4 = vld1q_f16((float16_t *)a4);
v5 = vld1q_f16((float16_t *)a5);
v6 = vld1q_f16((float16_t *)a6);
v7 = vld1q_f16((float16_t *)a7);
float16x8_t rows[8] = {v0, v1, v2, v3, v4, v5, v6, v7};
float16x8_t cols[8];
transpose8x8(rows, cols);
vst1q_f16((float16_t *)b_offset, cols[0]);
vst1q_f16((float16_t *)b_offset + 8, cols[1]);
vst1q_f16((float16_t *)b_offset + 16, cols[2]);
vst1q_f16((float16_t *)b_offset + 24, cols[3]);
vst1q_f16((float16_t *)b_offset + 32, cols[4]);
vst1q_f16((float16_t *)b_offset + 40, cols[5]);
vst1q_f16((float16_t *)b_offset + 48, cols[6]);
vst1q_f16((float16_t *)b_offset + 56, cols[7]);
a0 += 8;
a1 += 8;
a2 += 8;
a3 += 8;
a4 += 8;
a5 += 8;
a6 += 8;
a7 += 8;
b_offset += 64;
}
BLASLONG i = (m & 7);
if (i > 0) {
for (BLASLONG k = 0; k < i; k++) {
*(b_offset + 0) = *a0;
*(b_offset + 1) = *a1;
*(b_offset + 2) = *a2;
*(b_offset + 3) = *a3;
*(b_offset + 4) = *a4;
*(b_offset + 5) = *a5;
*(b_offset + 6) = *a6;
*(b_offset + 7) = *a7;
a0++;
a1++;
a2++;
a3++;
a4++;
a5++;
a6++;
a7++;
b_offset += 8;
}
}
}
if (n & 4) {
IFLOAT *a0 = a_offset;
IFLOAT *a1 = a0 + lda;
IFLOAT *a2 = a1 + lda;
IFLOAT *a3 = a2 + lda;
a_offset += 4 * lda;
BLASLONG m4 = m >> 2;
for (i = 0; i < m4; i++) {
v8 = vld1_f16((float16_t *)a0);
v9 = vld1_f16((float16_t *)a1);
v10 = vld1_f16((float16_t *)a2);
v11 = vld1_f16((float16_t *)a3);
float16x4_t rows[4] = {v8, v9, v10, v11};
float16x4_t cols[4];
transpose_4x4(rows, cols);
vst1_f16((float16_t *)b_offset, cols[0]);
vst1_f16((float16_t *)b_offset + 4, cols[1]);
vst1_f16((float16_t *)b_offset + 8, cols[2]);
vst1_f16((float16_t *)b_offset + 12, cols[3]);
a0 += 4;
a1 += 4;
a2 += 4;
a3 += 4;
b_offset += 16;
}
BLASLONG i = (m & 3);
if (i > 0) {
for (BLASLONG k = 0; k < i; k++) {
*(b_offset + 0) = *a0;
*(b_offset + 1) = *a1;
*(b_offset + 2) = *a2;
*(b_offset + 3) = *a3;
a0++;
a1++;
a2++;
a3++;
b_offset += 4;
}
}
}
if (n & 2) {
IFLOAT *a0 = a_offset;
IFLOAT *a1 = a0 + lda;
a_offset += 2 * lda;
BLASLONG m2 = m >> 1;
for (i = 0; i < m2; i++) {
v8 = vld1_f16((float16_t *)a0);
v9 = vld1_f16((float16_t *)a1);
float16_t col0[2] = {vget_lane_f16(v8, 0), vget_lane_f16(v9, 0)};
float16_t col1[2] = {vget_lane_f16(v8, 1), vget_lane_f16(v9, 1)};
b_offset[0] = col0[0];
b_offset[1] = col0[1];
b_offset[2] = col1[0];
b_offset[3] = col1[1];
a0 += 2;
a1 += 2;
b_offset += 4;
}
if (m & 1) {
b_offset[0] = *a0;
b_offset[1] = *a1;
b_offset += 2;
}
}
if (n & 1) {
IFLOAT *a0 = a_offset;
for (i = 0; i < m; i++) {
*b_offset++ = *a0;
a0++;
}
}
return 0;
}
+87
View File
@@ -0,0 +1,87 @@
/***************************************************************************
* Copyright (c) 2026, The OpenBLAS Project
* All rights reserved.
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions are
* met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in
* the documentation and/or other materials provided with the
* distribution.
* 3. Neither the name of the OpenBLAS project nor the names of
* its contributors may be used to endorse or promote products
* derived from this software without specific prior written permission.
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
* POSSIBILITY OF SUCH DAMAGE.
* *****************************************************************************/
#include <arm_sve.h>
#include "common.h"
int CNAME(BLASLONG m, BLASLONG n, IFLOAT *a, BLASLONG lda, IFLOAT *b) {
BLASLONG i, j;
IFLOAT *aoffset, *aoffset1;
IFLOAT *boffset, *boffset1;
IFLOAT *boffset2, *boffset3, *boffset4;
aoffset = a;
boffset = b;
boffset2 = b + m * (n & ~7);
boffset3 = b + m * (n & ~3);
boffset4 = b + m * (n & ~1);
svbool_t pg8 = svwhilelt_b16(0, 8);
svbool_t pg4 = svwhilelt_b16(0, 4);
for (j = 0; j < m; j++) {
aoffset1 = aoffset;
boffset1 = boffset;
aoffset += lda;
boffset += 8;
for (i = 0; i < (n >> 3); i++) {
svfloat16_t v0 = svld1_f16(pg8, (float16_t *)aoffset1);
svst1_f16(pg8, (float16_t *)boffset1, v0);
aoffset1 += 8;
boffset1 += 8 * m;
}
if (n & 4) {
svfloat16_t v0 = svld1_f16(pg4, (float16_t *)aoffset1);
svst1_f16(pg4, (float16_t *)boffset2, v0);
aoffset1 += 4;
boffset2 += 4;
}
if (n & 2) {
boffset3[0] = aoffset1[0];
boffset3[1] = aoffset1[1];
aoffset1 += 2;
boffset3 += 2;
}
if (n & 1) {
boffset4[0] = aoffset1[0];
aoffset1 += 1;
boffset4 += 1;
}
}
return 0;
}
+245
View File
@@ -0,0 +1,245 @@
/***************************************************************************
Copyright (c) 2017, The OpenBLAS Project
All rights reserved.
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the following conditions are
met:
1. Redistributions of source code must retain the above copyright
notice, this list of conditions and the following disclaimer.
2. Redistributions in binary form must reproduce the above copyright
notice, this list of conditions and the following disclaimer in
the documentation and/or other materials provided with the
distribution.
3. Neither the name of the OpenBLAS project nor the names of
its contributors may be used to endorse or promote products
derived from this software without specific prior written permission.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
ARE DISCLAIMED. IN NO EVENT SHALL THE OPENBLAS PROJECT OR CONTRIBUTORS BE
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*****************************************************************************/
#include "common.h"
#include <arm_neon.h>
#define N "x0" /* vector length */
#define X "x1" /* "X" vector address */
#define INC_X "x2" /* "X" stride */
#define J "x5" /* loop variable */
#define REG0 "wzr"
#define SUMF "s0"
#define SUMFD "d0"
/******************************************************************************/
#define KERNEL_F1 \
"ldr s1, ["X"] \n" \
"add "X", "X", #4 \n" \
"fadd "SUMF", "SUMF", s1 \n"
#define KERNEL_F64 \
"ldr q16, ["X"] \n" \
"ldr q17, ["X", #16] \n" \
"ldr q18, ["X", #32] \n" \
"ldr q19, ["X", #48] \n" \
"ldp q20, q21, ["X", #64] \n" \
"ldp q22, q23, ["X", #96] \n" \
"ldp q24, q25, ["X", #128] \n" \
"ldp q26, q27, ["X", #160] \n" \
"fadd v16.4s, v16.4s, v17.4s \n" \
"fadd v18.4s, v18.4s, v19.4s \n" \
"ldp q28, q29, ["X", #192] \n" \
"ldp q30, q31, ["X", #224] \n" \
"add "X", "X", #256 \n" \
"fadd v20.4s, v20.4s, v21.4s \n" \
"fadd v22.4s, v22.4s, v23.4s \n" \
"PRFM PLDL1KEEP, ["X", #1024] \n" \
"PRFM PLDL1KEEP, ["X", #1024+64] \n" \
"fadd v24.4s, v24.4s, v25.4s \n" \
"fadd v26.4s, v26.4s, v27.4s \n" \
"fadd v0.4s, v0.4s, v16.4s \n" \
"fadd v1.4s, v1.4s, v18.4s \n" \
"fadd v2.4s, v2.4s, v20.4s \n" \
"fadd v3.4s, v3.4s, v22.4s \n" \
"PRFM PLDL1KEEP, ["X", #1024+128] \n" \
"PRFM PLDL1KEEP, ["X", #1024+192] \n" \
"fadd v28.4s, v28.4s, v29.4s \n" \
"fadd v30.4s, v30.4s, v31.4s \n" \
"fadd v4.4s, v4.4s, v24.4s \n" \
"fadd v5.4s, v5.4s, v26.4s \n" \
"fadd v6.4s, v6.4s, v28.4s \n" \
"fadd v7.4s, v7.4s, v30.4s \n"
#define KERNEL_F64_FINALIZE \
"fadd v0.4s, v0.4s, v1.4s \n" \
"fadd v2.4s, v2.4s, v3.4s \n" \
"fadd v4.4s, v4.4s, v5.4s \n" \
"fadd v6.4s, v6.4s, v7.4s \n" \
"fadd v0.4s, v0.4s, v2.4s \n" \
"fadd v4.4s, v4.4s, v6.4s \n" \
"fadd v0.4s, v0.4s, v4.4s \n" \
"ext v1.16b, v0.16b, v0.16b, #8 \n" \
"fadd v0.2s, v0.2s, v1.2s \n" \
"faddp "SUMF", v0.2s \n"
#define INIT_S \
"lsl "INC_X", "INC_X", #2 \n"
#define KERNEL_S1 \
"ldr s1, ["X"] \n" \
"add "X", "X", "INC_X" \n" \
"fadd "SUMF", "SUMF", s1 \n"
#if defined(SMP)
extern int blas_level1_thread_with_return_value(int mode, BLASLONG m, BLASLONG n,
BLASLONG k, void *alpha, void *a, BLASLONG lda, void *b, BLASLONG ldb,
void *c, BLASLONG ldc, int (*function)(), int nthreads);
#endif
static FLOAT ssum_compute(BLASLONG n, FLOAT *x, BLASLONG inc_x)
{
FLOAT ssum = 0.0 ;
if ( n < 0 ) return(ssum);
__asm__ __volatile__ (
" mov "N", %[N_] \n"
" mov "X", %[X_] \n"
" mov "INC_X", %[INCX_] \n"
" fmov "SUMF", "REG0" \n"
" fmov s1, "REG0" \n"
" fmov s2, "REG0" \n"
" fmov s3, "REG0" \n"
" fmov s4, "REG0" \n"
" fmov s5, "REG0" \n"
" fmov s6, "REG0" \n"
" fmov s7, "REG0" \n"
" cmp "N", xzr \n"
" ble 9f //ssum_kernel_L999 \n"
" cmp "INC_X", xzr \n"
" ble 9f //ssum_kernel_L999 \n"
" cmp "INC_X", #1 \n"
" bne 5f //ssum_kernel_S_BEGIN \n"
"1: //sum_kernel_F_BEGIN: \n"
" asr "J", "N", #6 \n"
" cmp "J", xzr \n"
" beq 3f //ssum_kernel_F1 \n"
#if !(defined(__clang__) && defined(OS_WINDOWS))
".align 5 \n"
#endif
"2: //ssum_kernel_F64: \n"
" "KERNEL_F64" \n"
" subs "J", "J", #1 \n"
" bne 2b //ssum_kernel_F64 \n"
" "KERNEL_F64_FINALIZE" \n"
"3: //ssum_kernel_F1: \n"
" ands "J", "N", #63 \n"
" ble 9f //ssum_kernel_L999 \n"
"4: //ssum_kernel_F10: \n"
" "KERNEL_F1" \n"
" subs "J", "J", #1 \n"
" bne 4b //ssum_kernel_F10 \n"
" b 9f //ssum_kernel_L999 \n"
"5: //ssum_kernel_S_BEGIN: \n"
" "INIT_S" \n"
" asr "J", "N", #2 \n"
" cmp "J", xzr \n"
" ble 7f //ssum_kernel_S1 \n"
"6: //ssum_kernel_S4: \n"
" "KERNEL_S1" \n"
" "KERNEL_S1" \n"
" "KERNEL_S1" \n"
" "KERNEL_S1" \n"
" subs "J", "J", #1 \n"
" bne 6b //ssum_kernel_S4 \n"
"7: //ssum_kernel_S1: \n"
" ands "J", "N", #3 \n"
" ble 9f //ssum_kernel_L999 \n"
"8: //ssum_kernel_S10: \n"
" "KERNEL_S1" \n"
" subs "J", "J", #1 \n"
" bne 8b //ssum_kernel_S10 \n"
"9: //ssum_kernel_L999: \n"
" fmov %[SSUM_], "SUMFD" \n"
: [SSUM_] "=r" (ssum) //%0
: [N_] "r" (n), //%1
[X_] "r" (x), //%2
[INCX_] "r" (inc_x) //%3
: "cc",
"memory",
"x0", "x1", "x2", "x3", "x4", "x5",
"d0", "d1", "d2", "d3", "d4", "d5", "d6", "d7"
);
return ssum;
}
#if defined(SMP)
static int ssum_thread_function(BLASLONG n, BLASLONG dummy0,
BLASLONG dummy1, FLOAT dummy2, FLOAT *x, BLASLONG inc_x, FLOAT *y,
BLASLONG inc_y, FLOAT *result, BLASLONG dummy3)
{
*result = ssum_compute(n, x, inc_x);
return 0;
}
#endif
FLOAT CNAME(BLASLONG n, FLOAT *x, BLASLONG inc_x)
{
#if defined(SMP)
int nthreads;
FLOAT dummy_alpha;
#endif
FLOAT ssum = 0.0;
#if defined(SMP)
if (inc_x == 0 || n <= 10000)
nthreads = 1;
else
nthreads = num_cpu_avail(1);
if (nthreads == 1) {
ssum = ssum_compute(n, x, inc_x);
} else {
int mode, i;
char result[MAX_CPU_NUMBER * sizeof(double) * 2];
FLOAT *ptr;
mode = BLAS_SINGLE;
blas_level1_thread_with_return_value(mode, n, 0, 0, &dummy_alpha,
x, inc_x, NULL, 0, result, 0,
( void *)ssum_thread_function, nthreads);
ptr = (FLOAT *)result;
for (i = 0; i < nthreads; i++) {
ssum = ssum + (*ptr);
ptr = (FLOAT *)(((char *)ptr) + sizeof(double) * 2);
}
}
#else
ssum = ssum_compute(n, x, inc_x);
#endif
return ssum;
}

Some files were not shown because too many files have changed in this diff Show More