Compare commits

...
Author SHA1 Message Date
Ryan C. Bleile b86dd9e455 Updated NVCC build with device linkable option 2019-05-13 14:18:51 -07:00
Kenneth Weiss 20becdcab0 Updates Axom TPL setup 2019-04-30 21:45:30 -07:00
Kenneth Weiss 690eb80767 Updates axom library names in build system 2019-04-30 20:50:44 -07:00
Kenneth Weiss 7798771ccb Updates SidreDataCollection due to changes to Axom's include directory structure
axom::sidre::SidreLength was also renamed as axom::sidre::IndexType.
2019-04-30 20:50:32 -07:00
Tzanio Kolev c097bda546 Merge pull request #870 from mfem/4.0-rc2-docs
Updated documentation to mention GPU support [4.0-rc2-docs]
2019-04-24 14:02:40 -07:00
Tzanio Kolev 56608342e8 Merge pull request #875 from mfem/okina-ex6fix
Okina ex6fix [okina-ex6fix]
2019-04-24 14:02:05 -07:00
Tzanio Kolev a1c1da8e9c Merge pull request #882 from mfem/autotest-devices
Update device sample runs [autotest-devices]
2019-04-24 14:01:50 -07:00
Tzanio 72968077c6 CUDA driver no longer needed in top-level CMakeList.txt.
See https://github.com/mfem/mfem/pull/862#issuecomment-485171820.
2019-04-24 13:59:45 -07:00
Tzanio 9cebf45288 Merge branch 'master' into okina-ex6fix
Conflicts:
	general/cuda.hpp
2019-04-24 13:58:31 -07:00
Tzanio Kolev ffc2dfc70b Merge pull request #862 from mfem/okina-cmake
Okina-CMake: CUDA, OCCA, RAJA + MPI [okina-cmake]
2019-04-24 13:54:24 -07:00
Tzanio Kolev d996ee2d39 Merge pull request #868 from mfem/okina-nodrv
Remove CU driver calls [okina-nodrv]
2019-04-24 13:54:02 -07:00
Veselin Dobrev 49c25eec31 Fix a typo. 2019-04-24 12:10:25 -07:00
camierjs 27f8d46aae Add the -dev option to launch devices tests 2019-04-24 10:14:10 -07:00
Tzanio 4a64afedc2 doxygen fix 2019-04-24 09:00:16 -07:00
Tzanio 03d36aa518 Set the RC2 date to today 2019-04-24 07:07:45 -07:00
Veselin Dobrev b2f154112c Update the script config/sample-runs.sh to filter out device runs. 2019-04-23 21:47:50 -07:00
Tzanio 247f9c7445 Small edits 2019-04-23 21:39:21 -07:00
Tzanio 2baece3ab1 Adjusted documentation, mentioned in CHANGELOG 2019-04-23 21:32:43 -07:00
Veselin Dobrev 5b5769dea3 In class SparseMatrix:
* Add methods BuildTranspose() and ResetTranspose() that control
  the use of the internal transpose matrix, At.
* The method AddMultTranspose() will always use At, if it is built.
  If At is not build and the Device is enabled, an error will be
  generated pointing to BuildTranspose().
* Introduce separate non-const and const versions of the methods
  GetI(), GetJ(), and GetData().
* Made the method ActualWidth() const.
* Some edits in the documentation, the code formatting, and the
  error messages.

In the method BilinearForm::FormLinearSystem(), call the method
SparseMatrix::BuildTranspose() for the nonconforming prolongation
matrix, when necessary.
2019-04-23 17:48:40 -07:00
Tzanio f617acf414 Removed a comment about the CUDA driver (no longer needed). 2019-04-22 18:30:14 -07:00
Veselin Dobrev 68fbe31aa1 In config/defaults.mk, use '=' to set {OCCA,RAJA}_DIR.
This makes it easier to configure mfem by copying defaults.mk to
user.mk and editting it: if using '?=', the value given in user.mk
will not overwrite the one from defaults.mk.
2019-04-22 16:51:13 -07:00
Veselin Dobrev 8142e822d8 Fix non-CUDA builds. 2019-04-22 16:41:09 -07:00
Veselin Dobrev a7303349e0 Add checks that the memory manager is enabled when CUDA is enabled. 2019-04-22 16:26:46 -07:00
Veselin Dobrev 10b3988449 Reworked the macro MFEM_CUDA_CHECK:
* It always performs the error check, no just in debug mode.
* All 'cuda*' runtime calls are now wrapped with this macro.
2019-04-22 15:17:04 -07:00
Veselin Dobrev bc876e1c64 Remove the CUDA driver from the GNU make build system. 2019-04-22 15:14:18 -07:00
Tzanio Kolev 1a8c36e92f Merge pull request #871 from mfem/bugfix/windows
Bugfix/windows
2019-04-21 21:44:57 -07:00
Veselin Dobrev aa027c2b8d In the CMake build system, define CUDA_ARCH in defaults.cmake. 2019-04-19 21:06:57 -07:00
Veselin Dobrev 8a8d9419d1 Some small modifications in the build systems. 2019-04-19 20:51:16 -07:00
camierjs 6e82bd6ada Update new nodes before rebalancing 2019-04-19 17:01:07 -07:00
Tzanio c9d80fc64f minor 2019-04-19 15:58:20 -07:00
Tzanio c9762fe73e Update INSTALL: CMake + CUDA build, list Homebrew/Science as deprecated. 2019-04-19 15:40:04 -07:00
Tzanio 41f45474b2 Merge branch 'okina-cmake' of github.com:mfem/mfem into okina-cmake 2019-04-19 15:39:38 -07:00
Tzanio 4313a7b00f Small adjustemnt in XSDKDefaults.cmake 2019-04-19 15:29:03 -07:00
camierjs 3e4301a4cb Install the okl files 2019-04-19 14:57:04 -07:00
camierjs 6991239cc0 XSDKDefaults tweaks 2019-04-19 14:36:28 -07:00
camierjs 9655ceaaef Merge branch 'okina-cmake' of github.com:mfem/mfem into okina-cmake 2019-04-19 14:07:38 -07:00
camierjs b92b3acc1a Add CMAKE_CUDA_STANDARD/REQUIRED/EXTENSIONS 2019-04-19 14:06:50 -07:00
Tzanio 226ccf8db7 RC2-related changes in CHANGELOG 2019-04-19 13:15:01 -07:00
Tzanio 4ffe4a4beb styling 2019-04-19 12:40:49 -07:00
Tzanio dbaa40a116 Renamed tA to At 2019-04-19 12:31:58 -07:00
camierjs 8ad78ae156 Add MPI + CUDA support 2019-04-19 12:00:59 -07:00
camierjs 118a4dcde4 MFEM_CUDA_CHECK fix 2019-04-19 11:39:46 -07:00
camierjs 3e34d4b99a Cleanup, style and remove all CUdevice, CUcontext & CUstream 2019-04-19 11:00:23 -07:00
camierjs 06ee78f67c Switch to use Device::IsEnable 2019-04-19 10:59:01 -07:00
camierjs be8eac8997 Keep legacy AddMultTranspose code for non-accelerated runs 2019-04-19 10:16:02 -07:00
camierjs 890579e228 Cleanup and add a self-transposed sparse matrix that is used 2019-04-19 09:48:15 -07:00
camierjs ceb8bb1417 Remove AtomicAdd from sparsemat AddMultTranspose 2019-04-18 18:21:06 -07:00
Veselin Dobrev a06fe30a73 Fix the "check" CMake target for Visual Studio. 2019-04-18 16:07:54 -07:00
jonesholger 2bb423434c Update .appveyor.yml 2019-04-17 23:02:30 -07:00
jonesholger 1bf5b9098f Update .appveyor.yml 2019-04-17 22:29:28 -07:00
jonesholger ed431414c2 Update .appveyor.yml 2019-04-17 22:17:43 -07:00
Holger Jones 0bbe93c26f Need to specify release config 2019-04-17 21:52:48 -07:00
Holger Jones d41d992798 modify test target to RUN_TESTS, which is known to work under windows 2019-04-17 21:31:32 -07:00
Holger Jones 881cc50cfd fix to bring in platform specific rmdir; lowered pts threshold in inversetransform test 2019-04-17 21:21:08 -07:00
Veselin Dobrev fb7be12a77 Merge pull request #863 from mfem/cmake-unit-tests-fix
Fix a bug in the CMake file for the unit tests [cmake-unit-tests-fix]
2019-04-17 19:43:59 -07:00
Veselin Dobrev 4e48ebc0cf Merge pull request #837 from rcarson3/hypre-dep-dev
Update hypre version and point users to the LLNL repository for hypre [rcarson3:hypre-dep-dev]
2019-04-17 19:41:39 -07:00
Tzanio ac12259cba Mention that MFEM_USE_LEGACY_OPENMP is deprecated in INSTALL. 2019-04-17 19:25:23 -07:00
Tzanio 99fbdcdf73 Mentioned GPU classes in doc/CodeDocumentation.dox 2019-04-17 18:12:55 -07:00
Tzanio aaead6c866 Updated README and CONTRIBUTING to mention GPUs 2019-04-17 18:00:21 -07:00
camierjs d3a1685cb6 Remove CU driver calls 2019-04-17 17:36:43 -07:00
camierjs 16ca9883eb Use CMake 3.8 CUDA native support to compile MFEM + examples 2019-04-17 16:15:26 -07:00
Veselin Dobrev 2a0c8f25d3 Fix a bug in tests/unit/CMakeLists.txt that prevented building of
the unit tests with CMake.
2019-04-16 22:38:30 -07:00
Tzanio Kolev e9691ba40e Merge pull request #859 from mfem/install-okl-fix
Fix error messages when installing *.okl files [install-okl-fix]
2019-04-16 21:45:36 -07:00
camierjs 21edf56417 First cmake MFEM_USE_CUDA/OCCA/RAJA pass 2019-04-16 18:54:01 -07:00
Tzanio 80a7cbaafe Hypre clarification 2019-04-16 10:36:16 -07:00
Veselin Dobrev 90b5e07681 In the main makefile, install *.okl files in a separate loop to
avoid error messages.
2019-04-15 19:47:20 -07:00
camierjs e9de1fcf7b Remove '>' in front of devices tests 2019-04-15 16:05:18 -07:00
Tzanio 99759b6e7c Updated hypre's URL 2019-04-13 22:36:19 -07:00
Veselin Dobrev a63cbf6841 Update .travis.yml
Link `hypre-2.10.0b` as `hypre`.
2019-04-07 14:19:20 -07:00
rcarson3 311f538fd5 Update hypre dependency version and point to github source
The hypre dependency version has been updated to now be the latest available on the LLNL github repository for hypre. The INSTALL file has also been updated to point people to the github page to have them download/clone the repository. Next, the build make/cmake files have also been updated to reflect that the hypre directory is now just hypre rather than hypre-2.10.0b.
2019-04-04 15:34:26 -07:00
41 changed files with 570 additions and 336 deletions
+3 -1
View File
@@ -43,7 +43,9 @@ before_build:
build_script:
- cmake --build build_parallel
- cmake --build build_serial
- cmake --build build_serial --target exec
after_build:
# - cmake --build build_parallel --target check
- cmake --build build_serial --target check
- cmake --build build_serial --target RUN_TESTS
+1
View File
@@ -205,6 +205,7 @@ install:
else
echo "Reusing cached hypre-2.10.0b/";
fi;
ln -s hypre-2.10.0b hypre;
else
echo "Serial build, not using hypre";
fi
+9 -6
View File
@@ -8,22 +8,21 @@
http://mfem.org
Version 4.0-RC1, Apr 11, 2019
Version 4.0-RC2, Apr 24, 2019
=============================
Requirements and Limitations
----------------------------
- This is a release candidate for mfem-4.0.
- Use at your own risk -- not everything will work, the API may change.
- Use at your own risk -- not everything will work and the API may change.
- We are looking for feedback from friendly users.
- Unlike previous MFEM releases, this version requires a C++11 compiler.
- GPU-related limitations:
* NVCC is not supported in the CMake build system yet.
* Element batching is currently ignored.
* Hypre preconditioners are not yet available in GPU mode.
* Only constant coefficients are currently supported on GPUs.
* Full-assembly (on device), element assembly, and matrix-free bilinear forms
are not supported yet.
* FunctionCoefficients do not currently work on GPUs.
are not supported yet. Element batching is currently ignored.
* Partial assembly kernels are not implemented yet for simplices.
GPU support
@@ -145,6 +144,10 @@ New and improved solvers and preconditioners
Miscellaneous
-------------
- In SparseMatrix added the option to perform MultTranspose() by matvec with
computed and stored transpose matrix. This is required for deterministic
results when using devices such as CUDA and OpenMP.
- Added unit tests based on the Catch++ library.
- Renamed the option MFEM_USE_OPENMP to MFEM_USE_LEGACY_OPENMP. This legacy
+56 -4
View File
@@ -86,6 +86,13 @@ include("${CMAKE_CURRENT_SOURCE_DIR}/config/XSDKDefaults.cmake")
# Enable languages.
enable_language(CXX)
if (MFEM_USE_CUDA)
# MFEM_USE_CUDA requires CMake 3.8 or newer (for direct CUDA support)
cmake_minimum_required(VERSION 3.8 FATAL_ERROR)
enable_language(CUDA)
message(STATUS "Using CUDA architecture: ${CUDA_ARCH}")
endif()
if (XSDK_ENABLE_C)
enable_language(C)
endif()
@@ -247,7 +254,7 @@ endif()
# Axom/Sidre
if (MFEM_USE_SIDRE)
find_package(Axom REQUIRED Sidre SLIC axom_utils)
find_package(Axom REQUIRED Axom)
endif()
# PUMI
@@ -266,6 +273,32 @@ if (MFEM_USE_PUMI)
endif()
endif()
# CUDA
if (MFEM_USE_CUDA)
set(CMAKE_CUDA_STANDARD 11)
set(CMAKE_CUDA_STANDARD_REQUIRED ON)
set(CMAKE_CUDA_EXTENSIONS OFF)
set(CMAKE_CUDA_FLAGS "-arch=${CUDA_ARCH} --expt-extended-lambda"
CACHE STRING "CUDA flags set for MFEM" FORCE)
if (MFEM_USE_MPI)
set(CUDA_CCBIN_COMPILER ${MPI_CXX_COMPILER})
else()
set(CUDA_CCBIN_COMPILER ${CMAKE_CXX_COMPILER})
endif()
string(APPEND CMAKE_CUDA_FLAGS " -ccbin ${CUDA_CCBIN_COMPILER}")
set(MFEM_USE_MM YES CACHE BOOL "Enable MFEM's memory manager" FORCE)
endif()
# OCCA
if (MFEM_USE_OCCA)
find_package(OCCA REQUIRED)
endif()
# RAJA
if (MFEM_USE_RAJA)
find_package(RAJA REQUIRED)
endif()
# MFEM_TIMER_TYPE
if (NOT DEFINED MFEM_TIMER_TYPE)
if (APPLE)
@@ -291,7 +324,7 @@ endif()
# be before SuiteSparse.
set(MFEM_TPLS MPI_CXX OPENMP BLAS LAPACK METIS HYPRE SuiteSparse SUNDIALS PETSC
MESQUITE SuperLUDist STRUMPACK AXOM CONDUIT GECKO GNUTLS NETCDF MPFR PUMI
POSIXCLOCKS MFEMBacktrace ZLIB)
POSIXCLOCKS MFEMBacktrace ZLIB OCCA RAJA)
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
set(TPL_LIBRARIES "")
set(TPL_INCLUDE_DIRS "")
@@ -327,6 +360,13 @@ set(MFEM_SOURCE_DIRS general linalg mesh fem)
foreach(DIR IN LISTS MFEM_SOURCE_DIRS)
add_subdirectory(${DIR})
endforeach()
if (MFEM_USE_CUDA)
foreach(file IN LISTS SOURCES)
set_property(SOURCE ${file} PROPERTY LANGUAGE CUDA)
endforeach()
endif()
add_subdirectory(config)
set(MASTER_HEADERS
${PROJECT_SOURCE_DIR}/mfem.hpp
@@ -337,6 +377,11 @@ set(CMAKE_INSTALL_RPATH_USE_LINK_PATH ON CACHE BOOL "")
set(CMAKE_INSTALL_RPATH "${_lib_path}" CACHE PATH "")
set(CMAKE_INSTALL_NAME_DIR "${_lib_path}" CACHE PATH "")
set(MFEM_SOURCE_DIR ${CMAKE_CURRENT_SOURCE_DIR} CACHE PATH
"The MFEM source directory" FORCE)
set(MFEM_INSTALL_DIR ${CMAKE_INSTALL_PREFIX} CACHE PATH
"The MFEM install directory" FORCE)
# Declaring the library
add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
@@ -434,12 +479,12 @@ endif()
# Add 'check' target - quick test
if (NOT MFEM_USE_MPI)
add_custom_target(check
${CMAKE_CTEST_COMMAND} -R '^ex1_ser' -C ${CMAKE_CFG_INTDIR}
${CMAKE_CTEST_COMMAND} -R \"^ex1_ser\" -C ${CMAKE_CFG_INTDIR}
USES_TERMINAL)
add_dependencies(check ex1)
else()
add_custom_target(check
${CMAKE_CTEST_COMMAND} -R '^ex1p' -C ${CMAKE_CFG_INTDIR}
${CMAKE_CTEST_COMMAND} -R \"^ex1p\" -C ${CMAKE_CFG_INTDIR}
USES_TERMINAL)
add_dependencies(check ex1p)
endif()
@@ -484,6 +529,13 @@ install(DIRECTORY ${MFEM_SOURCE_DIRS}
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem
FILES_MATCHING PATTERN "*.hpp")
# Install the okl files
if (MFEM_USE_OCCA)
install(DIRECTORY ${MFEM_SOURCE_DIRS}
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem
FILES_MATCHING PATTERN "*.okl")
endif()
# Install ${HEADERS}
# ---
# foreach (HDR ${HEADERS})
+10
View File
@@ -142,6 +142,16 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
+ [`HypreParMatrix`](http://mfem.github.io/doxygen/html/classmfem_1_1HypreParMatrix.html) and [`HypreParVector`](http://mfem.github.io/doxygen/html/classmfem_1_1HypreParVector.html)
+ [`HypreSolver`](http://mfem.github.io/doxygen/html/classmfem_1_1HypreSolver.html) and other [hypre classes](http://mfem.github.io/doxygen/html/hypre_8hpp.html)
- GPU and multi-core CPU support is based on device kernels supporting different
backends (CUDA, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
device/host memory manager.
- The main device-relevant classes and sources are:
+ [`Device`](http://mfem.github.io/doxygen/html/device_8hpp.html)
+ [`MemoryManager`](http://mfem.github.io/doxygen/html/mem_manager_8hpp.html)
+ the [`MFEM_FORALL`](http://mfem.github.io/doxygen/html/forall_8hpp.html) macro
+ the [`cuda.hpp`](http://mfem.github.io/doxygen/html/cuda_8hpp.html) and [`occa.hpp`](http://mfem.github.io/doxygen/html/occa_8hpp.html) files
- The `general/` directory contains C++ classes that serve as utilities for
communication, error handling, arrays, (Boolean) tables, timing, etc.
+30 -13
View File
@@ -13,11 +13,17 @@ of MFEM is a (modern) C++ compiler, such as g++. The parallel version of MFEM
requires an MPI C++ compiler, as well as the following external libraries:
- hypre (a library of high-performance preconditioners)
http://www.llnl.gov/CASC/hypre
https://github.com/hypre-space/hypre
- METIS (a family of multilevel partitioning algorithms)
http://glaros.dtc.umn.edu/gkhome/metis/metis/overview
The hypre dependency can be downloaded as a tarball from GitHub or from the
project webpage https://www.llnl.gov/casc/hypre. For example, the 2.16.0 release
of hypre is available at
https://github.com/hypre-space/hypre/archive/v2.16.0.tar.gz
The METIS dependency can be disabled but that is not generally recommended, see
the option MFEM_USE_METIS.
@@ -48,7 +54,7 @@ following package managers:
- Spack, https://github.com/spack/spack
- OpenHPC, http://openhpc.community
- Homebrew/Science, https://github.com/Homebrew/homebrew-science
- Homebrew/Science, https://github.com/Homebrew/homebrew-science (deprecated)
We also recommend downloading and building the MFEM-based GLVis visualization
tool which can be used to visualize the meshes and solution in MFEM's examples
@@ -60,9 +66,9 @@ Serial build:
make serial -j 4
Parallel build:
(download hypre 2.10.0b and METIS 4 from above URLs)
(download hypre and METIS 4 from above URLs)
(build METIS 4 in ../metis-4.0 relative to mfem/)
(build hypre 2.10.0b in ../hypre-2.10.0b relative to mfem/)
(build hypre in ../hypre relative to mfem/)
make parallel -j 4
CUDA build:
@@ -87,13 +93,19 @@ Serial build:
make -j 4 (assuming "UNIX Makefiles" generator)
Parallel build:
(download hypre 2.10.0b and METIS 4 from above URLs)
(download hypre and METIS 4 from above URLs)
(build METIS 4 in ../metis-4.0 relative to mfem/)
(build hypre 2.10.0b in ../hypre-2.10.0b relative to mfem/)
(build hypre in ../hypre relative to mfem/)
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES
make -j 4
CUDA build:
(this build requires CMake 3.8 or newer)
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_CUDA=YES
make -j 4
Example codes (serial/parallel, depending on the build):
make examples -j 4
@@ -278,6 +290,7 @@ MFEM_THREAD_SAFE = YES/NO
MFEM_USE_LEGACY_OPENMP = YES/NO
Enable (basic) experimental OpenMP support. Requires MFEM_THREAD_SAFE.
This option is deprecated.
MFEM_USE_OPENMP = YES/NO
Enable the OpenMP backend.
@@ -393,7 +406,8 @@ MFEM_USE_PUMI = YES/NO
MFEM_USE_MM = YES/NO
Enables support for the MFEM's memory manager (MM), which is required to
support devices with different memory spaces.
support devices with different memory spaces. This option is required when
CUDA support is enabled, i.e. when MFEM_USE_CUDA=YES.
MFEM_USE_CUDA = YES/NO
Enables support for CUDA devices in MFEM. CUDA is a parallel computing
@@ -406,13 +420,15 @@ MFEM_USE_CUDA = YES/NO
MFEM_USE_RAJA = YES/NO
Enable support for the RAJA performance portability layer in MFEM. RAJA
provides a portable abstraction for loops, supporting different programming
model backends. When using the RAJA CUDA backend, MFEM_USE_MM is required.
model backends. When using RAJA built with CUDA support, CUDA support must be
also enabled in MFEM, i.e. MFEM_USE_CUDA=YES must be set.
MFEM_USE_OCCA = YES/NO
Enables support for the OCCA library in MFEM. OCCA is an open-source library
which aims to make it easy to program different types of devices (e.g. CPU,
GPU, FPGA) by providing an unified API for interacting with JIT-compiled
backends. When using the OCCA CUDA backend, MFEM_USE_MM is required.
backends. In order to use the OCCA CUDA backend, CUDA support must be enabled
in MFEM as well, i.e. MFEM_USE_CUDA=YES must be set.
MFEM_BUILD_TAG = (any value)
An optional tag to characterize the build. Exported to config/config.mk.
@@ -435,7 +451,7 @@ directory and use the string @MFEM_DIR@, e.g. HYPRE_OPT = -I@MFEM_DIR@/../hypre.
The specific libraries and their options are:
- HYPRE, required for the parallel build, i.e. when MFEM_USE_MPI = YES.
URL: http://www.llnl.gov/CASC/hypre
URL: https://github.com/hypre-space/hypre and https://www.llnl.gov/casc/hypre
Options: HYPRE_OPT, HYPRE_LIB.
- METIS, used when MFEM_USE_METIS = YES. If using METIS 5, set
@@ -645,6 +661,8 @@ Configuration variables (CMake)
===============================
See the configuration file config/defaults.cmake for the default settings.
Note: the option MFEM_USE_CUDA requires CMake version 3.8 or newer!
Non-standard CMake variables for compilers:
CXX - If set, overwrite the auto-detected C++ compiler, serial build
MPICXX - If set, overwrite the auto-detected MPI C++ compiler, parallel build
@@ -675,9 +693,6 @@ MFEM_USE_NETCDF
MFEM_USE_MPFR
MFEM_USE_GZSTREAM
MFEM_USE_PUMI
The following GNU make options are not supported with CMake yet:
MFEM_USE_CUDA
MFEM_USE_OCCA
MFEM_USE_RAJA
@@ -728,6 +743,8 @@ The CMake build system adds auto-detection for the following packages/libraries:
- LIBUNWIND
- POSIXCLOCKS
- PUMI
- OCCA
- RAJA
The following built-in CMake packages are also used:
+17 -16
View File
@@ -8,9 +8,9 @@
http://mfem.org
MFEM is a modular parallel C++ library for finite element methods. Its goal is
to enable the research and development of scalable finite element discretization
and solver algorithms through general finite element abstractions, accurate and
flexible visualization, and tight integration with the hypre library.
to enable high-performance scalable finite element discretization research and
application development on a wide variety of platforms, ranging from laptops to
supercomputers.
* For building instructions, see the file INSTALL, or type "make help".
@@ -39,23 +39,24 @@ conforming and non-conforming (AMR) adaptive refinement. Arbitrary element
transformations, allowing for high-order mesh elements with curved boundaries,
are also supported.
MFEM is commonly used as a "finite element to linear algebra translator", since
it can take a problem described in terms of finite element-type objects, and
produce the corresponding linear algebra vectors and sparse matrices. In order
to facilitate this, MFEM uses compressed sparse row (CSR) sparse matrix storage
and includes simple smoothers and Krylov solvers, such as PCG, MINRES and GMRES,
as well as support for sequential sparse direct solvers from the SuiteSparse
When used as a "finite element to linear algebra translator", MFEM can take a
problem described in terms of finite element-type objects, and produce the
corresponding linear algebra vectors and fully or partially assembled operators,
e.g. in the form of global sparse matrices or matrix-free operators. The library
includes simple smoothers and Krylov solvers, such as PCG, MINRES and GMRES, as
well as support for sequential sparse direct solvers from the SuiteSparse
library. Nonlinear solvers (the Newton method), eigensolvers (LOBPCG), and
several explicit and implicit Runge-Kutta time integrators are also available.
MFEM supports MPI-based parallelism throughout the library, and can readily be
used as a scalable unstructured finite element problem generator. MFEM-based
applications require minimal changes to transition from a serial to a
high-performing parallel version of the code, where they can take advantage of
the integrated scalable linear solvers from the hypre library. Comprehensive
support for other external packages, e.g. PETSc and SUNDIALS is also included,
giving access to many additional linear and nonlinear solvers, preconditioners,
time integrators, etc.
used as a scalable unstructured finite element problem generator. As of version
4.0, MFEM offers initial support for GPU acceleration, and programming models,
such as CUDA, OCCA, RAJA and OpenMP. MFEM-based applications require minimal
changes to switch from a serial to a high-performing MPI-parallel version of the
code, where they can take advantage of the integrated linear solvers from the
hypre library. Comprehensive support for other external packages, e.g. PETSc
and SUNDIALS is also included, giving access to many additional linear and
nonlinear solvers, preconditioners, time integrators, etc.
For examples of using MFEM, see the examples/ and miniapps/ directories, as well
as the OpenGL visualization tool GLVis which is available at http://glvis.org.
+23 -4
View File
@@ -74,7 +74,7 @@
IF (NOT COMMAND PRINT_VAR)
FUNCTION(PRINT_VAR VAR_NAME)
MESSAGE("-- " "${VAR_NAME} = '${${VAR_NAME}}'")
MESSAGE(STATUS "${VAR_NAME} = '${${VAR_NAME}}'")
ENDFUNCTION()
ENDIF()
@@ -166,14 +166,14 @@ IF (USE_XSDK_DEFAULTS)
ENDIF()
XSDK_HANDLE_LANG_DEFAULTS(Fortran FC "FFLAGS;FCFLAGS")
ENDIF()
# Set XSDK defaults for other CMake variables
IF ("${BUILD_SHARED_LIBS}" STREQUAL "")
MESSAGE("-- " "XSDK: Setting default BUILD_SHARED_LIBS=TRUE")
SET(BUILD_SHARED_LIBS TRUE CACHE BOOL "Set by default in XSDK mode")
ENDIF()
IF ("${CMAKE_BUILD_TYPE}" STREQUAL "")
MESSAGE("-- " "XSDK: Setting default CMAKE_BUILD_TYPE=DEBUG")
SET(CMAKE_BUILD_TYPE DEBUG CACHE STRING "Set by default in XSDK mode")
@@ -181,6 +181,13 @@ IF (USE_XSDK_DEFAULTS)
ENDIF()
##################################################################################
#
# MFEM-specific additions: set TPL MFEM_USE_* defaults
#
##################################################################################
IF (DEFINED TPL_ENABLE_MPI)
SET(MFEM_USE_MPI ${TPL_ENABLE_MPI} CACHE BOOL "Enable MPI parallel build" FORCE)
ENDIF()
@@ -252,3 +259,15 @@ ENDIF()
IF (DEFINED TPL_ENABLE_PUMI)
SET(MFEM_USE_PUMI ${TPL_ENABLE_PUMI} CACHE BOOL "Enable PUMI" FORCE)
ENDIF()
IF (DEFINED TPL_ENABLE_CUDA)
SET(MFEM_USE_CUDA ${TPL_ENABLE_CUDA} CACHE BOOL "Enable CUDA" FORCE)
ENDIF()
IF (DEFINED TPL_ENABLE_OCCA)
SET(MFEM_USE_OCCA ${TPL_ENABLE_OCCA} CACHE BOOL "Enable OCCA" FORCE)
ENDIF()
IF (DEFINED TPL_ENABLE_RAJA)
SET(MFEM_USE_RAJA ${TPL_ENABLE_RAJA} CACHE BOOL "Enable RAJA" FORCE)
ENDIF()
+4
View File
@@ -41,6 +41,10 @@ set(MFEM_USE_MPFR @MFEM_USE_MPFR@)
set(MFEM_USE_SIDRE @MFEM_USE_SIDRE@)
set(MFEM_USE_CONDUIT @MFEM_USE_CONDUIT@)
set(MFEM_USE_PUMI @MFEM_USE_PUMI@)
set(MFEM_USE_MM @MFEM_USE_MM@)
set(MFEM_USE_CUDA @MFEM_USE_CUDA@)
set(MFEM_USE_OCCA @MFEM_USE_OCCA@)
set(MFEM_USE_RAJA @MFEM_USE_RAJA@)
set(MFEM_CXX_COMPILER "@CMAKE_CXX_COMPILER@")
set(MFEM_CXX_FLAGS "@CMAKE_CXX_FLAGS@")
+19
View File
@@ -30,6 +30,12 @@
#define MFEM_VERSION_MINOR (((MFEM_VERSION)/100)%100)
#define MFEM_VERSION_PATCH ((MFEM_VERSION)%100)
// MFEM source directory.
#define MFEM_SOURCE_DIR "@MFEM_SOURCE_DIR@"
// MFEM install directory.
#define MFEM_INSTALL_DIR "@MFEM_INSTALL_DIR@"
// Description of the git commit used to build MFEM.
#cmakedefine MFEM_GIT_STRING "@MFEM_GIT_STRING@"
@@ -104,6 +110,19 @@
// Enable MFEM functionality based on the PUMI library
#cmakedefine MFEM_USE_PUMI
// Build the GPU/CUDA-enabled version of the MFEM library.
// Requires a CUDA compiler (nvcc).
#cmakedefine MFEM_USE_CUDA
// Enable MFEM functionality based on the RAJA library
#cmakedefine MFEM_USE_RAJA
// Enable MFEM functionality based on the OCCA library
#cmakedefine MFEM_USE_OCCA
// Enable MFEM's internal Memory Manager (needed e.g. for MFEM_USE_CUDA)
#cmakedefine MFEM_USE_MM
// Which library functions to use in class StopWatch for measuring time.
// For a list of the available options, see INSTALL.
// If not defined, an option is selected automatically.
+1 -3
View File
@@ -18,6 +18,4 @@ include(MfemCmakeUtilities)
# Note: components are enabled based on the find_package() parameters.
mfem_find_package(Axom AXOM AXOM_DIR "include" "" "lib" ""
"Paths to headers required by Axom." "Libraries required by Axom."
ADD_COMPONENT Sidre "include" sidre/sidre.hpp "lib" sidre
ADD_COMPONENT SLIC "include" slic/slic.hpp "lib" slic
ADD_COMPONENT axom_utils "include" axom_utils/Utilities.hpp "lib" axom_utils)
ADD_COMPONENT Axom "include" axom/config.hpp "lib" axom)
+19
View File
@@ -0,0 +1,19 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Defines the following variables:
# - OCCA_FOUND
# - OCCA_LIBRARIES
# - OCCA_INCLUDE_DIRS
include(MfemCmakeUtilities)
mfem_find_package(OCCA OCCA OCCA_DIR "include" "occa.hpp" "lib" "occa"
"Paths to headers required by OCCA." "Libraries required by OCCA.")
+30
View File
@@ -0,0 +1,30 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Defines the following variables:
# - RAJA_FOUND
# - RAJA_LIBRARIES
# - RAJA_INCLUDE_DIRS
include(MfemCmakeUtilities)
mfem_find_package(RAJA RAJA RAJA_DIR "include" "RAJA/RAJA.hpp" "lib" "RAJA"
"Paths to headers required by RAJA." "Libraries required by RAJA.")
if (NOT RAJA_CONFIG_CMAKE)
set(RAJA_CONFIG_CMAKE "${RAJA_DIR}/share/raja/cmake/raja-config.cmake")
endif()
if (EXISTS "${RAJA_CONFIG_CMAKE}")
include("${RAJA_CONFIG_CMAKE}")
if (ENABLE_CUDA AND NOT MFEM_USE_CUDA)
message(FATAL_ERROR
"RAJA is built with CUDA: MFEM_USE_CUDA=YES is required")
endif()
endif()
@@ -232,10 +232,12 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
# If we have the TPL_ versions of _INCLUDE_DIRS and _LIBRARIES then set the
# standard ${Prefix} versions
if (TPL_${Prefix}_INCLUDE_DIRS)
set(${Prefix}_INCLUDE_DIRS ${TPL_${Prefix}_INCLUDE_DIRS} CACHE STRING "TPL_${Prefix}_INCLUDE_DIRS was found." FORCE)
set(${Prefix}_INCLUDE_DIRS ${TPL_${Prefix}_INCLUDE_DIRS} CACHE STRING
"TPL_${Prefix}_INCLUDE_DIRS was found." FORCE)
endif()
if (TPL_${Prefix}_LIBRARIES)
set(${Prefix}_LIBRARIES ${TPL_${Prefix}_LIBRARIES} CACHE STRING "TPL_${Prefix}_LIBRARIES was found." FORCE)
set(${Prefix}_LIBRARIES ${TPL_${Prefix}_LIBRARIES} CACHE STRING
"TPL_${Prefix}_LIBRARIES was found." FORCE)
endif()
# Quick return
@@ -718,7 +720,8 @@ function(mfem_export_mk_files)
MFEM_USE_MEMALLOC MFEM_USE_SUNDIALS MFEM_USE_MESQUITE MFEM_USE_SUITESPARSE
MFEM_USE_SUPERLU MFEM_USE_STRUMPACK MFEM_USE_GECKO MFEM_USE_GNUTLS
MFEM_USE_NETCDF MFEM_USE_PETSC MFEM_USE_MPFR MFEM_USE_SIDRE
MFEM_USE_CONDUIT MFEM_USE_PUMI)
MFEM_USE_CONDUIT MFEM_USE_PUMI MFEM_USE_MM MFEM_USE_CUDA MFEM_USE_OCCA
MFEM_USE_RAJA)
foreach(var ${CONFIG_MK_BOOL_VARS})
if (${var})
set(${var} YES)
@@ -726,6 +729,7 @@ function(mfem_export_mk_files)
set(${var} NO)
endif()
endforeach()
# TODO: Add support for MFEM_USE_CUDA=YES
set(MFEM_CXX ${CMAKE_CXX_COMPILER})
set(MFEM_CPPFLAGS "")
string(STRIP "${CMAKE_CXX_FLAGS_${BUILD_TYPE}} ${CMAKE_CXX_FLAGS}"
+5
View File
@@ -56,4 +56,9 @@
#endif
#endif // MFEM_USE_MPI not defined
// CUDA requires the memory manager
#if defined(MFEM_USE_CUDA) && !defined(MFEM_USE_MM)
#error Building with CUDA (MFEM_USE_CUDA=YES) requires MFEM_USE_MM=YES
#endif
#endif // MFEM_CONFIG_HPP
+11 -1
View File
@@ -42,6 +42,10 @@ option(MFEM_USE_MPFR "Enable MPFR usage." OFF)
option(MFEM_USE_SIDRE "Enable Axom/Sidre usage" OFF)
option(MFEM_USE_CONDUIT "Enable Conduit usage" OFF)
option(MFEM_USE_PUMI "Enable PUMI" OFF)
option(MFEM_USE_MM "Enable MFEM's memory manager" OFF)
option(MFEM_USE_CUDA "Enable CUDA" OFF)
option(MFEM_USE_OCCA "Enable OCCA" OFF)
option(MFEM_USE_RAJA "Enable RAJA" OFF)
set(MFEM_MPI_NP 4 CACHE STRING "Number of processes used for MPI tests")
@@ -59,13 +63,16 @@ option(MFEM_ENABLE_MINIAPPS "Build all of the miniapps" OFF)
# set(CXX g++)
# set(MPICXX mpicxx)
# Set the target CUDA architecture
set(CUDA_ARCH "sm_60" CACHE STRING "Target CUDA architecture.")
set(MFEM_DIR ${CMAKE_CURRENT_SOURCE_DIR})
# The *_DIR paths below will be the first place searched for the corresponding
# headers and library. If these fail, then standard cmake search is performed.
# Note: if the variables are already in the cache, they are not overwritten.
set(HYPRE_DIR "${MFEM_DIR}/../hypre-2.10.0b/src/hypre" CACHE PATH
set(HYPRE_DIR "${MFEM_DIR}/../hypre/src/hypre" CACHE PATH
"Path to the hypre library.")
# If hypre was compiled to depend on BLAS and LAPACK:
# set(HYPRE_REQUIRED_PACKAGES "BLAS" "LAPACK" CACHE STRING
@@ -154,6 +161,9 @@ set(Axom_REQUIRED_PACKAGES "Conduit/relay" CACHE STRING
set(PUMI_DIR "${MFEM_DIR}/../pumi-2.1.0" CACHE STRING
"Directory where PUMI is installed")
set(OCCA_DIR "${MFEM_DIR}/../occa" CACHE PATH "Path to OCCA")
set(RAJA_DIR "${MFEM_DIR}/../raja" CACHE PATH "Path to RAJA")
set(BLAS_INCLUDE_DIRS "" CACHE STRING "Path to BLAS headers.")
set(BLAS_LIBRARIES "" CACHE STRING "The BLAS library.")
set(LAPACK_INCLUDE_DIRS "" CACHE STRING "Path to LAPACK headers.")
+6 -8
View File
@@ -136,7 +136,7 @@ LIBUNWIND_OPT = -g
LIBUNWIND_LIB = $(if $(NOTMAC),-lunwind -ldl,)
# HYPRE library configuration (needed to build the parallel version)
HYPRE_DIR = @MFEM_DIR@/../hypre-2.10.0b/src/hypre
HYPRE_DIR = @MFEM_DIR@/../hypre/src/hypre
HYPRE_OPT = -I$(HYPRE_DIR)/include
HYPRE_LIB = -L$(HYPRE_DIR)/lib -lHYPRE
@@ -291,7 +291,7 @@ SIDRE_LIB = \
-Wl,-rpath,$(SIDRE_DIR)/lib -L$(SIDRE_DIR)/lib \
-Wl,-rpath,$(CONDUIT_DIR)/lib -L$(CONDUIT_DIR)/lib \
-Wl,-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib \
-lsidre -lslic -laxom_utils -lconduit -lconduit_relay -lhdf5 $(ZLIB_LIB) -ldl
-laxom -lconduit -lconduit_relay -lhdf5 $(ZLIB_LIB) -ldl
# PUMI
# Note that PUMI_DIR is needed -- it is used to check for gmi_sim.h
@@ -300,19 +300,17 @@ PUMI_OPT = -I$(PUMI_DIR)/include
PUMI_LIB = -L$(PUMI_DIR)/lib -lpumi -lcrv -lma -lmds -lapf -lpcu -lgmi -lparma\
-llion -lmth -lapf_zoltan -lspr
# CUDA library configuration. Since we compile and link with nvcc (when CUDA is
# enabled) we only need to explicitly link with the CUDA driver, libcuda.*,
# which is usually in a system path.
# CUDA library configuration (currently not needed)
CUDA_OPT =
CUDA_LIB = $(if $(NOTMAC),,-L/usr/local/cuda/lib) -lcuda
CUDA_LIB =
# OCCA library configuration
OCCA_DIR ?= @MFEM_DIR@/../occa
OCCA_DIR = @MFEM_DIR@/../occa
OCCA_OPT = -I$(OCCA_DIR)/include
OCCA_LIB = $(XLINKER)-rpath,$(OCCA_DIR)/lib -L$(OCCA_DIR)/lib -locca
# RAJA library configuration
RAJA_DIR ?= @MFEM_DIR@/../raja
RAJA_DIR = @MFEM_DIR@/../raja
RAJA_OPT = -I$(RAJA_DIR)/include
ifdef CUB_DIR
RAJA_OPT += -I$(CUB_DIR)
+20 -1
View File
@@ -18,6 +18,8 @@ run_prefix=""
run_vg="valgrind --leak-check=full --show-reachable=yes --track-origins=yes"
run_suffix="-no-vis"
skip_gen_meshes="yes"
# filter-out device runs ("no") or non-device runs ("yes"):
device_runs="no"
cur_dir="${PWD}"
mfem_dir="$(cd "$(dirname "$0")"/.. && pwd)"
mfem_build_dir=""
@@ -148,6 +150,11 @@ function extract_sample_runs()
if [ "$skip_gen_meshes" == "yes" ]; then
runs=`printf "%s" "$runs" | grep -v ".* -m .*\.gen"`
fi
if [ "$device_runs" == "yes" ]; then
runs=`printf "%s" "$runs" | grep ".* -d .*"`
else
runs=`printf "%s" "$runs" | grep -v ".* -d .*"`
fi
IFS=$'\n'
runs=(${runs})
IFS="${old_IFS}"
@@ -169,6 +176,9 @@ function help_message()
-g <dir> <pattern>
Specify explicitly a group (dir + file pattern) to run; This
option can be used multiple times to define multiple groups
-dev configure only sample runs using devices.
To test with a parallel build, the parallel (-p|-par) option
should be set first on the command line.
-v Enable valgrind
-o <dir> [${output_dir:-"<empty>: output goes to stdout"}]
If not empty, save output to files inside <dir>
@@ -253,7 +263,7 @@ case "$1" in
-h|-help)
opt_help="yes"
;;
-p|-parallel)
-p|-par)
mfem_config="MFEM_USE_MPI=YES MFEM_DEBUG=NO"
;;
-g)
@@ -264,6 +274,11 @@ case "$1" in
groups=("${groups[@]}" "${test_group}")
shift 2
;;
-dev)
device_runs="yes"
mfem_config+=" MFEM_USE_CUDA=YES MFEM_USE_MM=YES \
MFEM_USE_OCCA=YES MFEM_USE_RAJA=YES MFEM_USE_OPENMP=YES"
;;
-v)
valgrind="yes"
;;
@@ -294,6 +309,10 @@ case "$1" in
-n)
run_prefix="echo"
;;
-*)
echo "unknown option: '$1'"
exit 1
;;
*=*)
eval $1
;;
+4
View File
@@ -35,6 +35,10 @@ namespace mfem {
* - HypreParMatrix and HypreParVector
* - HypreSolver and other \link hypre.hpp hypre classes\endlink
*
* <H3>Main GPU classes</H3>
* - Device
* - MemoryManager
*
* <H3>Example codes</H3>
* - <a class="el" href="examples_2ex1_8cpp_source.html">Example 1</a>: nodal H1 FEM for the Laplace problem
* - <a class="el" href="examples_2ex1p_8cpp_source.html">Example 1p</a>: parallel nodal H1 FEM for the Laplace problem
+6 -6
View File
@@ -26,12 +26,12 @@
// ex1 -m ../data/mobius-strip.mesh -o -1 -sc
//
// Device sample runs:
// > ex1 -pa -d cuda
// > ex1 -pa -d raja-cuda
// > ex1 -pa -d occa-cuda
// > ex1 -pa -d raja-omp
// > ex1 -pa -d occa-omp
// > ex1 -m ../data/beam-hex.mesh -pa -d cuda
// ex1 -pa -d cuda
// ex1 -pa -d raja-cuda
// ex1 -pa -d occa-cuda
// ex1 -pa -d raja-omp
// ex1 -pa -d occa-omp
// ex1 -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Laplace problem
+3 -3
View File
@@ -26,9 +26,9 @@
// mpirun -np 4 ex1p -m ../data/mobius-strip.mesh -o -1 -sc
//
// Device sample runs:
// > mpirun -np 4 ex1p -pa -d cuda
// > mpirun -np 4 ex1p -pa -d occa-cuda
// > mpirun -np 4 ex1p -pa -d raja-omp
// mpirun -np 4 ex1p -pa -d cuda
// mpirun -np 4 ex1p -pa -d occa-cuda
// mpirun -np 4 ex1p -pa -d raja-omp
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Laplace problem
+3 -3
View File
@@ -16,9 +16,9 @@
// ex6 -m ../data/amr-quad.mesh
//
// Device sample runs:
// > ex6 -pa -d cuda
// > ex6 -pa -d occa-cuda
// > ex6 -pa -d raja-omp
// ex6 -pa -d cuda
// ex6 -pa -d occa-cuda
// ex6 -pa -d raja-omp
//
// Description: This is a version of Example 1 with a simple adaptive mesh
// refinement loop. The problem being solved is again the Laplace
+3 -3
View File
@@ -16,9 +16,9 @@
// mpirun -np 4 ex6p -m ../data/amr-quad.mesh
//
// Device sample runs:
// > mpirun -np 4 ex6p -pa -d cuda
// > mpirun -np 4 ex6p -pa -d occa-cuda
// > mpirun -np 4 ex6p -pa -d raja-omp
// mpirun -np 4 ex6p -pa -d cuda
// mpirun -np 4 ex6p -pa -d occa-cuda
// mpirun -np 4 ex6p -pa -d raja-omp
//
// Description: This is a version of Example 1 with a simple adaptive mesh
// refinement loop. The problem being solved is again the Laplace
+6 -2
View File
@@ -588,14 +588,18 @@ void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
Vector &b, OperatorHandle &A, Vector &X,
Vector &B, int copy_interior)
{
const SparseMatrix *P = fes->GetConformingProlongation();
if (ext)
{
if (P != NULL && assembly != AssemblyLevel::FULL && Device::IsEnabled())
{
P->BuildTranspose();
}
ext->FormLinearSystem(ess_tdof_list, x, b, A, X, B, copy_interior);
return;
}
const SparseMatrix *P = fes->GetConformingProlongation();
FormSystemMatrix(ess_tdof_list, A);
// Transform the system and perform the elimination in B, based on the
+6 -8
View File
@@ -16,9 +16,7 @@
#include "fem.hpp"
#ifdef MFEM_USE_MPI
#include <sidre/IOManager.hpp>
#endif
#include <axom/sidre.hpp>
#include <string>
#include <iomanip> // for setw, setfill
@@ -204,10 +202,10 @@ SidreDataCollection::get_file_path(const std::string &filename) const
axom::sidre::View *
SidreDataCollection::AllocNamedBuffer(const std::string& buffer_name,
axom::sidre::SidreLength sz,
axom::sidre::IndexType sz,
axom::sidre::TypeID type)
{
sz = std::max(sz, sidre::SidreLength(0));
sz = std::max(sz, sidre::IndexType(0));
sidre::Group *f = named_buffers_grp();
sidre::View *v = NULL;
@@ -825,7 +823,7 @@ void SidreDataCollection::Save(const std::string& filename,
void SidreDataCollection::
addScalarBasedGridFunction(const std::string &field_name, GridFunction *gf,
const std::string &buffer_name,
axom::sidre::SidreLength offset)
axom::sidre::IndexType offset)
{
sidre::Group* grp = m_bp_grp->getGroup("fields/" + field_name);
MFEM_ASSERT(grp != NULL, "field " << field_name << " does not exist");
@@ -888,7 +886,7 @@ addScalarBasedGridFunction(const std::string &field_name, GridFunction *gf,
void SidreDataCollection::
addVectorBasedGridFunction(const std::string& field_name, GridFunction *gf,
const std::string &buffer_name,
axom::sidre::SidreLength offset)
axom::sidre::IndexType offset)
{
sidre::Group* grp = m_bp_grp->getGroup("fields/" + field_name);
MFEM_ASSERT(grp != NULL, "field " << field_name << " does not exist");
@@ -1013,7 +1011,7 @@ DeregisterFieldInBPIndex(const std::string& field_name)
void SidreDataCollection::RegisterField(const std::string &field_name,
GridFunction *gf,
const std::string &buffer_name,
axom::sidre::SidreLength offset)
axom::sidre::IndexType offset)
{
if ( field_name.empty() || buffer_name.empty() ||
gf == NULL || gf->FESpace() == NULL )
+5 -5
View File
@@ -25,7 +25,7 @@
# pragma GCC diagnostic ignored "-Wpedantic"
# endif
#endif
#include <sidre/sidre.hpp>
#include <axom/sidre.hpp>
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
# pragma GCC diagnostic pop
#endif
@@ -246,7 +246,7 @@ public:
*/
void RegisterField(const std::string &field_name, GridFunction *gf,
const std::string &buffer_name,
axom::sidre::SidreLength offset);
axom::sidre::IndexType offset);
/// Registers an attribute field in the Sidre DataStore
/** The registration process is similar to that of RegisterField()
@@ -385,7 +385,7 @@ public:
*/
axom::sidre::View *
AllocNamedBuffer(const std::string& buffer_name,
axom::sidre::SidreLength sz,
axom::sidre::IndexType sz,
axom::sidre::TypeID type =
axom::sidre::DOUBLE_ID);
@@ -469,7 +469,7 @@ private:
void addScalarBasedGridFunction(const std::string& field_name,
GridFunction* gf,
const std::string &buffer_name,
axom::sidre::SidreLength offset);
axom::sidre::IndexType offset);
/**
* \brief A private helper function to set up the views associated with the
@@ -483,7 +483,7 @@ private:
void addVectorBasedGridFunction(const std::string& field_name,
GridFunction* gf,
const std::string &buffer_name,
axom::sidre::SidreLength offset);
axom::sidre::IndexType offset);
/** @brief A private helper function to set up the Views associated with
attribute field named @a field_name */
+23 -39
View File
@@ -10,17 +10,27 @@
// Software Foundation) version 2.1 dated February 1999.
#include "cuda.hpp"
#include "globals.hpp"
namespace mfem
{
#ifdef MFEM_USE_CUDA
void mfem_cuda_error(cudaError_t err, const char *expr, const char *func,
const char *file, int line)
{
mfem::err << "CUDA error: (" << expr << ") failed with error:\n --> "
<< cudaGetErrorString(err)
<< "\n ... in function: " << func
<< "\n ... in file: " << file << ':' << line << '\n';
mfem_error();
}
#endif
void* CuMemAlloc(void** dptr, size_t bytes)
{
#ifdef MFEM_USE_CUDA
if (CUDA_SUCCESS != ::cuMemAlloc((CUdeviceptr*)dptr, bytes))
{
mfem_error("Error in CuMemAlloc");
}
MFEM_CUDA_CHECK(cudaMalloc(dptr, bytes));
#endif
return *dptr;
}
@@ -28,10 +38,7 @@ void* CuMemAlloc(void** dptr, size_t bytes)
void* CuMemFree(void *dptr)
{
#ifdef MFEM_USE_CUDA
if (CUDA_SUCCESS != ::cuMemFree((CUdeviceptr)dptr))
{
mfem_error("Error in CuMemFree");
}
MFEM_CUDA_CHECK(cudaFree(dptr));
#endif
return dptr;
}
@@ -39,22 +46,15 @@ void* CuMemFree(void *dptr)
void* CuMemcpyHtoD(void* dst, const void* src, size_t bytes)
{
#ifdef MFEM_USE_CUDA
if (CUDA_SUCCESS != ::cuMemcpyHtoD((CUdeviceptr)dst, src, bytes))
{
mfem_error("Error in CuMemcpyHtoD");
}
MFEM_CUDA_CHECK(cudaMemcpy(dst, src, bytes, cudaMemcpyHostToDevice));
#endif
return dst;
}
void* CuMemcpyHtoDAsync(void* dst, const void* src, size_t bytes, void *s)
void* CuMemcpyHtoDAsync(void* dst, const void* src, size_t bytes)
{
#ifdef MFEM_USE_CUDA
if (CUDA_SUCCESS !=
::cuMemcpyHtoDAsync((CUdeviceptr)dst, src, bytes, (CUstream)s))
{
mfem_error("Error in CuMemcpyHtoDAsync");
}
MFEM_CUDA_CHECK(cudaMemcpyAsync(dst, src, bytes, cudaMemcpyHostToDevice));
#endif
return dst;
}
@@ -62,24 +62,15 @@ void* CuMemcpyHtoDAsync(void* dst, const void* src, size_t bytes, void *s)
void* CuMemcpyDtoD(void* dst, void* src, size_t bytes)
{
#ifdef MFEM_USE_CUDA
if (CUDA_SUCCESS !=
::cuMemcpyDtoD((CUdeviceptr)dst, (CUdeviceptr)src, bytes))
{
mfem_error("Error in CuMemcpyDtoD");
}
MFEM_CUDA_CHECK(cudaMemcpy(dst, src, bytes, cudaMemcpyDeviceToDevice));
#endif
return dst;
}
void* CuMemcpyDtoDAsync(void* dst, void* src, size_t bytes, void *s)
void* CuMemcpyDtoDAsync(void* dst, void* src, size_t bytes)
{
#ifdef MFEM_USE_CUDA
if (CUDA_SUCCESS !=
::cuMemcpyDtoDAsync((CUdeviceptr)dst, (CUdeviceptr)src,
bytes, (CUstream)s))
{
mfem_error("Error in CuMemcpyDtoDAsync");
}
MFEM_CUDA_CHECK(cudaMemcpyAsync(dst, src, bytes, cudaMemcpyDeviceToDevice));
#endif
return dst;
}
@@ -87,10 +78,7 @@ void* CuMemcpyDtoDAsync(void* dst, void* src, size_t bytes, void *s)
void* CuMemcpyDtoH(void *dst, void *src, size_t bytes)
{
#ifdef MFEM_USE_CUDA
if (CUDA_SUCCESS != ::cuMemcpyDtoH(dst, (CUdeviceptr)src, bytes))
{
mfem_error("Error in CuMemcpyDtoH");
}
MFEM_CUDA_CHECK(cudaMemcpy(dst, src, bytes, cudaMemcpyDeviceToHost));
#endif
return dst;
}
@@ -98,11 +86,7 @@ void* CuMemcpyDtoH(void *dst, void *src, size_t bytes)
void* CuMemcpyDtoHAsync(void* dst, void* src, size_t bytes, void *s)
{
#ifdef MFEM_USE_CUDA
if (CUDA_SUCCESS !=
::cuMemcpyDtoHAsync(dst, (CUdeviceptr)src, bytes, (CUstream)s))
{
mfem_error("Error in CuMemcpyDtoHAsync");
}
MFEM_CUDA_CHECK(cudaMemcpyAsync(dst, src, bytes, cudaMemcpyDeviceToHost));
#endif
return dst;
}
+12 -69
View File
@@ -26,89 +26,33 @@
#ifdef MFEM_USE_CUDA
#define MFEM_ATTR_DEVICE __device__
#define MFEM_ATTR_HOST_DEVICE __host__ __device__
// Define the CUDA debug macros:
// - MFEM_CUDA_CHECK_DRV(x) where 'x' returns/is type 'CUresult'
// - MFEM_CUDA_CHECK_RT(x) where 'x' returns/is type 'cudaError_t'
#ifdef MFEM_DEBUG
#define MFEM_CUDA_CHECK_DRV(x) \
do \
{ \
CUresult err = (x); \
if (err != CUDA_SUCCESS) \
{ \
const char *error_string; \
cuGetErrorString(err, &error_string); \
_MFEM_MESSAGE("CUDA error: (" << #x \
<< ") failed with error:\n --> " \
<< error_string, 0); \
} \
} \
while (0)
#define MFEM_CUDA_CHECK_RT(x) \
// Define a CUDA error check macro, MFEM_CUDA_CHECK(x), where x returns/is of
// type 'cudaError_t'. This macro evaluates 'x' and raises an error if the
// result is not cudaSuccess.
#define MFEM_CUDA_CHECK(x) \
do \
{ \
cudaError_t err = (x); \
if (err != cudaSuccess) \
{ \
_MFEM_MESSAGE("CUDA error: (" << #x \
<< ") failed with error:\n --> " \
<< cudaGetErrorString(err), 0); \
mfem_cuda_error(err, #x, _MFEM_FUNC_NAME, __FILE__, __LINE__); \
} \
} \
while (0)
#else
#define MFEM_CUDA_CHECK_DRV(x) x
#define MFEM_CUDA_CHECK_RT(x) x
#endif
#else // MFEM_USE_CUDA
#define MFEM_ATTR_DEVICE
#define MFEM_ATTR_HOST_DEVICE
typedef int CUdevice;
typedef int CUcontext;
typedef void* CUstream;
#endif // MFEM_USE_CUDA
namespace mfem
{
// Define 'atomicAdd' function.
#ifdef __CUDA_ARCH__
#if __CUDA_ARCH__ < 600
static __device__ inline double atomicAdd(double* address, double val)
{
unsigned long long int* address_as_ull = (unsigned long long int*)address;
unsigned long long int old = *address_as_ull, assumed;
do
{
assumed = old;
old =
atomicCAS(address_as_ull, assumed,
__double_as_longlong(val +
__longlong_as_double(assumed)));
// Note: uses integer comparison to avoid hang in case of NaN
// (since NaN != NaN)
}
while (assumed != old);
return __longlong_as_double(old);
}
#endif // __CUDA_ARCH__ < 600
template<typename T> MFEM_ATTR_DEVICE
inline T AtomicAdd(T volatile *address, T val)
{
return atomicAdd((T *)address, val);
}
#else // __CUDA_ARCH__
template<typename T> inline T AtomicAdd(T volatile *address, T val)
{
#ifdef MFEM_USE_OPENMP
#pragma omp atomic
#ifdef MFEM_USE_CUDA
// Function used by the macro MFEM_CUDA_CHECK.
void mfem_cuda_error(cudaError_t err, const char *expr, const char *func,
const char *file, int line);
#endif
*address += val;
return *address;
}
#endif // __CUDA_ARCH__
/// Allocates device memory
void* CuMemAlloc(void **d_ptr, size_t bytes);
@@ -120,20 +64,19 @@ void* CuMemFree(void *d_ptr);
void* CuMemcpyHtoD(void *d_dst, const void *h_src, size_t bytes);
/// Copies memory from Host to Device
void* CuMemcpyHtoDAsync(void *d_dst, const void *h_src,
size_t bytes, void *stream);
void* CuMemcpyHtoDAsync(void *d_dst, const void *h_src, size_t bytes);
/// Copies memory from Device to Device
void* CuMemcpyDtoD(void *d_dst, void *d_src, size_t bytes);
/// Copies memory from Device to Device
void* CuMemcpyDtoDAsync(void *d_dst, void *d_src, size_t bytes, void *stream);
void* CuMemcpyDtoDAsync(void *d_dst, void *d_src, size_t bytes);
/// Copies memory from Device to Host
void* CuMemcpyDtoH(void *h_dst, void *d_src, size_t bytes);
/// Copies memory from Device to Host
void* CuMemcpyDtoHAsync(void *h_dst, void *d_src, size_t bytes, void *stream);
void* CuMemcpyDtoHAsync(void *h_dst, void *d_src, size_t bytes);
} // namespace mfem
+8 -27
View File
@@ -24,9 +24,6 @@ namespace mfem
namespace internal
{
CUstream *cuStream = NULL;
static CUdevice cuDevice;
static CUcontext cuContext;
OccaDevice occaDevice;
// Backends listed by priority, high to low:
@@ -100,14 +97,9 @@ void Device::Print(std::ostream &out)
#ifdef MFEM_USE_CUDA
static void DeviceSetup(const int dev, int &ngpu)
{
cudaGetDeviceCount(&ngpu);
MFEM_VERIFY(ngpu>0, "No CUDA device found!");
cuInit(0);
cuDeviceGet(&internal::cuDevice, dev);
cuCtxCreate(&internal::cuContext, CU_CTX_SCHED_AUTO, internal::cuDevice);
internal::cuStream = new CUstream;
MFEM_VERIFY(internal::cuStream, "CUDA stream could not be created!");
cuStreamCreate(internal::cuStream, CU_STREAM_DEFAULT);
MFEM_CUDA_CHECK(cudaGetDeviceCount(&ngpu));
MFEM_VERIFY(ngpu > 0, "No CUDA device found!");
MFEM_CUDA_CHECK(cudaSetDevice(dev));
}
#endif
@@ -125,7 +117,7 @@ static void RajaDeviceSetup(const int dev, int &ngpu)
#endif
}
static void OccaDeviceSetup(CUdevice cu_dev, CUcontext cu_ctx)
static void OccaDeviceSetup(const int dev)
{
#ifdef MFEM_USE_OCCA
const int cpu = Device::Allows(Backend::OCCA_CPU);
@@ -138,7 +130,8 @@ static void OccaDeviceSetup(CUdevice cu_dev, CUcontext cu_ctx)
if (cuda)
{
#if OCCA_CUDA_ENABLED
internal::occaDevice = occa::cuda::wrapDevice(cu_dev, cu_ctx);
std::string mode("mode: 'CUDA', device_id : ");
internal::occaDevice.setup(mode.append(1,'0'+dev));
#else
MFEM_ABORT("the OCCA CUDA backend requires OCCA built with CUDA!");
#endif
@@ -197,22 +190,10 @@ void Device::Setup(const int device)
"the OpenMP and RAJA OpenMP backends require MFEM built with"
" MFEM_USE_OPENMP=YES");
#endif
// The check for MFEM_USE_OCCA is in the function OccaDeviceSetup().
// We initialize CUDA and/or RAJA_CUDA first so OccaDeviceSetup() can reuse
// the same initialized cuDevice and cuContext objects when OCCA_CUDA is
// enabled.
if (Allows(Backend::CUDA)) { CudaDeviceSetup(dev, ngpu); }
if (Allows(Backend::RAJA_CUDA)) { RajaDeviceSetup(dev, ngpu); }
if (Allows(Backend::OCCA_MASK))
{
OccaDeviceSetup(internal::cuDevice, internal::cuContext);
}
}
Device::~Device()
{
delete internal::cuStream;
// The check for MFEM_USE_OCCA is in the function OccaDeviceSetup().
if (Allows(Backend::OCCA_MASK)) { OccaDeviceSetup(dev); }
}
} // mfem
-2
View File
@@ -181,8 +181,6 @@ public:
Backend::*_MASK, or combinations of those. */
static inline bool Allows(unsigned long b_mask)
{ return Get().allowed_backends & b_mask; }
~Device();
};
} // mfem
+4 -2
View File
@@ -22,6 +22,9 @@
#ifdef MFEM_USE_RAJA
#include "RAJA/RAJA.hpp"
#if defined(RAJA_ENABLE_CUDA) && !defined(MFEM_USE_CUDA)
#error When RAJA is built with CUDA, MFEM_USE_CUDA=YES is required
#endif
#endif
namespace mfem
@@ -106,8 +109,7 @@ void CuWrap(const int N, DBODY &&d_body)
if (N==0) { return; }
const int GRID = (N+BLOCKS-1)/BLOCKS;
CuKernel<<<GRID,BLOCKS>>>(N,d_body);
const cudaError_t last = cudaGetLastError();
MFEM_VERIFY(last == cudaSuccess, cudaGetErrorString(last));
MFEM_CUDA_CHECK(cudaGetLastError());
}
#else // MFEM_USE_CUDA
+1 -2
View File
@@ -312,7 +312,6 @@ void MemoryManager::Pull(const void *ptr, const std::size_t bytes)
{ mfem_error("Unknown pointer to pull from!"); }
}
namespace internal { extern CUstream *cuStream; }
void* MemoryManager::Memcpy(void *dst, const void *src,
const std::size_t bytes, const bool async)
{
@@ -322,7 +321,7 @@ void* MemoryManager::Memcpy(void *dst, const void *src,
const bool run_on_host = !Device::Allows(Backend::DEVICE_MASK);
if (run_on_host) { return std::memcpy(dst, src, bytes); }
if (!async) { return CuMemcpyDtoD(d_dst, d_src, bytes); }
return CuMemcpyDtoDAsync(d_dst, d_src, bytes, internal::cuStream);
return CuMemcpyDtoDAsync(d_dst, d_src, bytes);
}
void MemoryManager::RegisterCheck(void *ptr)
-1
View File
@@ -13,7 +13,6 @@
#define MFEM_OCCA_HPP
#include "../config/config.hpp"
#include "cuda.hpp" // for CUdevice, CUcontext
#ifdef MFEM_USE_OCCA
#include <occa.hpp>
+78 -52
View File
@@ -38,6 +38,7 @@ SparseMatrix::SparseMatrix(int nrows, int ncols)
current_row(-1),
ColPtrJ(NULL),
ColPtrNode(NULL),
At(NULL),
ownGraph(true),
ownData(true),
isSorted(false)
@@ -60,6 +61,7 @@ SparseMatrix::SparseMatrix(int *i, int *j, double *data, int m, int n)
Rows(NULL),
ColPtrJ(NULL),
ColPtrNode(NULL),
At(NULL),
ownGraph(true),
ownData(true),
isSorted(false)
@@ -78,6 +80,7 @@ SparseMatrix::SparseMatrix(int *i, int *j, double *data, int m, int n,
Rows(NULL),
ColPtrJ(NULL),
ColPtrNode(NULL),
At(NULL),
ownGraph(ownij),
ownData(owna),
isSorted(issorted)
@@ -103,6 +106,7 @@ SparseMatrix::SparseMatrix(int nrows, int ncols, int rowsize)
, Rows(NULL)
, ColPtrJ(NULL)
, ColPtrNode(NULL)
, At(NULL)
, ownGraph(true)
, ownData(true)
, isSorted(false)
@@ -183,6 +187,7 @@ SparseMatrix::SparseMatrix(const SparseMatrix &mat, bool copy_graph)
current_row = -1;
ColPtrJ = NULL;
ColPtrNode = NULL;
At = NULL;
isSorted = mat.isSorted;
}
@@ -191,6 +196,7 @@ SparseMatrix::SparseMatrix(const Vector &v)
, Rows(NULL)
, ColPtrJ(NULL)
, ColPtrNode(NULL)
, At(NULL)
, ownGraph(true)
, ownData(true)
, isSorted(true)
@@ -245,6 +251,7 @@ void SparseMatrix::SetEmpty()
current_row = -1;
ColPtrJ = NULL;
ColPtrNode = NULL;
At = NULL;
#ifdef MFEM_USE_MEMALLOC
NodesMem = NULL;
#endif
@@ -333,7 +340,7 @@ void SparseMatrix::SetWidth(int newWidth)
// Nothing to be done here
return;
}
else if ( newWidth == -1)
else if (newWidth == -1)
{
// Compute the actual width
width = ActualWidth();
@@ -552,12 +559,10 @@ void SparseMatrix::Mult(const Vector &x, Vector &y) const
void SparseMatrix::AddMult(const Vector &x, Vector &y, const double a) const
{
MFEM_ASSERT(width == x.Size(),
"Input vector size (" << x.Size() << ") must match matrix width (" << width
<< ")");
MFEM_ASSERT(height == y.Size(),
"Output vector size (" << y.Size() << ") must match matrix height (" << height
<< ")");
MFEM_ASSERT(width == x.Size(), "Input vector size (" << x.Size()
<< ") must match matrix width (" << width << ")");
MFEM_ASSERT(height == y.Size(), "Output vector size (" << y.Size()
<< ") must match matrix height (" << height << ")");
int i, j, end;
double *Ap = A, *yp = y.GetData();
@@ -636,12 +641,10 @@ void SparseMatrix::MultTranspose(const Vector &x, Vector &y) const
void SparseMatrix::AddMultTranspose(const Vector &x, Vector &y,
const double a) const
{
MFEM_ASSERT(height == x.Size(),
"Input vector size (" << x.Size() << ") must match matrix height (" << height
<< ")");
MFEM_ASSERT(width == y.Size(),
"Output vector size (" << y.Size() << ") must match matrix width (" << width
<< ")");
MFEM_ASSERT(height == x.Size(), "Input vector size (" << x.Size()
<< ") must match matrix height (" << height << ")");
MFEM_ASSERT(width == y.Size(), "Output vector size (" << y.Size()
<< ") must match matrix width (" << width << ")");
if (A == NULL)
{
@@ -658,23 +661,40 @@ void SparseMatrix::AddMultTranspose(const Vector &x, Vector &y,
}
return;
}
// Prepare the lambda capture and get our pointers from the memory manager
const int d_height = height;
const DeviceArray d_I(I);
const DeviceArray d_J(J);
const DeviceVector d_A(A);
const DeviceVector d_x(x, x.Size());
DeviceVector d_y(y, y.Size());
MFEM_FORALL(i, d_height,
if (At)
{
const double xi = a * d_x[i];
const int end = d_I[i+1];
for (int j = d_I[i]; j < end; j++)
At->AddMult(x, y, a);
}
else
{
MFEM_VERIFY(Device::IsDisabled(), "transpose action on device is not "
"enabled; see BuildTranspose() for details.");
for (int i = 0; i < height; i++)
{
const int Jj = d_J[j];
AtomicAdd(&d_y[Jj], d_A[j] * xi);
const double xi = a * x[i];
const int end = I[i+1];
for (int j = I[i]; j < end; j++)
{
const int Jj = J[j];
y[Jj] += A[j] * xi;
}
}
});
}
}
void SparseMatrix::BuildTranspose() const
{
if (At == NULL)
{
At = Transpose(*this);
}
}
void SparseMatrix::ResetTranspose() const
{
delete At;
At = NULL;
}
void SparseMatrix::PartMult(
@@ -2101,12 +2121,12 @@ void SparseMatrix::Set(const int i, const int j, const double A)
if ((gi=i) < 0) { gi = -1-gi, s = -1; }
else { s = 1; }
MFEM_ASSERT(gi < height,
"Trying to insert a row " << gi << " outside the matrix height "
"Trying to set a row " << gi << " outside the matrix height "
<< height);
if ((gj=j) < 0) { gj = -1-gj, t = -s; }
else { t = s; }
MFEM_ASSERT(gj < width,
"Trying to insert a column " << gj << " outside the matrix width "
"Trying to set a column " << gj << " outside the matrix width "
<< width);
if (t < 0) { a = -a; }
_Set_(gi, gj, a);
@@ -2142,7 +2162,7 @@ void SparseMatrix::SetSubMatrix(const Array<int> &rows, const Array<int> &cols,
if ((gi=rows[i]) < 0) { gi = -1-gi, s = -1; }
else { s = 1; }
MFEM_ASSERT(gi < height,
"Trying to insert a row " << gi << " outside the matrix height "
"Trying to set a row " << gi << " outside the matrix height "
<< height);
SetColPtr(gi);
for (j = 0; j < cols.Size(); j++)
@@ -2155,7 +2175,7 @@ void SparseMatrix::SetSubMatrix(const Array<int> &rows, const Array<int> &cols,
if ((gj=cols[j]) < 0) { gj = -1-gj, t = -s; }
else { t = s; }
MFEM_ASSERT(gj < width,
"Trying to insert a column " << gj << " outside the matrix width "
"Trying to set a column " << gj << " outside the matrix width "
<< width);
if (t < 0) { a = -a; }
_Set_(gj, a);
@@ -2177,7 +2197,7 @@ void SparseMatrix::SetSubMatrixTranspose(const Array<int> &rows,
if ((gi=rows[i]) < 0) { gi = -1-gi, s = -1; }
else { s = 1; }
MFEM_ASSERT(gi < height,
"Trying to insert a row " << gi << " outside the matrix height "
"Trying to set a row " << gi << " outside the matrix height "
<< height);
SetColPtr(gi);
for (j = 0; j < cols.Size(); j++)
@@ -2190,7 +2210,7 @@ void SparseMatrix::SetSubMatrixTranspose(const Array<int> &rows,
if ((gj=cols[j]) < 0) { gj = -1-gj, t = -s; }
else { t = s; }
MFEM_ASSERT(gj < width,
"Trying to insert a column " << gj << " outside the matrix width "
"Trying to set a column " << gj << " outside the matrix width "
<< width);
if (t < 0) { a = -a; }
_Set_(gj, a);
@@ -2210,7 +2230,7 @@ void SparseMatrix::GetSubMatrix(const Array<int> &rows, const Array<int> &cols,
if ((gi=rows[i]) < 0) { gi = -1-gi, s = -1; }
else { s = 1; }
MFEM_ASSERT(gi < height,
"Trying to insert a row " << gi << " outside the matrix height "
"Trying to read a row " << gi << " outside the matrix height "
<< height);
SetColPtr(gi);
for (j = 0; j < cols.Size(); j++)
@@ -2218,7 +2238,7 @@ void SparseMatrix::GetSubMatrix(const Array<int> &rows, const Array<int> &cols,
if ((gj=cols[j]) < 0) { gj = -1-gj, t = -s; }
else { t = s; }
MFEM_ASSERT(gj < width,
"Trying to insert a column " << gj << " outside the matrix width "
"Trying to read a column " << gj << " outside the matrix width "
<< width);
a = _Get_(gj);
subm(i, j) = (t < 0) ? (-a) : (a);
@@ -2236,7 +2256,7 @@ bool SparseMatrix::RowIsEmpty(const int row) const
gi = -1-gi;
}
MFEM_ASSERT(gi < height,
"Trying to insert a row " << gi << " outside the matrix height "
"Trying to query a row " << gi << " outside the matrix height "
<< height);
if (Rows)
{
@@ -2255,7 +2275,7 @@ int SparseMatrix::GetRow(const int row, Array<int> &cols, Vector &srow) const
if ((gi=row) < 0) { gi = -1-gi; }
MFEM_ASSERT(gi < height,
"Trying to insert a row " << gi << " outside the matrix height "
"Trying to read a row " << gi << " outside the matrix height "
<< height);
if (Rows)
{
@@ -2282,7 +2302,7 @@ int SparseMatrix::GetRow(const int row, Array<int> &cols, Vector &srow) const
j = I[gi];
cols.MakeRef(J + j, I[gi+1]-j);
srow.NewDataAndSize(A + j, cols.Size());
MFEM_ASSERT(row >= 0, "Row not valid: " << row );
MFEM_ASSERT(row >= 0, "Row not valid: " << row << ", height: " << height);
return 1;
}
}
@@ -2296,7 +2316,7 @@ void SparseMatrix::SetRow(const int row, const Array<int> &cols,
if ((gi=row) < 0) { gi = -1-gi, s = -1; }
else { s = 1; }
MFEM_ASSERT(gi < height,
"Trying to insert a row " << gi << " outside the matrix height "
"Trying to set a row " << gi << " outside the matrix height "
<< height);
if (!Finalized())
@@ -2307,7 +2327,7 @@ void SparseMatrix::SetRow(const int row, const Array<int> &cols,
if ((gj=cols[j]) < 0) { gj = -1-gj, t = -s; }
else { t = s; }
MFEM_ASSERT(gj < width,
"Trying to insert a column " << gj << " outside the matrix"
"Trying to set a column " << gj << " outside the matrix"
" width " << width);
a = srow(j);
if (t < 0) { a = -a; }
@@ -2325,7 +2345,7 @@ void SparseMatrix::SetRow(const int row, const Array<int> &cols,
if ((gj=cols[j]) < 0) { gj = -1-gj, t = -s; }
else { t = s; }
MFEM_ASSERT(gj < width,
"Trying to insert a column " << gj << " outside the matrix"
"Trying to set a column " << gj << " outside the matrix"
" width " << width);
J[i] = gj;
@@ -2771,9 +2791,10 @@ void SparseMatrix::Destroy()
delete NodesMem;
}
#endif
delete At;
}
int SparseMatrix::ActualWidth()
int SparseMatrix::ActualWidth() const
{
int awidth = 0;
if (A)
@@ -2817,8 +2838,10 @@ SparseMatrix *Transpose (const SparseMatrix &A)
"Finalize must be called before Transpose. Use TransposeRowMatrix instead");
int i, j, end;
int m, n, nnz, *A_i, *A_j, *At_i, *At_j;
double *A_data, *At_data;
const int *A_i, *A_j;
int m, n, nnz, *At_i, *At_j;
const double *A_data;
double *At_data;
m = A.Height(); // number of rows of A
n = A.Width(); // number of columns of A
@@ -2945,8 +2968,10 @@ SparseMatrix *Mult (const SparseMatrix &A, const SparseMatrix &B,
SparseMatrix *OAB)
{
int nrowsA, ncolsA, nrowsB, ncolsB;
int *A_i, *A_j, *B_i, *B_j, *C_i, *C_j, *B_marker;
double *A_data, *B_data, *C_data;
const int *A_i, *A_j, *B_i, *B_j;
int *C_i, *C_j, *B_marker;
const double *A_data, *B_data;
double *C_data;
int ia, ib, ic, ja, jb, num_nonzeros;
int row_start, counter;
double a_entry, b_entry;
@@ -3259,13 +3284,13 @@ SparseMatrix * Add(double a, const SparseMatrix & A, double b,
int * C_j;
double * C_data;
int * A_i = A.GetI();
int * A_j = A.GetJ();
double * A_data = A.GetData();
const int *A_i = A.GetI();
const int *A_j = A.GetJ();
const double *A_data = A.GetData();
int * B_i = B.GetI();
int * B_j = B.GetJ();
double * B_data = B.GetData();
const int *B_i = B.GetI();
const int *B_j = B.GetJ();
const double *B_data = B.GetData();
int * marker = new int[ncols];
std::fill(marker, marker+ncols, -1);
@@ -3493,6 +3518,7 @@ void SparseMatrix::Swap(SparseMatrix &other)
mfem::Swap(current_row, other.current_row);
mfem::Swap(ColPtrJ, other.ColPtrJ);
mfem::Swap(ColPtrNode, other.ColPtrNode);
mfem::Swap(At, other.At);
#ifdef MFEM_USE_MEMALLOC
mfem::Swap(NodesMem, other.NodesMem);
+74 -18
View File
@@ -64,6 +64,9 @@ protected:
mutable int* ColPtrJ;
mutable RowNode ** ColPtrNode;
/// Transpose of A. Owned. Used to perform MultTranspose() on devices.
mutable SparseMatrix *At;
#ifdef MFEM_USE_MEMALLOC
typedef MemAlloc <RowNode, 1024> RowNodeAlloc;
RowNodeAlloc * NodesMem;
@@ -97,6 +100,9 @@ public:
/** @brief Create a sparse matrix in CSR format. Ownership of @a i, @a j, and
@a data is optionally transferred to the SparseMatrix. */
/** If the parameter @a data is NULL, then the internal #A array is allocated
by this constructor (initializing it with zeros and taking ownership,
regardless of the parameter @a owna). */
SparseMatrix(int *i, int *j, double *data, int m, int n, bool ownij,
bool owna, bool issorted);
@@ -112,7 +118,7 @@ public:
ownership. */
SparseMatrix(const SparseMatrix &mat, bool copy_graph = true);
/// Create a SparseMatrix with diagonal v, i.e. A = Diag(v)
/// Create a SparseMatrix with diagonal @a v, i.e. A = Diag(v)
SparseMatrix(const Vector & v);
@@ -134,21 +140,35 @@ public:
/// Check if the SparseMatrix is empty.
bool Empty() const { return (A == NULL) && (Rows == NULL); }
/// Return the array #I
inline int *GetI() const { return I; }
/// Return the array #J
inline int *GetJ() const { return J; }
/// Return element data, i.e. array #A
inline double *GetData() const { return A; }
/// Returns the number of elements in row @a i
/// Return the array #I.
inline int *GetI() { return I; }
/// Return the array #I, const version.
inline const int *GetI() const { return I; }
/// Return the array #J.
inline int *GetJ() { return J; }
/// Return the array #J, const version.
inline const int *GetJ() const { return J; }
/// Return the element data, i.e. the array #A.
inline double *GetData() { return A; }
/// Return the element data, i.e. the array #A, const version.
inline const double *GetData() const { return A; }
/// Returns the number of elements in row @a i.
int RowSize(const int i) const;
/// Returns the maximum number of elements among all rows
/// Returns the maximum number of elements among all rows.
int MaxRowSize() const;
/// Return a pointer to the column indices in a row
/// Return a pointer to the column indices in a row.
int *GetRowColumns(const int row);
/// Return a pointer to the column indices in a row, const version.
const int *GetRowColumns(const int row) const;
/// Return a pointer to the entries in a row
/// Return a pointer to the entries in a row.
double *GetRowEntries(const int row);
/// Return a pointer to the entries in a row, const version.
const double *GetRowEntries(const int row) const;
/// Change the width of a SparseMatrix.
@@ -163,7 +183,7 @@ public:
/// Returns the actual Width of the matrix.
/*! This method can be called for matrices finalized or not. */
int ActualWidth();
int ActualWidth() const;
/// Sort the column indices corresponding to each row.
void SortColumnIndices();
@@ -206,13 +226,45 @@ public:
void AddMultTranspose(const Vector &x, Vector &y,
const double a = 1.0) const;
/** @brief Build and store internally the transpose of this matrix which will
be used in the methods AddMultTranspose() and MultTranspose(). */
/** If this method has been called, the internal transpose matrix will be
used to perform the action of the transpose matrix in AddMultTranspose(),
and MultTranspose().
Warning: any changes in this matrix will invalidate the internal
transpose. To rebuild the transpose, call ResetTranspose() followed by a
call to this method. If the internal transpose is already built, this
method has no effect.
When any non-default backend is enabled, i.e. Device::IsEnabled() is
true, the methods AddMultTranspose(), and MultTranspose(), require the
internal transpose to be built. If that is not the case (i.e. the
internal transpose is not built), these methods will raise an error with
an appropriate message pointing to this method. When using the default
backend, calling this method is optional.
This method can only be used when the sparse matrix is finalized. */
void BuildTranspose() const;
/** Reset (destroy) the internal transpose matrix. See BuildTranspose() for
more details. */
void ResetTranspose() const;
void PartMult(const Array<int> &rows, const Vector &x, Vector &y) const;
void PartAddMult(const Array<int> &rows, const Vector &x, Vector &y,
const double a=1.0) const;
/// y = A * x, but treat all elements as booleans (zero=false, nonzero=true).
/// y = A * x, treating all entries as booleans (zero=false, nonzero=true).
/** The actual values stored in the data array, #A, are not used - this means
and that all entries in the sparsity pattern are considered to be true by
this method. */
void BooleanMult(const Array<int> &x, Array<int> &y) const;
/// y = At * x, but treat all elements as booleans (zero=false, nonzero=true).
/// y = At * x, treating all entries as booleans (zero=false, nonzero=true).
/** The actual values stored in the data array, #A, are not used - this means
and that all entries in the sparsity pattern are considered to be true by
this method. */
void BooleanMultTranspose(const Array<int> &x, Array<int> &y) const;
/// Compute y^t A x
@@ -452,11 +504,11 @@ SparseMatrix *Transpose(const SparseMatrix &A);
SparseMatrix *TransposeAbstractSparseMatrix (const AbstractSparseMatrix &A,
int useActualWidth);
/** Matrix product A.B.
If OAB is not NULL, we assume it has the structure
of A.B and store the result in OAB.
If OAB is NULL, we create a new SparseMatrix to store
/// Matrix product A.B.
/** If @a OAB is not NULL, we assume it has the structure of A.B and store the
result in @a OAB. If @a OAB is NULL, we create a new SparseMatrix to store
the result and return a pointer to it.
All matrices must be finalized. */
SparseMatrix *Mult(const SparseMatrix &A, const SparseMatrix &B,
SparseMatrix *OAB = NULL);
@@ -555,16 +607,20 @@ inline void SparseMatrix::SetColPtr(const int row) const
inline void SparseMatrix::ClearColPtr() const
{
if (Rows)
{
for (RowNode *node_p = Rows[current_row]; node_p != NULL;
node_p = node_p->Prev)
{
ColPtrNode[node_p->Column] = NULL;
}
}
else
{
for (int j = I[current_row], end = I[current_row+1]; j < end; j++)
{
ColPtrJ[J[j]] = -1;
}
}
}
inline double &SparseMatrix::SearchRow(const int col)
+9 -9
View File
@@ -857,11 +857,11 @@ static double cuVectorMin(const int N, const double *X)
const int bytes = min_sz*sizeof(double);
static double *h_min = NULL;
if (!h_min) { h_min = (double*)calloc(min_sz,sizeof(double)); }
static CUdeviceptr gdsr = (CUdeviceptr) NULL;
if (!gdsr) { ::cuMemAlloc(&gdsr,bytes); }
static void *gdsr = NULL;
if (!gdsr) { MFEM_CUDA_CHECK(cudaMalloc(&gdsr, bytes)); }
cuKernelMin<<<gridSize,blockSize>>>(N, (double*)gdsr, x);
MFEM_CUDA_CHECK_RT(cudaGetLastError());
::cuMemcpy((CUdeviceptr)h_min,(CUdeviceptr)gdsr,bytes);
MFEM_CUDA_CHECK(cudaGetLastError());
MFEM_CUDA_CHECK(cudaMemcpy(h_min, gdsr, bytes, cudaMemcpyDeviceToHost));
double min = std::numeric_limits<double>::infinity();
for (int i = 0; i < min_sz; i++) { min = fmin(min, h_min[i]); }
return min;
@@ -909,19 +909,19 @@ static double cuVectorDot(const int N, const double *X, const double *Y)
if (h_dot) { free(h_dot); }
h_dot = (double*)calloc(dot_sz,sizeof(double));
}
static CUdeviceptr gdsr = (CUdeviceptr) NULL;
static void *gdsr = NULL;
if (!gdsr or dot_block_sz!=dot_sz)
{
if (gdsr) { MFEM_CUDA_CHECK_DRV(::cuMemFree(gdsr)); }
MFEM_CUDA_CHECK_DRV(::cuMemAlloc(&gdsr,bytes));
if (gdsr) { MFEM_CUDA_CHECK(cudaFree(gdsr)); }
MFEM_CUDA_CHECK(cudaMalloc(&gdsr,bytes));
}
if (dot_block_sz!=dot_sz)
{
dot_block_sz = dot_sz;
}
cuKernelDot<<<gridSize,blockSize>>>(N, (double*)gdsr, x, y);
MFEM_CUDA_CHECK_RT(cudaGetLastError());
MFEM_CUDA_CHECK_DRV(::cuMemcpy((CUdeviceptr)h_dot,(CUdeviceptr)gdsr,bytes));
MFEM_CUDA_CHECK(cudaGetLastError());
MFEM_CUDA_CHECK(cudaMemcpy(h_dot, gdsr, bytes, cudaMemcpyDeviceToHost));
double dot = 0.0;
for (int i = 0; i < dot_sz; i++) { dot += h_dot[i]; }
return dot;
+27 -5
View File
@@ -211,6 +211,9 @@ ifneq ($(MFEM_USE_CUDA),YES)
XCOMPILER = $(CXX_XCOMPILER)
XLINKER = $(CXX_XLINKER)
else
ifneq ($(MFEM_USE_MM),YES)
$(error MFEM_USE_CUDA=YES requires MFEM_USE_MM=YES)
endif
MFEM_CXX ?= $(CUDA_CXX)
CXXFLAGS += $(CUDA_FLAGS) -ccbin $(CXX_OR_MPICXX)
XCOMPILER = $(CUDA_XCOMPILER)
@@ -230,7 +233,7 @@ endif
# List of MFEM dependencies, that require the *_LIB variable to be non-empty
MFEM_REQ_LIB_DEPS = SUPERLU METIS CONDUIT SIDRE LAPACK SUNDIALS MESQUITE\
SUITESPARSE STRUMPACK GECKO GNUTLS NETCDF PETSC MPFR PUMI CUDA OCCA RAJA
SUITESPARSE STRUMPACK GECKO GNUTLS NETCDF PETSC MPFR PUMI OCCA RAJA
PETSC_ERROR_MSG = $(if $(PETSC_FOUND),,. PETSC config not found: $(PETSC_VARS))
define mfem_check_dependency
@@ -246,7 +249,7 @@ ifeq ($(MAKECMDGOALS),config)
endif
# List of MFEM dependencies, processed below
MFEM_DEPENDENCIES = $(MFEM_REQ_LIB_DEPS) LIBUNWIND OPENMP
MFEM_DEPENDENCIES = $(MFEM_REQ_LIB_DEPS) LIBUNWIND OPENMP CUDA
# List of deprecated MFEM dependencies, processed below
MFEM_LEGACY_DEPENDENCIES = OPENMP
@@ -319,8 +322,8 @@ MFEM_TEST_MK ?= @MFEM_DIR@/config/test.mk
# Use "\n" (interpreted by sed) to add a newline.
MFEM_CONFIG_EXTRA ?= $(if $(BUILD_DIR_DEF),MFEM_BUILD_DIR ?= @MFEM_DIR@,)
MFEM_SOURCE_DIR := $(MFEM_REAL_DIR)
MFEM_INSTALL_DIR := $(BUILD_REAL_DIR)
MFEM_SOURCE_DIR = $(MFEM_REAL_DIR)
MFEM_INSTALL_DIR = $(abspath $(MFEM_PREFIX))
# If we have 'config' target, export variables used by config/makefile
ifneq (,$(filter config,$(MAKECMDGOALS)))
@@ -344,6 +347,11 @@ ifneq (,$(filter install,$(MAKECMDGOALS)))
MFEM_LIBS = $(if $(shared),$(INSTALL_RPATH)) -L@MFEM_LIB_DIR@ -lmfem\
@MFEM_EXT_LIBS@
MFEM_LIB_FILE = @MFEM_LIB_DIR@/libmfem.$(if $(shared),$(SO_VER),a)
ifeq ($(MFEM_USE_OCCA),YES)
ifneq ($(MFEM_INSTALL_DIR),$(abspath $(PREFIX))
$(error OCCA is enabled: PREFIX must be set during configuration!)
endif
endif
MFEM_PREFIX := $(abspath $(PREFIX))
MFEM_INC_DIR = $(abspath $(PREFIX_INC))
MFEM_LIB_DIR = $(abspath $(PREFIX_LIB))
@@ -358,6 +366,7 @@ DIRS = general linalg mesh fem
SOURCE_FILES = $(foreach dir,$(DIRS),$(wildcard $(SRC)$(dir)/*.cpp))
RELSRC_FILES = $(patsubst $(SRC)%,%,$(SOURCE_FILES))
OBJECT_FILES = $(patsubst $(SRC)%,$(BLD)%,$(SOURCE_FILES:.cpp=.o))
OKL_DIRS = fem
.PHONY: lib all clean distclean install config status info deps serial parallel\
debug pdebug cuda pcuda cudebug pcudebug style check test unittest\
@@ -395,10 +404,18 @@ doc:
-include $(BLD)deps.mk
ifneq ($(MFEM_USE_CUDA),YES)
$(BLD)libmfem.a: $(OBJECT_FILES)
$(AR) $(ARFLAGS) $(@) $(OBJECT_FILES)
$(RANLIB) $(@)
@$(MAKE) deprecation-warnings
else
$(BLD)libmfem.a: $(OBJECT_FILES)
nvcc -arch=sm_60 --device-link $(OBJECT_FILES) -o mfemCudaObjs.o
nvcc -arch=sm_60 --lib $(OBJECT_FILES) mfemCudaObjs.o -o $(@)
$(RANLIB) $(@)
@$(MAKE) deprecation-warnings
endif
$(BLD)libmfem.$(SO_EXT): $(BLD)libmfem.$(SO_VER)
cd $(@D) && ln -sf $(<F) $(@F)
@@ -498,7 +515,12 @@ install: $(if $(static),$(BLD)libmfem.a) $(if $(shared),$(BLD)libmfem.$(SO_EXT))
# install remaining includes in each subdirectory
for dir in $(DIRS); do \
mkdir -p $(PREFIX_INC)/mfem/$$dir && \
$(INSTALL) -m 640 $(SRC)$$dir/*.hpp $(SRC)$$dir/*.okl $(PREFIX_INC)/mfem/$$dir; \
$(INSTALL) -m 640 $(SRC)$$dir/*.hpp $(PREFIX_INC)/mfem/$$dir; \
done
# install *.okl files
for dir in $(OKL_DIRS); do \
mkdir -p $(PREFIX_INC)/mfem/$$dir && \
$(INSTALL) -m 640 $(SRC)$$dir/*.okl $(PREFIX_INC)/mfem/$$dir; \
done
# install config.mk in $(PREFIX_SHARE)
mkdir -p $(PREFIX_SHARE)
+17 -16
View File
@@ -3090,6 +3090,23 @@ void ParMesh::Rebalance()
" meshes.");
}
// Make sure the Nodes use a ParFiniteElementSpace
if (Nodes && dynamic_cast<ParFiniteElementSpace*>(Nodes->FESpace()) == NULL)
{
ParFiniteElementSpace *pfes =
new ParFiniteElementSpace(*Nodes->FESpace(), *this);
ParGridFunction *new_nodes = new ParGridFunction(pfes);
*new_nodes = *Nodes;
if (Nodes->OwnFEC())
{
new_nodes->MakeOwner(Nodes->OwnFEC());
Nodes->MakeOwner(NULL); // takes away ownership of 'fec' and 'fes'
delete Nodes->FESpace();
}
delete Nodes;
Nodes = new_nodes;
}
DeleteFaceNbrData();
pncmesh->Rebalance();
@@ -3110,22 +3127,6 @@ void ParMesh::Rebalance()
last_operation = Mesh::REBALANCE;
sequence++;
// Make sure the Nodes use a ParFiniteElementSpace
if (Nodes && dynamic_cast<ParFiniteElementSpace*>(Nodes->FESpace()) == NULL)
{
ParFiniteElementSpace *pfes =
new ParFiniteElementSpace(*Nodes->FESpace(), *this);
ParGridFunction *new_nodes = new ParGridFunction(pfes);
*new_nodes = *Nodes;
if (Nodes->OwnFEC())
{
new_nodes->MakeOwner(Nodes->OwnFEC());
Nodes->MakeOwner(NULL); // takes away ownership of 'fec' and 'fes'
delete Nodes->FESpace();
}
delete Nodes;
Nodes = new_nodes;
}
UpdateNodes();
}
+2 -2
View File
@@ -9,11 +9,11 @@
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Include the build directory where mfem.hpp and mfem-performance.hpp are.
include_directories(BEFORE ${PROJECT_BINARY_DIR})
# Include the top mfem source directory - needed by some tests, e.g. to
# #include "general/text.hpp".
include_directories(BEFORE ${PROJECT_SOURCE_DIR})
# Include the build directory where mfem.hpp and mfem-performance.hpp are.
include_directories(BEFORE ${PROJECT_BINARY_DIR})
# Include the source directory for the unit tests - catch.hpp is there.
include_directories(BEFORE ${CMAKE_CURRENT_SOURCE_DIR})
+7 -1
View File
@@ -12,7 +12,13 @@
#include "mfem.hpp"
#include "catch.hpp"
#include <stdio.h>
#include <unistd.h> // rmdir
#ifndef _WIN32
#include <unistd.h> // rmdir
#else
#include <direct.h> // _rmdir
#define rmdir(dir) _rmdir(dir)
#endif
using namespace mfem;
+1 -1
View File
@@ -125,7 +125,7 @@ TEST_CASE("InverseElementTransformation",
REQUIRE( mesh_file.good() );
const int npts = 100; // number of random points to test
const int min_found_pts = 94;
const int min_found_pts = 93;
const int rand_seed = 189548;
srand(rand_seed);