Compare commits
7
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e176bb8d06 | ||
|
|
c1509fde91 | ||
|
|
44f25808b9 | ||
|
|
8296bca522 | ||
|
|
d4825be0ca | ||
|
|
ff143322bd | ||
|
|
b08c652314 |
@@ -1,49 +0,0 @@
|
||||
version: '{build}'
|
||||
|
||||
# https://www.appveyor.com/docs/build-environment/#build-worker-images
|
||||
image: Visual Studio 2017
|
||||
|
||||
install:
|
||||
|
||||
# Install MS-MPI
|
||||
- ps: Start-FileDownload 'https://download.microsoft.com/download/B/2/E/B2EB83FE-98C2-4156-834A-E1711E6884FB/MSMpiSetup.exe'
|
||||
- MSMpiSetup.exe -unattend
|
||||
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
|
||||
|
||||
# Install MS-MPI SDK
|
||||
- ps: Start-FileDownload 'https://download.microsoft.com/download/B/2/E/B2EB83FE-98C2-4156-834A-E1711E6884FB/msmpisdk.msi'
|
||||
- msmpisdk.msi /passive
|
||||
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
|
||||
|
||||
# Install METIS
|
||||
- ps: Start-FileDownload 'http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz'
|
||||
- 7z x metis-5.1.0.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd metis-5.1.0
|
||||
- ps: ( get-content "GKlib\gk_arch.h") | % { If ($_.ReadCount -ge 52) {$_ -replace "#ifdef __MSC__","#ifdef DISABLE_THIS_ANCIENT_MSC_CHECK"} Else {$_} } | set-content "GKlib\gk_arch.h"
|
||||
- cmake -H. -Bbuild
|
||||
# -DCMAKE_BUILD_TYPE=Release
|
||||
- cmake --build build
|
||||
- cd ..
|
||||
|
||||
# Install hypre
|
||||
- ps: Start-FileDownload 'https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz'
|
||||
- 7z x hypre-2.10.0b.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd hypre-2.10.0b
|
||||
- cmake -Hsrc -Bbuild -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
# - cmake -Hsrc -Bbuild -DCMAKE_BUILD_TYPE=Release -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
- cmake --build build
|
||||
- cmake --build build --target install
|
||||
- cd ..
|
||||
|
||||
# MFEM
|
||||
before_build:
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2.10.0b\src\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2.10.0b\src\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_serial -DMFEM_USE_MPI=FALSE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2.10.0b\src\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2.10.0b\src\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
|
||||
build_script:
|
||||
- cmake --build build_parallel
|
||||
- cmake --build build_serial
|
||||
|
||||
after_build:
|
||||
# - cmake --build build_parallel --target check
|
||||
- cmake --build build_serial --target check
|
||||
-158
@@ -1,158 +0,0 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# Ignore files that are generated from the repository sources by either building
|
||||
# the code or running it. These should be the same as the files erased by
|
||||
# `make distclean`.
|
||||
#
|
||||
# Also ignore OS-specific files like .DS_Store on Mac
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
# Object and library files
|
||||
*.o
|
||||
/libmfem.*
|
||||
|
||||
# CMake generated files
|
||||
CMakeCache.txt
|
||||
CMakeFiles/
|
||||
|
||||
# Backup files
|
||||
*~
|
||||
|
||||
# Default install location
|
||||
/mfem/
|
||||
|
||||
# Generated files in main directory, config/ and docs/
|
||||
/deps.mk
|
||||
config/_config.hpp
|
||||
config/config.mk
|
||||
config/sample-runs-build.log
|
||||
doc/CodeDocumentation.conf
|
||||
doc/CodeDocumentation.html
|
||||
doc/CodeDocumentation
|
||||
|
||||
# Temporary files created by the tests.
|
||||
*.stderr
|
||||
|
||||
# Totalview breakpoint files
|
||||
*.TVD.*breakpoints
|
||||
|
||||
# OS-specific: Mac
|
||||
*.dSYM
|
||||
.DS_Store
|
||||
|
||||
# Example and miniapp binaries and outputs
|
||||
|
||||
examples/ex[1-9]
|
||||
examples/ex[1-9]p
|
||||
examples/ex1[04-9]
|
||||
examples/ex1[0-9]p
|
||||
|
||||
examples/refined.mesh
|
||||
examples/displaced.mesh
|
||||
examples/mesh.*
|
||||
examples/ex5.mesh
|
||||
examples/Example5*
|
||||
examples/Example9*
|
||||
examples/Example15*
|
||||
examples/Example16*
|
||||
examples/sphere_refined.*
|
||||
examples/sol.*
|
||||
examples/sol_u.*
|
||||
examples/sol_p.*
|
||||
examples/ex9.mesh
|
||||
examples/ex9-mesh.*
|
||||
examples/ex9-init.*
|
||||
examples/ex9-final.*
|
||||
examples/deformed.*
|
||||
examples/velocity.*
|
||||
examples/elastic_energy.*
|
||||
examples/mode_*
|
||||
examples/ex16.mesh
|
||||
examples/ex16-mesh.*
|
||||
examples/ex16-init.*
|
||||
examples/ex16-final.*
|
||||
examples/vortex-mesh.*
|
||||
examples/vortex.mesh
|
||||
examples/vortex-?-init.*
|
||||
examples/vortex-?-final.*
|
||||
examples/deformation.*
|
||||
examples/pressure.*
|
||||
|
||||
examples/sundials/ex9
|
||||
examples/sundials/ex1[06]
|
||||
examples/sundials/ex9p
|
||||
examples/sundials/ex1[06]p
|
||||
|
||||
examples/sundials/ex9.mesh
|
||||
examples/sundials/ex9-mesh.*
|
||||
examples/sundials/ex9-init.*
|
||||
examples/sundials/ex9-final.*
|
||||
examples/sundials/Example9*
|
||||
examples/sundials/deformed.*
|
||||
examples/sundials/velocity.*
|
||||
examples/sundials/elastic_energy.*
|
||||
examples/sundials/ex16.mesh
|
||||
examples/sundials/ex16-mesh.*
|
||||
examples/sundials/ex16-init.*
|
||||
examples/sundials/ex16-final.*
|
||||
examples/sundials/Example16*
|
||||
|
||||
examples/petsc/ex[1-69]p
|
||||
examples/petsc/ex10p
|
||||
|
||||
examples/petsc/mesh.*
|
||||
examples/petsc/sol.*
|
||||
examples/petsc/sol_p.*
|
||||
examples/petsc/sol_u.*
|
||||
examples/petsc/Example5*
|
||||
examples/petsc/ex9-mesh.*
|
||||
examples/petsc/ex9-init.*
|
||||
examples/petsc/ex9-final.*
|
||||
examples/petsc/Example9*
|
||||
examples/petsc/deformed.*
|
||||
examples/petsc/velocity.*
|
||||
examples/petsc/elastic_energy.*
|
||||
|
||||
miniapps/electromagnetics/volta
|
||||
miniapps/electromagnetics/tesla
|
||||
miniapps/electromagnetics/maxwell
|
||||
miniapps/electromagnetics/joule
|
||||
|
||||
miniapps/electromagnetics/Volta-AMR*
|
||||
miniapps/electromagnetics/Tesla-AMR*
|
||||
miniapps/electromagnetics/Maxwell-Parallel*
|
||||
miniapps/electromagnetics/Joule_*
|
||||
|
||||
miniapps/meshing/mobius-strip
|
||||
miniapps/meshing/klein-bottle
|
||||
miniapps/meshing/mesh-explorer
|
||||
miniapps/meshing/shaper
|
||||
miniapps/meshing/mesh-optimizer
|
||||
miniapps/meshing/pmesh-optimizer
|
||||
|
||||
miniapps/meshing/mobius-strip.mesh
|
||||
miniapps/meshing/klein-bottle.mesh
|
||||
miniapps/meshing/mesh-explorer.mesh
|
||||
miniapps/meshing/partitioning.txt
|
||||
miniapps/meshing/shaper.mesh
|
||||
miniapps/meshing/optimized*
|
||||
miniapps/meshing/perturbed*
|
||||
|
||||
miniapps/performance/ex1
|
||||
miniapps/performance/ex1p
|
||||
|
||||
miniapps/performance/refined.mesh
|
||||
miniapps/performance/mesh.*
|
||||
miniapps/performance/sol.*
|
||||
|
||||
miniapps/tools/display-basis
|
||||
miniapps/tools/load-dc
|
||||
miniapps/tools/convert-dc
|
||||
|
||||
miniapps/nurbs/ex1
|
||||
miniapps/nurbs/ex1p
|
||||
miniapps/nurbs/ex11p
|
||||
miniapps/nurbs/refined.mesh
|
||||
miniapps/nurbs/mesh.*
|
||||
miniapps/nurbs/sol.*
|
||||
miniapps/nurbs/mode_*
|
||||
miniapps/nurbs/Example1*
|
||||
+67
-230
@@ -1,207 +1,55 @@
|
||||
sudo: false
|
||||
|
||||
language: cpp
|
||||
|
||||
matrix:
|
||||
include:
|
||||
#
|
||||
# Linux
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
env: DEBUG=YES
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
env: DEBUG=NO
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=test
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
addons:
|
||||
apt:
|
||||
# sources:
|
||||
# - ubuntu-toolchain-r-test
|
||||
packages:
|
||||
# GCC 4.9
|
||||
# - g++-4.9
|
||||
# MPICH
|
||||
- mpich
|
||||
- libmpich-dev
|
||||
# OpenMPI
|
||||
# - openmpi-bin
|
||||
# - libopenmpi-dev
|
||||
env: DEBUG=YES
|
||||
MPI=YES
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
NPROCS=2
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
addons:
|
||||
apt:
|
||||
# sources:
|
||||
# - ubuntu-toolchain-r-test
|
||||
packages:
|
||||
# GCC 4.9
|
||||
# - g++-4.9
|
||||
# MPICH
|
||||
- mpich
|
||||
- libmpich-dev
|
||||
# OpenMPI
|
||||
# - openmpi-bin
|
||||
# - libopenmpi-dev
|
||||
env: DEBUG=NO
|
||||
MPI=YES
|
||||
CODECOV=YES
|
||||
MFEM_TEST_TARGET=test
|
||||
NPROCS=2
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
#
|
||||
# Mac OS X
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=YES
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=NO
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=test
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=YES
|
||||
MPI=YES
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
NPROCS=4
|
||||
TMPDIR=/tmp
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
- $HOME/local-cached
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=NO
|
||||
MPI=YES
|
||||
CODECOV=YES
|
||||
MFEM_TEST_TARGET=test
|
||||
NPROCS=4
|
||||
TMPDIR=/tmp
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
- $HOME/local-cached
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
compiler:
|
||||
- gcc
|
||||
- clang
|
||||
|
||||
os:
|
||||
- linux
|
||||
- osx
|
||||
|
||||
env:
|
||||
- DEBUG=YES MPI=YES TMPDIR=/tmp
|
||||
- DEBUG=NO MPI=YES TMPDIR=/tmp
|
||||
- DEBUG=YES MPI=NO
|
||||
- DEBUG=NO MPI=NO
|
||||
|
||||
# Test with GCC on Linux an Clang on Mac
|
||||
matrix:
|
||||
exclude:
|
||||
- compiler: clang
|
||||
os: linux
|
||||
- compiler: gcc
|
||||
os: osx
|
||||
|
||||
before_install:
|
||||
# No addon for brew yet, have to install OSX packages this way.
|
||||
# - if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
|
||||
# brew install open-mpi;
|
||||
# fi
|
||||
|
||||
# On Mac OS X, build and cache OpenMPI 2.1.1:
|
||||
- if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
|
||||
if [ ! -e $HOME/local-cached/bin/mpicc ]; then
|
||||
mkdir -p $HOME/builds && cd $HOME/builds &&
|
||||
wget https://www.open-mpi.org/software/ompi/v2.1/downloads/openmpi-2.1.1.tar.bz2 &&
|
||||
tar jxf openmpi-2.1.1.tar.bz2 &&
|
||||
mkdir openmpi-build && cd openmpi-build &&
|
||||
../openmpi-2.1.1/configure --prefix=$HOME/local-cached &&
|
||||
make -j3 all && make install;
|
||||
fi;
|
||||
PATH=$HOME/local-cached/bin:$PATH;
|
||||
cd $TRAVIS_BUILD_DIR;
|
||||
fi
|
||||
|
||||
# Update environment to find g++ 4.9 installation first.
|
||||
# - if [ $TRAVIS_OS_NAME == "linux" ]; then
|
||||
# mkdir -p latest-gcc-symlinks;
|
||||
# ln -s /usr/bin/g++-4.9 latest-gcc-symlinks/g++;
|
||||
# ln -s /usr/bin/gcc-4.9 latest-gcc-symlinks/gcc;
|
||||
# ln -s /usr/bin/gcov-4.9 latest-gcc-symlinks/gcov;
|
||||
# export PATH=$PWD/latest-gcc-symlinks:$PATH;
|
||||
# fi
|
||||
|
||||
# Install tool to upload code coverage reports to coveralls.io
|
||||
- if [ "$CODECOV" == "YES" ]; then
|
||||
export PYTHONUSERBASE=$HOME/local;
|
||||
pip install --user cpp-coveralls;
|
||||
pip install --user pyyaml;
|
||||
PATH=$HOME/local/bin:$PATH;
|
||||
fi
|
||||
- if [ $TRAVIS_OS_NAME == "linux" -a "$CXX" == "g++" ]; then sudo add-apt-repository -y ppa:ubuntu-toolchain-r/test; fi
|
||||
- if [ $TRAVIS_OS_NAME == "linux" ]; then sudo apt-get update; fi || true
|
||||
|
||||
install:
|
||||
# Set MPI compilers, print compiler version
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ "$TRAVIS_OS_NAME" == "linux" ]; then
|
||||
export MPICH_CC="$CC";
|
||||
export MPICH_CXX="$CXX";
|
||||
else
|
||||
export OMPI_CC="$CC";
|
||||
export OMPI_CXX="$CXX";
|
||||
mpic++ --showme:version;
|
||||
fi;
|
||||
mpic++ -v;
|
||||
else
|
||||
$CXX -v;
|
||||
fi
|
||||
# g++-4.9
|
||||
- if [ $TRAVIS_OS_NAME == "linux" -a "$CXX" == "g++" ]; then sudo apt-get install -qq g++-4.9; fi
|
||||
- if [ $TRAVIS_OS_NAME == "linux" -a "$CXX" == "g++" ]; then export CXX="g++-4.9"; fi
|
||||
|
||||
# Back out of the mfem directory to install the libraries
|
||||
- cd ..
|
||||
|
||||
# OpenMPI
|
||||
- if [ $TRAVIS_OS_NAME == "linux" ]; then
|
||||
sudo apt-get install openmpi-bin openmpi-common openssh-client openssh-server libopenmpi1.3 libopenmpi-dbg libopenmpi-dev;
|
||||
else
|
||||
travis_wait brew install open-mpi;
|
||||
fi
|
||||
|
||||
# hypre
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ ! -e hypre-2.10.0b/src/hypre/lib/libHYPRE.a ]; then
|
||||
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
|
||||
rm -rf hypre-2.10.0b;
|
||||
tar xvzf hypre-2.10.0b.tar.gz;
|
||||
cd hypre-2.10.0b/src;
|
||||
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
|
||||
make -j3;
|
||||
cd ../..;
|
||||
if [ ! -d hypre-2.10.0b ]; then
|
||||
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
|
||||
tar xvzf hypre-2.10.0b.tar.gz;
|
||||
cd hypre-2.10.0b/src;
|
||||
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
|
||||
make -j 4;
|
||||
cd ../..;
|
||||
else
|
||||
echo "Reusing cached hypre-2.10.0b/";
|
||||
fi;
|
||||
@@ -210,54 +58,43 @@ install:
|
||||
fi
|
||||
|
||||
# METIS
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ ! -e metis-4.0/libmetis.a ]; then
|
||||
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
|
||||
tar xvzf metis-4.0.3.tar.gz;
|
||||
make -j3 -C metis-4.0.3/Lib CC="$CC" OPTFLAGS="-O2";
|
||||
rm -rf metis-4.0;
|
||||
mv metis-4.0.3 metis-4.0;
|
||||
else
|
||||
echo "Reusing cached metis-4.0/";
|
||||
fi;
|
||||
- if [ ! -d metis-4.0 ]; then
|
||||
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
|
||||
tar xvzf metis-4.0.3.tar.gz;
|
||||
cd metis-4.0.3;
|
||||
make -j 4;
|
||||
cd ..;
|
||||
mv metis-4.0.3 metis-4.0;
|
||||
else
|
||||
echo "Reusing cached metis-4.0/";
|
||||
fi
|
||||
|
||||
# # Delete an expired cache here: https://travis-ci.org/mfem/mfem/caches
|
||||
# cache:
|
||||
# directories:
|
||||
# - $TRAVIS_BUILD_DIR/../hypre-2.10.0b
|
||||
# - $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
|
||||
script:
|
||||
# Compiler
|
||||
- if [ $MPI == "YES" ]; then
|
||||
export MYCXX=mpic++;
|
||||
export OMPI_CXX="$CXX";
|
||||
$MYCXX --showme:version;
|
||||
else
|
||||
export MYCXX="$CXX";
|
||||
fi
|
||||
|
||||
# Print the compiler version
|
||||
- $MYCXX -v
|
||||
|
||||
# Set some variables
|
||||
- cd $TRAVIS_BUILD_DIR;
|
||||
CPPFLAGS="";
|
||||
SKIP_TEST_DIRS="";
|
||||
if [ "$CODECOV" == "YES" ]; then
|
||||
CPPFLAGS="--coverage -g";
|
||||
fi;
|
||||
if [ "$CXX" == "clang++" ]; then
|
||||
export MFEM_PERF_SW=clang;
|
||||
fi
|
||||
|
||||
# Configure the library
|
||||
- make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG MFEM_CXX="$MYCXX"
|
||||
MFEM_MPI_NP=$NPROCS CPPFLAGS="$CPPFLAGS"
|
||||
# Show the configuration
|
||||
- make info
|
||||
# Build the library
|
||||
- make -j3
|
||||
# Build the examples and the miniapps
|
||||
- make -j3 all
|
||||
# Run tests
|
||||
- make $MFEM_TEST_TARGET SKIP_TEST_DIRS="$SKIP_TEST_DIRS"
|
||||
|
||||
after_success:
|
||||
- if [ "$CODECOV" == "YES" ]; then
|
||||
coveralls --include fem --include general --include linalg --include
|
||||
mesh --exclude /usr --gcov-options '\-lp' --root $TRAVIS_BUILD_DIR;
|
||||
# Build the code and do a quick check (debug mode) or a full tests run (non-debug mode)
|
||||
- if [ $DEBUG == "NO" ]; then
|
||||
export MFEM_TEST_TARGET="test";
|
||||
else
|
||||
export MFEM_TEST_TARGET="check";
|
||||
fi
|
||||
# Build and check/test MFEM, its examples and miniapps
|
||||
- cd $TRAVIS_BUILD_DIR &&
|
||||
make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG MFEM_CXX="$MYCXX" &&
|
||||
make info &&
|
||||
make all -j 4 &&
|
||||
make $MFEM_TEST_TARGET
|
||||
|
||||
@@ -8,279 +8,16 @@
|
||||
http://mfem.org
|
||||
|
||||
|
||||
Version 3.3.3 (development)
|
||||
===========================
|
||||
Development version, not released yet
|
||||
=====================================
|
||||
|
||||
More efficient non-conforming adaptive mesh refinement
|
||||
------------------------------------------------------
|
||||
- Significantly reduced MPI communication in the construction of the parallel
|
||||
prolongation matrix in ParFiniteElementSpace, for much improved parallel
|
||||
scaling of non-conforming AMR on hundreds of thousands of MPI tasks. The
|
||||
memory footprint of the ParNCMesh class has also been reduced.
|
||||
|
||||
- In FiniteElementSpace, the fully assembled refinement matrix is now replaced
|
||||
by default by a specialized refinement operator. The operator option is both
|
||||
faster and more memory efficient than using the fully assembled matrix. The
|
||||
old approach is still available and can be enabled, if needed, using the new
|
||||
method FiniteElementSpace::SetUpdateOperatorType().
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Added support for a general "high-order"-to-"low-order refined" transfer of
|
||||
GridFunction and true-dof data from a "high-order" finite element space
|
||||
defined on a coarse mesh, to a "low-order refined" space defined on a refined
|
||||
mesh. The new methods, GetTransferOperator and GetTrueTransferOperator in the
|
||||
FiniteElementSpace classes, work in both serial and parallel and support
|
||||
matrix-based as well as matrix-free transfer operator representations. They
|
||||
use a new method, GetTransferMatrix, in the FiniteElement class similar to
|
||||
GetLocalInterpolation, that allows the coarse FiniteElement to be different
|
||||
from the fine FiniteElement.
|
||||
|
||||
- Added class ComplexOperator, that implements the action of a complex operator
|
||||
through the equivalent 2x2 real formulation. Both symmetric and antisymmetric
|
||||
block structures are supported.
|
||||
|
||||
- Added classes for general block nonlinear finite element operators (deriving
|
||||
from BlockNonlinearForm and ParBlockNonlinearForm) enabling solution of
|
||||
nonlinear systems with multiple unknowns in different function spaces. Such
|
||||
operators have assemble-based action and also support assembly of the gradient
|
||||
operator to enable inversion with Newton iteration.
|
||||
|
||||
- Added variable order NURBS: for each space each knot vector in the mesh can
|
||||
have a different order. The order information is now part of the finite
|
||||
element space header in the NURBS mesh output, so NURBS meshes in the old
|
||||
format need to be updated.
|
||||
|
||||
- In the classes NonlinearForm and ParNonlinearForm, added support for
|
||||
non-conforming AMR meshes; see also the "API changes" section.
|
||||
|
||||
- New specialized time integrators: symplectic integrators of orders 1-4 for
|
||||
systems of first order ODEs derived from a Hamiltonian and generalized-alpha
|
||||
ODE solver for the filtered Navier–Stokes equations with stabilization. See
|
||||
classes SIASolver and GeneralizedAlphaSolver in linalg/ode.hpp.
|
||||
|
||||
- Inherit finite element classes from the new base class TensorBasisElement,
|
||||
whenever the basis can be represented by a tensor product of 1D bases.
|
||||
|
||||
- Added support for elimination of boundary conditions in block matrices.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added a new serial and parallel example (ex19) that solves the quasi-static
|
||||
incompressible hyperelastic equations. The example demonstrates the use of
|
||||
block nonlinear forms as well as custom block preconditioners.
|
||||
|
||||
- Added a new electromagnetics miniapp, Maxwell, for simulating time-domain
|
||||
electromagnetics phenomena as a coupled first order system of equations.
|
||||
|
||||
- A simple local refinement option has been added to the mesh-explorer miniapp
|
||||
(menu option 'r', sub-option 'l') that selects elements for refinement based
|
||||
on their spatial location - see the function 'region()' in the source file.
|
||||
|
||||
- Added a set of miniapps specifically focused on Isogeometric Analysis (IGA) on
|
||||
NURBS meshes in the miniapps/nurbs directory. Currently the directory contains
|
||||
variable order NURBS versions of examples 1, 1p and 11p.
|
||||
|
||||
- Added two new miniapps related to DataCollection I/O in miniapps/tools:
|
||||
load-dc.cpp can be used to visualize fields saved via DataCollection classes;
|
||||
convert-dc.cpp demonstrates how to convert between MFEM's different concrete
|
||||
DataCollection options.
|
||||
|
||||
- Example 10p with its SUNDIALS and PETSc versions have been updated to reflect
|
||||
the change in the behavior of the method ParNonlinearForm::GetLocalGradient()
|
||||
(see the "API changes" section) and now works correctly on non-conforming AMR
|
||||
meshes. Example 10 and its SUNDIALS version have also been updated to support
|
||||
non-conforming ARM meshes.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Documented project workflow and provided contribution guidelines in the new
|
||||
top-level file, CONTRIBUTING.md.
|
||||
|
||||
- Added (optional) Conduit Mesh Blueprint support of MFEM data for both in-core
|
||||
and I/O use cases. This includes a new ConduitDataCollection that provides
|
||||
json, simple binary, and HDF5-based I/O. Support requires Conduit >= v0.3.1
|
||||
and VisIt >= v2.13.1 will read the new Data Collection outputs.
|
||||
|
||||
- Added a new developer tool, config/sample-runs.sh, that extracts the sample
|
||||
runs from all examples and miniapps and runs them. Optionally, it can save the
|
||||
output from the execution to files, allowing comparison between different
|
||||
versions and builds of the library.
|
||||
|
||||
- Support for building a shared version of the MFEM library with GNU make.
|
||||
|
||||
- Added a build option, MFEM_USE_EXCEPTIONS=YES, to throw an exception instead
|
||||
of calling abort on mfem errors.
|
||||
|
||||
- When building with the GnuTLS library, switch to using X.509 certificates for
|
||||
secure socket authentication. Support for the previously used OpenPGP keys has
|
||||
been deprecated in GnuTLS 3.5.x and removed in 3.6.0. For secure communication
|
||||
with the visualization tool GLVis, a new set of certificates can be generated
|
||||
using the latest version of the script 'glvis-keygen.sh' from GLVis.
|
||||
|
||||
- Upgraded MFEM to support Axom 0.2.8. Prior versions are no longer supported.
|
||||
|
||||
API changes
|
||||
-----------
|
||||
- Introduced a new enum, Matrix::DiagonalPolicy, that replaces the integer
|
||||
parameters in many methods that perform elimination of rows and/or columns in
|
||||
matrices. Some examples of such methods are:
|
||||
* class SparseMatrix: EliminateRow(), EliminateCol(), EliminateRowCol(), ...
|
||||
* class BilinearForm: EliminateEssentialBC(), EliminateVDofs(), ...
|
||||
* class StaticCondensation: EliminateReducedTrueDofs()
|
||||
* class BlockMatrix: EliminateRowCol()
|
||||
Calling these methods with an explicitly given (integer) constants, will now
|
||||
generate compilation errors, please use one of the new enum constants instead.
|
||||
|
||||
- Modified the virtual method AbstractSparseMatrix::EliminateZeroRows() and its
|
||||
implementations in derived classes, to accept an optional 'threshold'
|
||||
parameter, replacing previously hard-coded threshold values.
|
||||
|
||||
- In the classes NonlinearForm and ParNonlinearForm:
|
||||
* The method GetLocalGradient() no longer imposes boundary conditions. The
|
||||
motivation for the change is that, in the case of non-conforming AMR,
|
||||
performing the elimination at the local level is incorrect - it must be
|
||||
applied at the true-dof level.
|
||||
* The method SetEssentialVDofs() is now deprecated.
|
||||
|
||||
|
||||
Version 3.3.2, released on Nov 10, 2017
|
||||
=======================================
|
||||
|
||||
High-order mesh optimization
|
||||
----------------------------
|
||||
- Added support for mesh optimization via node-movement based on the Target-
|
||||
Matrix Optimization Paradigm (TMOP) developed by P.Knupp et al. A variety of
|
||||
mesh quality metrics, with their first and second derivatives have been
|
||||
implemented. The combination of targets & quality metrics is used to optimize
|
||||
the physical node positions, i.e., they must be as close as possible to the
|
||||
shape, size and/or alignment of their targets. The optimization of arbitrary
|
||||
high-order meshes in 2D, 3D, serial and parallel is supported.
|
||||
|
||||
- The new Mesh Optimizer miniapp can be used to perform mesh optimization with
|
||||
TMOP in serial and parallel versions. The miniapp also demonstrates the use of
|
||||
nonlinear operators and their coupling to Newton methods for solving
|
||||
minimization problems.
|
||||
|
||||
New and improved solvers and preconditioners
|
||||
--------------------------------------------
|
||||
- MFEM is now included in the xSDK project, the Extreme-scale Scientific
|
||||
Software Development Kit, as of xSDK-0.3.0. Various changes were made to
|
||||
comply with xSDK's community policies, https://xsdk.info/policies, including:
|
||||
xSDK-specific options in CMake, support for user-provided MPI communicators,
|
||||
runtime API for version number, and the ability to disable/redirect output.
|
||||
For more details, see general/globals.hpp and in particular the mfem::err and
|
||||
mfem::out streams replacing std::err and std::out respectively.
|
||||
|
||||
- Added (optional) support for the STRUMPACK parallel sparse direct solver and
|
||||
preconditioner. STRUMPACK uses Hierarchically Semi-Separable (HSS) compression
|
||||
in a fully algebraic manner, with interface similar to SuperLU_DIST. See
|
||||
http://portal.nersc.gov/project/sparse/strumpack for more details.
|
||||
|
||||
- Added a block lower triangular preconditioner based (only) on the actions of
|
||||
each block, see class BlockLowerTriangularPreconditioner.
|
||||
|
||||
- Added an optional operator in LOBPCG to projects vectors onto a desired
|
||||
subspace (e.g. divergence-free). Other small changes in LOBPCG include the
|
||||
ability to set the starting vectors and support for relative tolerance.
|
||||
|
||||
- The Newton solver supports an optional scaling factor, that can limit the
|
||||
increment in the Newton step, see e.g. the Mesh Optimizer miniapp.
|
||||
|
||||
- Updated MFEM integration to support the new SUNDIALS 3.0.0 interface.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added a new serial and parallel example (ex18) that solves the transient Euler
|
||||
equations on a periodic domain with explicit time integrators. In the process
|
||||
extended the NonlinearForm class to allow for integrals over faces and
|
||||
exchanging face-neighbor data in parallel.
|
||||
|
||||
- Added a new meshing miniapp, Shaper, that can be used to resolve complicated
|
||||
material interfaces by mesh refinement, e.g. as a tool for initial mesh
|
||||
generation from prescribed "material()" function. Both conforming and
|
||||
non-conforming (isotropic and anisotropic) refinements are supported.
|
||||
|
||||
- Added a new meshing miniapp, Mesh Optimizer, that demonstrates the use of TMOP
|
||||
for mesh optimization (serial and parallel version.)
|
||||
|
||||
- Added SUNDIALS version of Example 16/16p.
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Added a FindPoints method of the Mesh and ParMesh classes that returns the
|
||||
elements that contain a given set of points, together with the coordinates of
|
||||
the points in the reference space of the corresponding element. In parallel,
|
||||
if a point is shared by multiple processors, only one of them will mark that
|
||||
point as found. Note that the current implementation of this method is not
|
||||
optimal and/or 100% reliable. See the mesh-explorer miniapp for an example.
|
||||
|
||||
- Added a new class InverseElementTransformation, that supports a number of
|
||||
algorithms for inversion of general ElementTransformations. This class can be
|
||||
used as a more flexible and extensible alternative to ElementTransformation's
|
||||
TransformBack method. It is also used in the FindPoints methods as a tunable
|
||||
and customizable inversion algorithm.
|
||||
|
||||
- Memory optimizations in the NCMesh class, which now uses 50% less memory than
|
||||
before. The average cost of an element in a uniformly refined mesh (including
|
||||
the refinement hierarchy, but excluding the temporary face_list and edge_list)
|
||||
- Memory optimizations in the NCMesh class, which now uses 50% less memory.
|
||||
The average cost of an NC element in a uniformly refined mesh (including the
|
||||
refinement hierarchy, but excluding the temporary face_list and edge_list)
|
||||
is now only about 290 bytes. This also makes the class faster.
|
||||
|
||||
- Added the ability to integrate delta functions on the right-hand side (by
|
||||
sampling the test function at the center of the delta coefficient). Currently
|
||||
this is supported in the DomainLFIntegrator, VectorDomainLFIntegrator and
|
||||
VectorFEDomainLFIntegrator classes.
|
||||
|
||||
- Added five new linear interpolators in fem/bilininteg.cpp to compute products
|
||||
of scalar and vector fields or products with arbitrary coefficients.
|
||||
|
||||
- Added matrix coefficient support to CurlCurlIntegrator.
|
||||
|
||||
- Extend the method NodalFiniteElement::Project for VectorCoefficient to work
|
||||
with arbitrary number of vector components.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Added a .gitignore file that ignores all files erased by "make distclean",
|
||||
i.e. the files that can be generated from the source but we don't want to
|
||||
track in the repository, as well as a few platform-specific files.
|
||||
|
||||
- Added Linux, Mac and Windows CI testing on GitHub with Travis CI and Appveyor.
|
||||
|
||||
- Added a new macro, MFEM_VERSION, defined as a single integer of the form
|
||||
(major*100 + minor)*100 + patch. The convention is that an even number
|
||||
(i.e. even patch number) denotes a "release" version, while an odd number
|
||||
denotes a "development" version. See config/config.hpp.in.
|
||||
|
||||
- Added an option for building in parallel without a METIS dependency. This is
|
||||
used for example the Laghos miniapp, https://github.com/CEED/Laghos.
|
||||
|
||||
- Modified the installation layout: all headers, except the master headers
|
||||
(mfem.hpp and mfem-performance.hpp), are installed in <PREFIX>/include/mfem;
|
||||
the master headers are installed in both <PREFIX>/include/mfem and in
|
||||
<PREFIX>/include. The mfem configuration and testing makefiles (config.mk and
|
||||
test.mk) are installed in <PREFIX>/share/mfem, instead of <PREFIX>.
|
||||
|
||||
- Add three more options for MFEM_TIMER_TYPE.
|
||||
|
||||
- Support independent number of digits for cycle and rank in DataCollection.
|
||||
|
||||
- Converted Sidre usage from "asctoolkit" to "axom" namespace.
|
||||
|
||||
- Various small fixes and styling updates.
|
||||
|
||||
API changes
|
||||
-----------
|
||||
- The methods GetCoeff of VectorArrayCoefficient and MatrixArrayCoefficient now
|
||||
return a pointer to Coefficient (instead of reference). Note that NULL pointer
|
||||
is a valid entry for these two classes - it is treated as the zero function.
|
||||
|
||||
- When building with PETSc, the required PETSc version is now 3.8.0. Newer
|
||||
versions may work too, as long as there are no interface changes in PETSc.
|
||||
|
||||
- The class GeometryRefiner now uses the enum in Quadrature1D for its type
|
||||
specification. In particular, this will affect older versions of GLVis. A
|
||||
simple upgrade to the latest version of GLVis should resolve this issue.
|
||||
- Add a block lower triangular preconditioner in using a matrix-free
|
||||
implementation, see class BlockLowerTriangularPreconditioner.
|
||||
|
||||
|
||||
Version 3.3, released on Jan 28, 2017
|
||||
@@ -436,7 +173,7 @@ Improved file output
|
||||
- Added experimental support for an HDF5-based output file format following the
|
||||
Conduit (https://github.com/LLNL/conduit) mesh blueprint specification for
|
||||
visualization and/or restart capability. This functionality is aimed primarily
|
||||
at user of LLNL's axom project (Sidre component) that run problems at extreme
|
||||
at user of LLNL's ASC Toolkit (Sidre component) that run problems at extreme
|
||||
scales. Users desiring a small scale binary format may want to look at the
|
||||
gzstream functionality instead.
|
||||
|
||||
|
||||
+46
-133
@@ -38,14 +38,8 @@ if (NOT CMAKE_CXX_COMPILER)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# Project name and version
|
||||
#-------------------------------------------------------------------------------
|
||||
project(mfem NONE)
|
||||
# Current version of MFEM, see also `makefile`.
|
||||
# mfem_VERSION = (string)
|
||||
# MFEM_VERSION = (int) [automatically derived from mfem_VERSION]
|
||||
set(${PROJECT_NAME}_VERSION 3.3.3)
|
||||
project(mfem CXX)
|
||||
set(${PROJECT_NAME}_VERSION 3.3)
|
||||
|
||||
# Prohibit in-source build
|
||||
if (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
|
||||
@@ -53,65 +47,18 @@ if (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
|
||||
"MFEM does not support in-source CMake builds at this time.")
|
||||
endif (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
|
||||
|
||||
# Set xSDK defaults.
|
||||
set(USE_XSDK_DEFAULTS_DEFAULT OFF)
|
||||
set(XSDK_ENABLE_CXX ON)
|
||||
set(XSDK_ENABLE_C OFF)
|
||||
set(XSDK_ENABLE_Fortran OFF)
|
||||
|
||||
# Check if we need to enable C or Fortran.
|
||||
if (CMAKE_VERSION VERSION_LESS 3.2 OR
|
||||
MFEM_USE_CONDUIT OR
|
||||
MFEM_USE_SIDRE OR
|
||||
MFEM_USE_PETSC)
|
||||
if (CMAKE_VERSION VERSION_LESS 3.2 OR MFEM_USE_SIDRE)
|
||||
# This seems to be needed by:
|
||||
# * find_package(BLAS REQUIRED) and
|
||||
# * find_package(HDF5 REQUIRED) needed, in turn, by:
|
||||
# - find_package(AXOM REQUIRED)
|
||||
# * find_package(PETSc REQUIRED)
|
||||
set(XSDK_ENABLE_C ON)
|
||||
endif()
|
||||
if (MFEM_USE_STRUMPACK)
|
||||
# Just needed to find the MPI_Fortran libraries to link with
|
||||
set(XSDK_ENABLE_Fortran ON)
|
||||
endif()
|
||||
|
||||
# Include xSDK default CMake file.
|
||||
include("${CMAKE_CURRENT_SOURCE_DIR}/config/XSDKDefaults.cmake")
|
||||
|
||||
# Enable languages.
|
||||
enable_language(CXX)
|
||||
if (XSDK_ENABLE_C)
|
||||
# - find_package(ATK REQUIRED)
|
||||
enable_language(C)
|
||||
endif()
|
||||
if (XSDK_ENABLE_Fortran)
|
||||
enable_language(Fortran)
|
||||
endif()
|
||||
|
||||
# Suppress warnings about MACOSX_RPATH
|
||||
set(CMAKE_MACOSX_RPATH OFF CACHE BOOL "")
|
||||
|
||||
# CMake needs to know where to find things
|
||||
set(MFEM_CMAKE_PATH ${PROJECT_SOURCE_DIR}/config)
|
||||
set(CMAKE_MODULE_PATH ${MFEM_CMAKE_PATH}/cmake/modules)
|
||||
|
||||
# Load MFEM CMake utilities.
|
||||
include(MfemCmakeUtilities)
|
||||
|
||||
string(TOUPPER "${PROJECT_NAME}" PROJECT_NAME_UC)
|
||||
mfem_version_to_int(${${PROJECT_NAME}_VERSION} ${PROJECT_NAME_UC}_VERSION)
|
||||
set(${PROJECT_NAME_UC}_VERSION_STRING ${${PROJECT_NAME}_VERSION})
|
||||
if (EXISTS ${PROJECT_SOURCE_DIR}/.git)
|
||||
execute_process(
|
||||
COMMAND git describe --all --long --abbrev=40 --dirty --always
|
||||
WORKING_DIRECTORY "${PROJECT_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE ${PROJECT_NAME_UC}_GIT_STRING
|
||||
ERROR_QUIET OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
endif()
|
||||
if (NOT ${PROJECT_NAME_UC}_GIT_STRING)
|
||||
set(${PROJECT_NAME_UC}_GIT_STRING "(unknown)")
|
||||
endif()
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# Process configuration options
|
||||
#-------------------------------------------------------------------------------
|
||||
@@ -123,23 +70,25 @@ else()
|
||||
set(MFEM_DEBUG OFF)
|
||||
endif()
|
||||
|
||||
# MPI -> hypre; PETSc (optional)
|
||||
# MPI -> hypre, METIS
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(MPI REQUIRED)
|
||||
set(MPI_CXX_INCLUDE_DIRS ${MPI_CXX_INCLUDE_PATH})
|
||||
# Parallel MFEM depends on hypre
|
||||
include_directories(${MPI_CXX_INCLUDE_PATH})
|
||||
# Parallel MFEM depends on hypre and METIS
|
||||
find_package(HYPRE REQUIRED)
|
||||
set(MFEM_HYPRE_VERSION ${HYPRE_VERSION})
|
||||
include_directories(${HYPRE_INCLUDE_DIRS})
|
||||
find_package(METIS REQUIRED)
|
||||
include_directories(${METIS_INCLUDE_DIRS})
|
||||
if (MFEM_USE_PETSC)
|
||||
find_package(PETSc REQUIRED)
|
||||
message(STATUS "Found PETSc version ${PETSC_VERSION}")
|
||||
if (PETSC_VERSION AND (PETSC_VERSION VERSION_LESS 3.8.0))
|
||||
message(FATAL_ERROR "PETSc version >= 3.8.0 is required")
|
||||
if (PETSC_VERSION AND (PETSC_VERSION VERSION_LESS 3.7.5.99))
|
||||
message(FATAL_ERROR "PETSc version >= 3.7.5.99 is required")
|
||||
endif()
|
||||
set(PETSC_INCLUDE_DIRS ${PETSC_INCLUDES})
|
||||
include_directories(${PETSC_INCLUDES})
|
||||
endif()
|
||||
else()
|
||||
set(PKGS_NEED_MPI SUPERLU PETSC STRUMPACK)
|
||||
set(PKGS_NEED_MPI SUPERLU PETSC)
|
||||
foreach(PKG IN LISTS PKGS_NEED_MPI)
|
||||
if (MFEM_USE_${PKG})
|
||||
message(STATUS "Disabling package ${PKG} - requires MPI")
|
||||
@@ -148,19 +97,17 @@ else()
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_METIS)
|
||||
find_package(METIS REQUIRED)
|
||||
endif()
|
||||
|
||||
# GZSTREAM -> zlib
|
||||
if (MFEM_USE_GZSTREAM)
|
||||
find_package(ZLIB REQUIRED)
|
||||
include_directories(${ZLIB_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# Backtrace with libunwind
|
||||
if (MFEM_USE_LIBUNWIND)
|
||||
set(MFEMBacktrace_REQUIRED_PACKAGES "Libunwind" "LIBDL" "CXXABIDemangle")
|
||||
find_package(MFEMBacktrace REQUIRED)
|
||||
include_directories(${LIBUNWIND_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# BLAS, LAPACK
|
||||
@@ -182,6 +129,7 @@ endif()
|
||||
if (MFEM_USE_SUITESPARSE)
|
||||
find_package(SuiteSparse REQUIRED
|
||||
UMFPACK KLU AMD BTF CHOLMOD COLAMD CAMD CCOLAMD config)
|
||||
include_directories(${SuiteSparse_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# SUNDIALS
|
||||
@@ -192,66 +140,63 @@ if (MFEM_USE_SUNDIALS)
|
||||
find_package(SUNDIALS REQUIRED
|
||||
NVector_Serial NVector_Parallel NVector_ParHyp CVODE ARKODE KINSOL)
|
||||
endif()
|
||||
include_directories(${SUNDIALS_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# Mesquite
|
||||
if (MFEM_USE_MESQUITE)
|
||||
find_package(Mesquite REQUIRED)
|
||||
include_directories(${MESQUITE_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# SuperLU_DIST can only be enabled in parallel
|
||||
# SuperLU_DIST can only be enabled if parallel
|
||||
if (MFEM_USE_SUPERLU)
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(SuperLUDist REQUIRED)
|
||||
include_directories(${SuperLUDist_INCLUDE_DIRS})
|
||||
else()
|
||||
message(FATAL_ERROR " *** SuperLU_DIST requires that MPI be enabled.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# STRUMPACK can only be enabled in parallel
|
||||
if (MFEM_USE_STRUMPACK)
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(STRUMPACK REQUIRED)
|
||||
else()
|
||||
message(FATAL_ERROR " *** STRUMPACK requires that MPI be enabled.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Gecko
|
||||
if (MFEM_USE_GECKO)
|
||||
find_package(Gecko REQUIRED)
|
||||
include_directories(${GECKO_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# GnuTLS
|
||||
if (MFEM_USE_GNUTLS)
|
||||
find_package(_GnuTLS REQUIRED)
|
||||
include_directories(${GNUTLS_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# NetCDF
|
||||
if (MFEM_USE_NETCDF)
|
||||
find_package(NetCDF REQUIRED)
|
||||
include_directories(${NETCDF_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# MPFR
|
||||
if (MFEM_USE_MPFR)
|
||||
find_package(MPFR REQUIRED)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_CONDUIT)
|
||||
find_package(Conduit REQUIRED conduit relay blueprint )
|
||||
include_directories(${MPFR_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# Axom/Sidre
|
||||
if (MFEM_USE_SIDRE)
|
||||
find_package(Axom REQUIRED Sidre SLIC axom_utils)
|
||||
if (NOT MFEM_USE_MPI)
|
||||
find_package(ATK REQUIRED Sidre SLIC common)
|
||||
else()
|
||||
find_package(ATK REQUIRED Sidre SPIO SLIC common)
|
||||
endif()
|
||||
include_directories(${ATK_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# MFEM_TIMER_TYPE
|
||||
if (NOT DEFINED MFEM_TIMER_TYPE)
|
||||
if (APPLE)
|
||||
# use std::clock from <ctime> for UserTime and
|
||||
# use mach_absolute_time from <mach/mach_time.h> for RealTime
|
||||
set(MFEM_TIMER_TYPE 4)
|
||||
set(MFEM_TIMER_TYPE 0) # use std::clock from <ctime>
|
||||
elseif (WIN32)
|
||||
set(MFEM_TIMER_TYPE 3) # QueryPerformanceCounter from <windows.h>
|
||||
else()
|
||||
@@ -265,27 +210,17 @@ if (NOT DEFINED MFEM_TIMER_TYPE)
|
||||
endif()
|
||||
|
||||
# List all possible libraries in order of dependencies.
|
||||
# [METIS < SuiteSparse]:
|
||||
# With newer versions of SuiteSparse which include METIS header using 64-bit
|
||||
# integers, the METIS header (with 32-bit indices, as used by mfem) needs to
|
||||
# be before SuiteSparse.
|
||||
set(MFEM_TPLS MPI_CXX OPENMP BLAS LAPACK METIS HYPRE SuiteSparse SUNDIALS PETSC
|
||||
MESQUITE SuperLUDist STRUMPACK AXOM CONDUIT GECKO GNUTLS NETCDF MPFR POSIXCLOCKS
|
||||
set(MFEM_TPLS HYPRE OPENMP SUNDIALS MESQUITE SuiteSparse SuperLUDist
|
||||
ParMETIS METIS LAPACK BLAS GECKO GNUTLS NETCDF PETSC MPFR ATK POSIXCLOCKS
|
||||
MFEMBacktrace ZLIB)
|
||||
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
|
||||
set(TPL_LIBRARIES "")
|
||||
set(TPL_INCLUDE_DIRS "")
|
||||
foreach(TPL IN LISTS MFEM_TPLS)
|
||||
if (${TPL}_FOUND)
|
||||
message(STATUS "MFEM: using package ${TPL}")
|
||||
list(APPEND TPL_LIBRARIES ${${TPL}_LIBRARIES})
|
||||
list(APPEND TPL_INCLUDE_DIRS ${${TPL}_INCLUDE_DIRS})
|
||||
endif()
|
||||
endforeach(TPL)
|
||||
list(REMOVE_DUPLICATES TPL_LIBRARIES)
|
||||
list(REMOVE_DUPLICATES TPL_INCLUDE_DIRS)
|
||||
# message(STATUS "TPL_INCLUDE_DIRS = ${TPL_INCLUDE_DIRS}")
|
||||
include_directories(${TPL_INCLUDE_DIRS})
|
||||
|
||||
if (OPENMP_FOUND)
|
||||
message(STATUS "MFEM: using package OpenMP")
|
||||
@@ -293,8 +228,6 @@ if (OPENMP_FOUND)
|
||||
endif()
|
||||
|
||||
message(STATUS "MFEM build type: CMAKE_BUILD_TYPE = ${CMAKE_BUILD_TYPE}")
|
||||
message(STATUS "MFEM version: v${MFEM_VERSION_STRING}")
|
||||
message(STATUS "MFEM git string: ${MFEM_GIT_STRING}")
|
||||
|
||||
# Windows specific
|
||||
set(_USE_MATH_DEFINES ${WIN32})
|
||||
@@ -304,6 +237,7 @@ set(_USE_MATH_DEFINES ${WIN32})
|
||||
#-------------------------------------------------------------------------------
|
||||
|
||||
# Headers and sources
|
||||
include(MfemCmakeUtilities)
|
||||
set(SOURCES "")
|
||||
set(HEADERS "")
|
||||
set(MFEM_SOURCE_DIRS general linalg mesh fem)
|
||||
@@ -315,30 +249,21 @@ set(MASTER_HEADERS
|
||||
${PROJECT_SOURCE_DIR}/mfem.hpp
|
||||
${PROJECT_SOURCE_DIR}/mfem-performance.hpp)
|
||||
|
||||
set(_lib_path "${CMAKE_INSTALL_PREFIX}/lib")
|
||||
set(CMAKE_INSTALL_RPATH_USE_LINK_PATH ON CACHE BOOL "")
|
||||
set(CMAKE_INSTALL_RPATH "${_lib_path}" CACHE PATH "")
|
||||
set(CMAKE_INSTALL_NAME_DIR "${_lib_path}" CACHE PATH "")
|
||||
|
||||
# Declaring the library
|
||||
add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
|
||||
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
|
||||
list(REMOVE_DUPLICATES TPL_LIBRARIES)
|
||||
# message(STATUS " TPL_LIBRARIES = ${TPL_LIBRARIES}")
|
||||
if (CMAKE_VERSION VERSION_GREATER 2.8.11)
|
||||
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES})
|
||||
else()
|
||||
target_link_libraries(mfem ${TPL_LIBRARIES})
|
||||
endif()
|
||||
if (MINGW)
|
||||
target_link_libraries(mfem ws2_32)
|
||||
endif()
|
||||
set_target_properties(mfem PROPERTIES VERSION "${mfem_VERSION}")
|
||||
set_target_properties(mfem PROPERTIES SOVERSION "${mfem_VERSION}")
|
||||
|
||||
# If building out-of-source, define MFEM_BUILD_DIR to point to the build
|
||||
# directory.
|
||||
if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
|
||||
target_compile_definitions(mfem PRIVATE
|
||||
"MFEM_BUILD_DIR=${PROJECT_BINARY_DIR}")
|
||||
"-DMFEM_BUILD_DIR=${PROJECT_BINARY_DIR}")
|
||||
endif()
|
||||
|
||||
# Generate configuration file in the build directory: config/_config.hpp.
|
||||
@@ -356,11 +281,6 @@ if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
|
||||
"// Auto-generated file.
|
||||
#define MFEM_BUILD_DIR ${PROJECT_BINARY_DIR}
|
||||
#include \"${PROJECT_SOURCE_DIR}/${Header}\"
|
||||
")
|
||||
# This version will be installed in the top include directory:
|
||||
file(WRITE "${PROJECT_BINARY_DIR}/InstallHeaders/${Header}"
|
||||
"// Auto-generated file.
|
||||
#include \"mfem/${Header}\"
|
||||
")
|
||||
endforeach()
|
||||
endif()
|
||||
@@ -412,12 +332,12 @@ endif()
|
||||
# Add 'check' target - quick test
|
||||
if (NOT MFEM_USE_MPI)
|
||||
add_custom_target(check
|
||||
${CMAKE_CTEST_COMMAND} -R '^ex1_ser' -C ${CMAKE_CFG_INTDIR}
|
||||
${CMAKE_CTEST_COMMAND} -R ex1_ser -E performance -C ${CMAKE_CFG_INTDIR}
|
||||
USES_TERMINAL)
|
||||
add_dependencies(check ex1)
|
||||
else()
|
||||
add_custom_target(check
|
||||
${CMAKE_CTEST_COMMAND} -R '^ex1p' -C ${CMAKE_CFG_INTDIR}
|
||||
${CMAKE_CTEST_COMMAND} -R ex1p -E performance -C ${CMAKE_CFG_INTDIR}
|
||||
USES_TERMINAL)
|
||||
add_dependencies(check ex1p)
|
||||
endif()
|
||||
@@ -432,6 +352,7 @@ add_subdirectory(doc)
|
||||
#-------------------------------------------------------------------------------
|
||||
|
||||
message(STATUS "CMAKE_INSTALL_PREFIX = ${CMAKE_INSTALL_PREFIX}")
|
||||
string(TOUPPER "${PROJECT_NAME}" PROJECT_NAME_UC)
|
||||
set(INSTALL_INCLUDE_DIR include
|
||||
CACHE PATH "Relative path for installing header files.")
|
||||
set(INSTALL_LIB_DIR lib
|
||||
@@ -451,15 +372,11 @@ install(TARGETS ${PROJECT_NAME}
|
||||
DESTINATION ${INSTALL_LIB_DIR})
|
||||
|
||||
# Install the master headers
|
||||
foreach(Header mfem.hpp mfem-performance.hpp)
|
||||
install(FILES ${PROJECT_BINARY_DIR}/InstallHeaders/${Header}
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR})
|
||||
endforeach()
|
||||
install(FILES ${MASTER_HEADERS} DESTINATION ${INSTALL_INCLUDE_DIR}/mfem)
|
||||
install(FILES ${MASTER_HEADERS} DESTINATION ${INSTALL_INCLUDE_DIR})
|
||||
|
||||
# Install the headers; currently, the miniapps headers are excluded
|
||||
install(DIRECTORY ${MFEM_SOURCE_DIRS}
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}
|
||||
FILES_MATCHING PATTERN "*.hpp")
|
||||
|
||||
# Install ${HEADERS}
|
||||
@@ -472,11 +389,11 @@ install(DIRECTORY ${MFEM_SOURCE_DIRS}
|
||||
|
||||
# Install the configuration header files
|
||||
install(FILES ${PROJECT_BINARY_DIR}/config/_config.hpp
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem/config
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/config
|
||||
RENAME config.hpp)
|
||||
|
||||
install(FILES ${PROJECT_SOURCE_DIR}/config/tconfig.hpp
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem/config)
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/config)
|
||||
|
||||
# Package the whole thing up nicely
|
||||
include(CMakePackageConfigHelpers)
|
||||
@@ -486,15 +403,11 @@ export(TARGETS ${PROJECT_NAME}
|
||||
FILE "${PROJECT_BINARY_DIR}/MFEMTargets.cmake")
|
||||
|
||||
# Export the package for use from the build-tree (this registers the build-tree
|
||||
# with the CMake user package registry.)
|
||||
# TODO: How do we register the install-tree? Replacing the build-tree?
|
||||
# with a global CMake-registry)
|
||||
export(PACKAGE ${PROJECT_NAME})
|
||||
|
||||
# Extract the include directories required to use MFEM
|
||||
get_target_property(MFEM_TPL_INCLUDE_DIRS mfem INCLUDE_DIRECTORIES)
|
||||
if (NOT MFEM_TPL_INCLUDE_DIRS)
|
||||
set(MFEM_TPL_INCLUDE_DIRS "")
|
||||
endif()
|
||||
|
||||
# This is the build-tree version
|
||||
set(INCLUDE_INSTALL_DIRS ${PROJECT_BINARY_DIR} ${MFEM_TPL_INCLUDE_DIRS})
|
||||
|
||||
-437
@@ -1,437 +0,0 @@
|
||||
# How to Contribute
|
||||
|
||||
The MFEM team welcomes contributions at all levels: bugfixes; code
|
||||
improvements; simplifications; new mesh, discretization or solver
|
||||
capabilities; improved documentation; new examples and miniapps;
|
||||
HPC performance improvements; ...
|
||||
|
||||
Use a pull request (PR) toward the `mfem:master` branch to propose your
|
||||
contribution. If you are planning significant code changes, or have any
|
||||
questions, you can also open an [issue](https://github.com/mfem/mfem/issues)
|
||||
before issuing a PR. We also welcome your [simulation
|
||||
images](http://mfem.org/gallery/), which you can submit via a pull request in
|
||||
[mfem/web](https://github.com/mfem/web).
|
||||
|
||||
See the [Quick Summary](#quick-summary) section for the main highlights of our
|
||||
GitHub workflow. For more details, consult the following sections and refer
|
||||
back to them before issuing pull requests:
|
||||
|
||||
- [GitHub Workflow](#github-workflow)
|
||||
- [MFEM Organization](#mfem-organization)
|
||||
- [New Feature Development](#new-feature-development)
|
||||
- [Developer Guidelines](#developer-guidelines)
|
||||
- [Pull Requests](#pull-requests)
|
||||
- [Pull Request Checklist](#pull-request-checklist)
|
||||
- [Master/Next Workflow](#masternext-workflow)
|
||||
- [Releases](#releases)
|
||||
- [Release Checklist](#release-checklist)
|
||||
- [LLNL Workflow](#llnl-workflow)
|
||||
- [Automated Testing](#automated-testing)
|
||||
- [Contact Information](#contact-information)
|
||||
|
||||
Contributing to MFEM requires knowledge of Git and, likely, finite elements. If
|
||||
you are new to Git, see the [GitHub learning
|
||||
resources](https://help.github.com/articles/git-and-github-learning-resources/).
|
||||
To learn more about the finite element method, see our [FEM page](http://mfem.org/fem).
|
||||
|
||||
*By submitting a pull request, you are affirming the [Developer's Certificate of
|
||||
Origin](#developers-certificate-of-origin-11) at the end of this file.*
|
||||
|
||||
|
||||
## Quick Summary
|
||||
|
||||
- We encourage you to [join the MFEM organization](#mfem-organization) and create
|
||||
development branches off `mfem:master`.
|
||||
- Please follow the [developer guidelines](#developer-guidelines), in particular
|
||||
with regards to documentation and code styling.
|
||||
- Pull requests should be issued toward `mfem:master`. Make sure
|
||||
to check the items off the [Pull Request Checklist](#pull-request-checklist).
|
||||
- After approval, MFEM developers merge the PR manually in the [mfem:next branch](#masternext-workflow).
|
||||
- After a week of testing in `mfem:next`, the original PR is merged in `mfem:master`.
|
||||
- We use [milestones](https://github.com/mfem/mfem/milestones) to coordinate the
|
||||
work on different PRs toward a release.
|
||||
- Don't hesitate to [contact us](#contact-information) if you have any questions.
|
||||
|
||||
|
||||
## GitHub Workflow
|
||||
|
||||
The GitHub organization, https://github.com/mfem, is the main developer hub for
|
||||
the MFEM project.
|
||||
|
||||
If you plan to make contributions or will like to stay up-to-date with changes
|
||||
in the code, *we strongly encourage you to [join the MFEM organization](#mfem-organization)*.
|
||||
|
||||
This will simplify the workflow (by providing you additional permissions), and
|
||||
will allow us to reach you directly with project announcements.
|
||||
|
||||
|
||||
### MFEM Organization
|
||||
|
||||
- Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
+ Create the account at: github.com/join.
|
||||
+ For easy identification, please add your name and maybe a picture of you at: https://github.com/settings/profile.
|
||||
+ To receive notification, set a primary email at: https://github.com/settings/emails.
|
||||
+ For password-less pull/push over SSH, add your SSH keys at: https://github.com/settings/keys.
|
||||
|
||||
- [Contact us](#contact-information) for an invitation to join the MFEM GitHub
|
||||
organization.
|
||||
|
||||
- You should receive an invitation email, which you can directly accept.
|
||||
Alternatively, *after logging into GitHub*, you can accept the invitation at
|
||||
the top of https://github.com/mfem.
|
||||
|
||||
- Consider making your membership public by going to https://github.com/orgs/mfem/people
|
||||
and clicking on the organization visibility dropbox next to your name.
|
||||
|
||||
- Project discussions and announcements will be posted at
|
||||
https://github.com/orgs/mfem/teams/everyone.
|
||||
|
||||
- The MFEM source code is in the [mfem](https://github.com/mfem/mfem)
|
||||
repository.
|
||||
|
||||
- The website and corresponding documentation are in the
|
||||
[web](https://github.com/mfem/web) repository.
|
||||
|
||||
- The [PyMFEM](https://github.com/mfem/PyMFEM) repository contains a Python
|
||||
wrapper for MFEM.
|
||||
|
||||
- The [data](https://github.com/mfem/data) repository contains additional
|
||||
(large) datafiles for MFEM.
|
||||
|
||||
|
||||
### New Feature Development
|
||||
|
||||
- A new feature should be important enough that at least one person, the
|
||||
proposer, is willing to work on it and be its champion.
|
||||
|
||||
- The proposer creates a branch for the new feature (with suffix `-dev`), off
|
||||
the `master` branch, or another existing feature branch, for example:
|
||||
|
||||
```
|
||||
# Clone assuming you have setup your ssh keys on GitHub:
|
||||
git clone git@github.com:mfem/mfem.git
|
||||
|
||||
# Alternatively, clone using the "https" protocol:
|
||||
git clone https://github.com/mfem/mfem.git
|
||||
|
||||
# Create a new feature branch starting from "master":
|
||||
git checkout master
|
||||
git pull
|
||||
git checkout -b feature-dev
|
||||
|
||||
# Work on "feature-dev", add local commits
|
||||
# ...
|
||||
|
||||
# One time only) push the branch to github and setup your local
|
||||
# branch to track the github branch (for "git pull"):
|
||||
git push -u origin feature-dev
|
||||
|
||||
```
|
||||
|
||||
- **We prefer that you create the new feature branch inside the MFEM organization
|
||||
as opposed to in a fork.** This allows everyone in the community to collaborate
|
||||
in one central place.
|
||||
|
||||
- If you prefer to work in your fork, please [enable upstream edits](https://help.github.com/articles/allowing-changes-to-a-pull-request-branch-created-from-a-fork/).
|
||||
|
||||
- Never use the `next` branch to start a new feature branch!
|
||||
|
||||
- The typical feature branch name is `new-feature-dev`, e.g. `pumi-dev`. While
|
||||
not frequent in MFEM, other suffixes are possible, e.g. `-fix`, `-doc`, etc.
|
||||
|
||||
|
||||
### Developer Guidelines
|
||||
|
||||
- *Keep the code lean and as simple as possible*
|
||||
- Well-designed simple code is frequently more general and powerful.
|
||||
- Lean code base is easier to understand by new collaborators.
|
||||
- New features should be added only if they are necessary or generally useful.
|
||||
- Introduction of language constructions not currently used in MFEM should be
|
||||
justified and generally avoided (so we can build on cutting-edge systems).
|
||||
- We prefer basic C++ and the C++03 standard, to keep the code readable by
|
||||
a large audience and to make sure it compiles anywhere.
|
||||
|
||||
- *Keep the code general and reasonably efficient*
|
||||
- Main goal is fast prototyping for research.
|
||||
- When in doubt, generality wins over efficiency.
|
||||
- Respect the needs of different users (current and/or future).
|
||||
|
||||
- *Keep things separate and logically organized*
|
||||
- General usage features go in MFEM (implemented in as much generality as
|
||||
possible), non-general features go into external apps.
|
||||
- Inside MFEM, compartmentalize between linalg, fem, mesh, GLVis, etc.
|
||||
- Contributions that are project-specific or have external dependencies are
|
||||
allowed (if they are of broader interest), but should be `#ifdef`-ed and not
|
||||
change the code by default.
|
||||
|
||||
- Code specifics
|
||||
- All significant new classes, methods and functions have Doxygen-style
|
||||
documentation in source comments.
|
||||
- Consistent code styling is enforced with `make style` in the top-level
|
||||
directory. This requires [Artistic Style](http://astyle.sourceforge.net) (we
|
||||
specifically use version 2.05.1). See also the file `config/mfem.astylerc`.
|
||||
- Use `mfem::out` and `mfem::err` instead of `std::cout` and `std::cerr` in
|
||||
internal library code. (You can use `std` in examples and miniapps.)
|
||||
- When manually resolving conflicts during a merge, make sure to mention the
|
||||
conflicted files in the commit message.
|
||||
|
||||
### Pull Requests
|
||||
|
||||
- When your branch is ready for other developers to review / comment on
|
||||
the code, create a pull request towards `mfem:master`.
|
||||
|
||||
- Pull request typically have titles like:
|
||||
|
||||
`Description [new-feature-dev]`
|
||||
|
||||
for example:
|
||||
|
||||
`Parallel Unstructured Mesh Infrastructure (PUMI) integration [pumi-dev]`
|
||||
|
||||
Note the branch name suffix (in square brackets).
|
||||
|
||||
- Titles may contain a prefix in square brackets to emphasize the type of PR.
|
||||
Common choices are: `[DON'T MERGE]`, `[WIP]` and `[DISCUSS]`, for example:
|
||||
|
||||
`[DISCUSS] Hybridized DG [hdg-dev]`
|
||||
|
||||
- Add a description, appropriate labels and assign yourself to the PR. The MFEM
|
||||
team will add reviewers as appropriate.
|
||||
|
||||
- List outstanding TODO items in the description, see PR #222 for an example.
|
||||
|
||||
- Track the Travis CI and Appveyor [continuous integration](#automated-testing)
|
||||
builds at the end of the PR. These should run clean, so address any errors as
|
||||
soon as possible.
|
||||
|
||||
|
||||
### Pull Request Checklist
|
||||
|
||||
Before a PR can be merged, it should satisfy the following:
|
||||
|
||||
- [ ] Code builds.
|
||||
- [ ] Code passes `make style`.
|
||||
- [ ] Update `CHANGELOG`:
|
||||
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
|
||||
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
|
||||
- [ ] Update `INSTALL`:
|
||||
- [ ] Has a new optional library been added? (*Make sure the external library is licensed under LGPL, not GPL!*)
|
||||
- [ ] Does `make` or `cmake` have a new target?
|
||||
- [ ] Did the requirements or the installation process change? *(rare)*.
|
||||
- [ ] Update `.gitignore`:
|
||||
- [ ] Check if `make distclean; git status` shows any files that are generated from the source but we don't want to track in the repository.
|
||||
- [ ] Add new patterns (just for the new files above) and re-run the above test.
|
||||
- [ ] New examples:
|
||||
- [ ] All sample runs at the top of the example work.
|
||||
- [ ] Update `examples/makefile`:
|
||||
- [ ] Add the example code to the appropriate `SEQ_EXAMPLES` and `PAR_EXAMPLES` variables.
|
||||
- [ ] Add any files generated by it to the `clean` target.
|
||||
- [ ] Add the example binary and any files generated by it to the top-level `.gitignore` file.
|
||||
- [ ] Update `examples/CMakeLists.txt`:
|
||||
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
|
||||
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
|
||||
- [ ] List the new example in `doc/CodeDocumentation.dox`.
|
||||
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
|
||||
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
- [ ] In `examples.md`, list the example under the appropriate categories, add new categories if necessary.
|
||||
- [ ] Add a short description of the example in the "Extensive Examples" section of `features.md`.
|
||||
- [ ] New miniapps:
|
||||
- [ ] All sample runs at the top of the miniapp work.
|
||||
- [ ] Update top-level `makefile` and `makefile` in corresponding miniapp directory.
|
||||
- [ ] Add the miniapp binary and any files generated by it to the top-level `.gitignore` file.
|
||||
- [ ] Update CMake build system:
|
||||
- [ ] Update the `CMakeLists.txt` file in the `miniapps` directory, if the new miniapp is in a new directory.
|
||||
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
|
||||
- [ ] Consider adding a new test for the new miniapp.
|
||||
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
|
||||
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
|
||||
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
|
||||
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
|
||||
- [ ] New capability:
|
||||
- [ ] All significant new classes, methods and functions have Doxygen-style documentation in source comments.
|
||||
- [ ] Consider adding new sample runs in existing examples to highlight the new capability.
|
||||
- [ ] Consider saving cool simulation pictures with the new capability in the Confluence gallery (LLNL only) or submitting them, via pull request, to the gallery section of the `mfem/web` repo.
|
||||
- [ ] If this is a major new feature, consider mentioning in the short summary inside `README` *(rare)*.
|
||||
- [ ] List major new classes in `doc/CodeDocumentation.dox` *(rare)*.
|
||||
- [ ] Update this checklist, if the new pull request affects it.
|
||||
- [ ] (LLNL only) Clone the `tests` repository and run the following tests, see `mfem/tests/README.md`:
|
||||
- [ ] `compilers`
|
||||
- [ ] `memcheck`
|
||||
- [ ] `unit-test`
|
||||
- [ ] `documentation`
|
||||
- [ ] (LLNL only) After merging:
|
||||
- [ ] Regenerate `README.html` files from companion documentation pull requests.
|
||||
- [ ] Update the `baseline` and `compiler` tests, add new tests if necessary.
|
||||
- [ ] Consider updating the script `mfem/tests/sample-runs` (`sample-runs-serial` and `sample-runs-parallel`).
|
||||
|
||||
### Master/Next Workflow
|
||||
|
||||
MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
|
||||
- The `master` branch should always be of release quality and changes should not
|
||||
be merged until they have been fully tested. This branch is protected, and
|
||||
changes can only be made through pull requests.
|
||||
|
||||
- After approval, a pull request is merged manually (by MFEM developers) in the
|
||||
`next` branch for testing and the `in-next` label is added to the PR.
|
||||
This can be done as follows:
|
||||
|
||||
```
|
||||
# Pull the latest version of the "feature-dev" branch
|
||||
git checkout feature-dev
|
||||
git pull
|
||||
|
||||
# Pull the latest version of the "next" branch
|
||||
git checkout next
|
||||
git pull
|
||||
|
||||
# Merge "feature-dev" into "next", resolving conflicts, if necessary.
|
||||
# Use the "--no-ff" flag to create a new commit with merge message.
|
||||
git merge --no-ff feature-dev
|
||||
|
||||
# Push the "next" branch to the server
|
||||
git push
|
||||
```
|
||||
|
||||
- After a week of testing in `next` (excluding bugfixes), both on GitHub, as
|
||||
well as [internally](#tests-at-llnl) at LLNL, the original PR is merged into
|
||||
`master` (provided there are no issues).
|
||||
|
||||
- After the merge, the feature branch is deleted (unless it is a long-term
|
||||
project with periodic PRs).
|
||||
|
||||
- The `next` branch is used just for integrated testing of all PRs approved for
|
||||
merging into `master` to verify that each works individually and that all of
|
||||
them work as a group. This branch can be discarded at any time, though we
|
||||
typically do that only at the end of a [release cycle](#releases).
|
||||
|
||||
|
||||
### Releases
|
||||
|
||||
- Releases are just tags in the `master` branch, e.g. https://github.com/mfem/mfem/releases/tag/v3.3.2,
|
||||
and have a version that ends in an even "patch" number, e.g. `v3.2.2` or
|
||||
`v3.4` (by convention `v3.4` is the same as `v3.4.0`.) Between releases, the
|
||||
version ends in an odd "patch" number, e.g. `v3.3.3`.
|
||||
|
||||
- We use [milestones](https://github.com/mfem/mfem/milestones) to coordinate the
|
||||
work on different PRs toward a release, see for example the
|
||||
[v3.3.2 release](https://github.com/mfem/mfem/milestone/1?closed=1).
|
||||
|
||||
- After a release is complete, the `next` branch is recreated, e.g. as follows
|
||||
(replace `3.3.2` with current release):
|
||||
- Rename the current `next` branch to `next-pre-v3.3.2`.
|
||||
- Create a new `next` branch starting from the `v3.3.2` release.
|
||||
- Local copies of `next` can then be updated with `git checkout -B next origin/next`.
|
||||
|
||||
### Release Checklist
|
||||
|
||||
- [ ] Update the MFEM version in the following files:
|
||||
- [ ] `CHANGELOG`
|
||||
- [ ] `makefile`
|
||||
- [ ] `CMakeLists.txt`
|
||||
- [ ] `doc/CodeDocumentation.conf`
|
||||
- [ ] (LLNL only) Make sure all `README.html` files in the source repo are up to date.
|
||||
- [ ] Tag the repository:
|
||||
|
||||
```
|
||||
git tag -a v3.1 -m "Official release v3.1"
|
||||
git push origin v3.1
|
||||
```
|
||||
- [ ] Create the release tarball and push to `mfem/releases`.
|
||||
- [ ] Recreate the `next` branch as described in previous section.
|
||||
- [ ] Update and push documentation to `mfem/doxygen`.
|
||||
- [ ] Update URL shorlinks:
|
||||
- [ ] Create a shortlink at [https://goo.gl/](https://goo.gl/) for the release tarball, e.g. http://mfem.github.io/releases/mfem-3.1.tgz.
|
||||
- [ ] (LLNL only) Add and commit the new shorlink in the `links` and `links-mfem` files of the internal `mfem/downloads` repo.
|
||||
- [ ] Add the new shortlinks to the MFEM packages in `spack`, `homebrew/science`, `VisIt`, etc.
|
||||
- [ ] Update website in `mfem/web` repo:
|
||||
- Update version and shortlinks in `src/index.md` and `src/download.md`.
|
||||
- Use [cloc-1.62.pl](http://cloc.sourceforge.net/) and `ls -lh` to estimate the SLOC and the tarball size in `src/download.md`.
|
||||
|
||||
|
||||
## LLNL Workflow
|
||||
|
||||
- The GitHub `master` and `next` branches are mirrored to the LLNL institutional
|
||||
Bitbucket repository as `gh-master` and `gh-next`.
|
||||
|
||||
- `gh-master` is merged into LLNL's internal `master` through pull requests; write
|
||||
permissions to `master` are restricted to ensure this is the only way in which it
|
||||
gets updated.
|
||||
|
||||
- We never push directly from LLNL to GitHub.
|
||||
|
||||
- Versions of the code on LLNL's internal server, from most to least stable:
|
||||
- MFEM official release on mfem.org -- Most stable, tested in many apps.
|
||||
- `mfem:master` -- Recent development version, guaranteed to work.
|
||||
- `mfem:gh-master` -- Stable development version, passed testing, you can use
|
||||
it to build your code between releases.
|
||||
- `mfem:gh-next` -- Bleeding-edge development version, may be broken, use at
|
||||
your own risk.
|
||||
|
||||
|
||||
## Automated Testing
|
||||
|
||||
MFEM has several levels of automated testing running on GitHub, as well as on
|
||||
local Mac and Linux workstations, and Livermore Computing clusters at LLNL.
|
||||
|
||||
### Linux and Mac smoke tests
|
||||
We use Travis CI to drive the default tests on the `master` and `next`
|
||||
branches. See the `.travis` file and the logs at
|
||||
[https://travis-ci.org/mfem/mfem](https://travis-ci.org/mfem/mfem).
|
||||
|
||||
Testing using Travis CI should be kept lightweight, as there is a 50 minute time
|
||||
constraint on jobs. Two virtual machines are configured - Mac (OS X) and Linux.
|
||||
|
||||
- Tests on the `master` branch are triggered whenever a PR is issued on this branch.
|
||||
- Tests on the `next` branch are currently scheduled to run each night.
|
||||
|
||||
### Windows smoke test
|
||||
We use Appveyor to test building with the MS Visual C++ compiler in a Windows
|
||||
environment, as well as to test the CMake build. See the `.appveyor` file and the
|
||||
build logs at
|
||||
[https://ci.appveyor.com/project/mfem/mfem](https://ci.appveyor.com/project/mfem/mfem).
|
||||
|
||||
CMake is used to generate the MSVC Project files and drive the build. A release
|
||||
and debug build is performed with a simple run of `ex1` to verify the executable.
|
||||
|
||||
### Tests at LLNL
|
||||
At LLNL, we mirror the `master` and `next` branches internally (to `gh-master`
|
||||
and `gh-next`) and run longer nightly tests via cron. On the weekends, a more
|
||||
extensive test is run which extracts and executes all the different sample runs
|
||||
from each example.
|
||||
|
||||
|
||||
## Contact Information
|
||||
|
||||
- Contact the MFEM team by posting to the [GitHub issue tracker](https://github.com/mfem/mfem).
|
||||
Please perform a search to make sure your question has not been answered already.
|
||||
|
||||
- Email communications should be sent to the MFEM developers mailing list,
|
||||
mfem-dev@llnl.gov.
|
||||
|
||||
|
||||
## [Developer's Certificate of Origin 1.1](https://developercertificate.org/)
|
||||
|
||||
By making a contribution to this project, I certify that:
|
||||
|
||||
(a) The contribution was created in whole or in part by me and I have the right
|
||||
to submit it under the open source license indicated in the file; or
|
||||
|
||||
(b) The contribution is based upon previous work that, to the best of my
|
||||
knowledge, is covered under an appropriate open source license and I have
|
||||
the right under that license to submit that work with modifications, whether
|
||||
created in whole or in part by me, under the same open source license
|
||||
(unless I am permitted to submit under a different license), as indicated in
|
||||
the file; or
|
||||
|
||||
(c) The contribution was provided directly to me by some other person who
|
||||
certified (a), (b) or (c) and I have not modified it.
|
||||
|
||||
(d) I understand and agree that this project and the contribution are public and
|
||||
that a record of the contribution (including all personal information I
|
||||
submit with it, including my sign-off) is maintained indefinitely and may be
|
||||
redistributed consistent with this project or the open source license(s)
|
||||
involved.
|
||||
@@ -18,19 +18,13 @@ requires an MPI C++ compiler, as well as the following external libraries:
|
||||
- METIS (a family of multilevel partitioning algorithms)
|
||||
http://glaros.dtc.umn.edu/gkhome/metis/metis/overview
|
||||
|
||||
The METIS dependency can be disabled but that is not generally recommended, see
|
||||
the option MFEM_USE_METIS.
|
||||
|
||||
The library supports two build systems: one based on GNU make, and a second one
|
||||
based on CMake. Both build systems are described below. Some hints for building
|
||||
without GNU make or CMake can be found at the end of this file.
|
||||
|
||||
In addition to the native build systems, MFEM packages are also available in the
|
||||
following package managers:
|
||||
|
||||
- Spack, https://github.com/spack/spack
|
||||
- OpenHPC, http://openhpc.community
|
||||
- Homebrew/Science, https://github.com/Homebrew/homebrew-science
|
||||
In addition to the native build systems, MFEM packages are also available in
|
||||
the Homebrew/Science, https://github.com/Homebrew/homebrew-science, and the
|
||||
Spack, https://github.com/LLNL/spack, package managers.
|
||||
|
||||
|
||||
Quick start with GNU make
|
||||
@@ -144,7 +138,7 @@ check the results from all the serial/parallel MFEM examples and miniapps use:
|
||||
|
||||
Note that by default MFEM uses "mpirun -np" in its test runs (this is also what
|
||||
is used in the sample runs of its examples and miniapps). The MPI launcher can
|
||||
be changed by the user as described in the "Specifying an MPI job launcher"
|
||||
be changed by the user as described in the "Specifying a MPI job launcher"
|
||||
section at the end of this file.
|
||||
|
||||
Running all the tests may take a while. Implementation details about the check
|
||||
@@ -156,8 +150,8 @@ An optional installation of the library and the headers can be performed with
|
||||
make install [PREFIX=<dir>]
|
||||
|
||||
The library will be installed in $(PREFIX)/lib, the headers in
|
||||
$(PREFIX)/include, and the configuration makefile (config.mk) in
|
||||
$(PREFIX)/share/mfem. The PREFIX option can also be set during configuration.
|
||||
$(PREFIX)/include, and the configuration makefile (config.mk) in $(PREFIX).
|
||||
The PREFIX option can also be set during configuration.
|
||||
|
||||
Information about the current build configuration can be viewed using
|
||||
|
||||
@@ -187,8 +181,6 @@ examples/ directory.
|
||||
|
||||
Configuration options (GNU make)
|
||||
================================
|
||||
See the configuration file config/defaults.mk for the default settings.
|
||||
|
||||
Compilers:
|
||||
CXX - C++ compiler, serial build
|
||||
MPICXX - MPI C++ compiler, parallel build
|
||||
@@ -199,14 +191,10 @@ Compiler options:
|
||||
CXXFLAGS - If not set, defined based on the above optimized/debug flags
|
||||
CPPFLAGS - Additional compiler options
|
||||
|
||||
Build options:
|
||||
STATIC - Build a static version of the library (YES/NO), default = YES
|
||||
SHARED - Build a shared version of the library (YES/NO), default = NO
|
||||
|
||||
Installation options:
|
||||
PREFIX - Specify the installation directory. The library (libmfem.a) will be
|
||||
installed in $(PREFIX)/lib, the headers in $(PREFIX)/include, and
|
||||
the configuration makefile (config.mk) in $(PREFIX)/share/mfem.
|
||||
the configuration makefile (config.mk) in $(PREFIX).
|
||||
INSTALL - Specify the install program, e.g /usr/bin/install
|
||||
|
||||
MFEM library features/options (GNU make)
|
||||
@@ -215,21 +203,10 @@ MFEM_USE_MPI = YES/NO
|
||||
Choose parallel/serial build. The parallel build requires proper setup of the
|
||||
HYPRE_* and METIS_* library options, see below.
|
||||
|
||||
MFEM_USE_METIS = YES/NO
|
||||
Enable/disable the use of the METIS library. By default, this option is set
|
||||
to the value of MFEM_USE_MPI. If this option is explicitly disabled in a
|
||||
parallel build, then the only parallel partitioning (domain decomposition)
|
||||
option in the library will be Cartesian partitioning with box meshes, and
|
||||
thus most of the parallel examples and miniapps will fail.
|
||||
|
||||
MFEM_DEBUG = YES/NO
|
||||
Choose debug/optimized build. The debug build enables a number of messages
|
||||
and consistency checks that may simplify bug-hunting.
|
||||
|
||||
MFEM_USE_EXCEPTIONS = YES/NO
|
||||
Enable the use of exceptions. In particular, modifies the default bahavior
|
||||
when errors are encountered: throw an exception, instead of aborting.
|
||||
|
||||
MFEM_USE_LIBUNWIND = YES/NO
|
||||
Use libunwind to print a stacktrace whenever mfem_error is raised. The
|
||||
information printed is enough to determine the line numbers where the
|
||||
@@ -254,16 +231,13 @@ MFEM_USE_MEMALLOC = YES/NO
|
||||
Internal MFEM option: enable batch allocation for some small objects.
|
||||
Recommended value is YES.
|
||||
|
||||
MFEM_TIMER_TYPE = 0/1/2/3/4/5/6/NO
|
||||
MFEM_TIMER_TYPE = 0/1/2/3/NO
|
||||
Specify which library functions to use in the class StopWatch used for
|
||||
measuring time. The available options are:
|
||||
0 - use std::clock from <ctime>, standard C++
|
||||
1 - use times from <sys/times.h>
|
||||
2 - use high-resolution POSIX clocks (see option POSIX_CLOCKS_LIB)
|
||||
3 - use QueryPerformanceCounter from <windows.h>
|
||||
4 - use mach_absolute_time from <mach/mach_time.h> + std::clock (Mac)
|
||||
5 - use gettimeofday from <sys/time.h>
|
||||
6 - use MPI_Wtime from <mpi.h>
|
||||
NO - use option 3 if the compiler macro _WIN32 is defined, 0 otherwise
|
||||
|
||||
MFEM_USE_SUNDIALS = YES/NO
|
||||
@@ -287,12 +261,6 @@ MFEM_USE_SUPERLU = YES/NO
|
||||
SuperLURowLocMatrix a distributed CSR matrix class needed by SuperLU. When
|
||||
enabled, this option uses the SUPERLU_* library options, see below.
|
||||
|
||||
MFEM_USE_STRUMPACK = YES/NO
|
||||
Enable MFEM functionality based on the STRUMPACK sparse direct solver and
|
||||
preconditioner through the STRUMPACKSolver and STRUMPACKRowLocMatrix
|
||||
classes. When enabled, this option uses the STRUMPACK_* library options, see
|
||||
below.
|
||||
|
||||
MFEM_USE_GNUTLS = YES/NO
|
||||
Enable secure socket support in class socketstream, using the auxiliary
|
||||
GnuTLS_* classes, based on the GnuTLS library. This option may be useful in
|
||||
@@ -304,9 +272,6 @@ MFEM_USE_GNUTLS = YES/NO
|
||||
the script 'glvis-keygen.sh' in the main GLVis directory can be used to do
|
||||
that:
|
||||
bash glvis-keygen.sh ["Your Name"] ["Your Email"]
|
||||
In MFEM v3.3.2 and earlier, the secure authentication is based on OpenPGP
|
||||
keys, while later versions use X.509 certificates. The latest version of the
|
||||
script 'glvis-keygen.sh' can be used to generate both types of keys.
|
||||
When MFEM_USE_GNUTLS is enabled, the additional build options, GNUTLS_*, are
|
||||
also used, see below.
|
||||
|
||||
@@ -333,13 +298,6 @@ MFEM_USE_SIDRE = YES/NO
|
||||
specification. When enabled, this option requires installation of HDF5 (see
|
||||
also MFEM_USE_NETCDF), Conduit and LLNL's axom project.
|
||||
|
||||
MFEM_USE_CONDUIT = YES/NO
|
||||
Enables support for converting MFEM Mesh and Grid Function objects to and
|
||||
from Conduit Mesh Blueprint Descriptions (https://github.com/LLNL/conduit/)
|
||||
and support for JSON and Binary I/O via Conduit Relay. This option requires
|
||||
an installation of Conduit. If Conduit was built with HDF5 support, it also
|
||||
requires an installation of HDF5 (see also MFEM_USE_NETCDF).
|
||||
|
||||
MFEM_USE_GZSTREAM = YES/NO
|
||||
Enables use of on-the-fly gzip compressed streams. With this feature enabled
|
||||
(YES), MFEM can compress its output files on-the-fly. In addition, it can
|
||||
@@ -350,7 +308,6 @@ MFEM_USE_GZSTREAM = YES/NO
|
||||
able to properly read an input file if it is gzip compressed. In that case,
|
||||
the solution is to uncompress the file with an external tool (such as gunzip)
|
||||
before attempting to use it with MFEM.
|
||||
When enabled, this option uses the ZLIB_* library options, see below.
|
||||
|
||||
MFEM_BUILD_TAG = (any value)
|
||||
An optional tag to characterize the build. Exported to config/config.mk.
|
||||
@@ -376,8 +333,8 @@ The specific libraries and their options are:
|
||||
URL: http://www.llnl.gov/CASC/hypre
|
||||
Options: HYPRE_OPT, HYPRE_LIB.
|
||||
|
||||
- METIS, used when MFEM_USE_METIS = YES. If using METIS 5, set
|
||||
MFEM_USE_METIS_5 = YES (default is to use METIS 4).
|
||||
- METIS, required for the parallel build, i.e. when MFEM_USE_MPI = YES. If using
|
||||
METIS 5, set MFEM_USE_METIS_5 = YES (default is to use METIS 4).
|
||||
URL: http://glaros.dtc.umn.edu/gkhome/metis/metis/overview
|
||||
Options: METIS_OPT, METIS_LIB.
|
||||
|
||||
@@ -395,10 +352,7 @@ The specific libraries and their options are:
|
||||
Option: POSIX_CLOCKS_LIB (default = -lrt).
|
||||
|
||||
- SUNDIALS (optional), used when MFEM_USE_SUNDIALS = YES.
|
||||
Beginning with MFEM v3.3, SUNDIALS v2.7.0 is supported.
|
||||
Beginning with MFEM v3.3.2, SUNDIALS v3.0.0 is also supported.
|
||||
If MFEM_USE_MPI is enabled, we expect that SUNDIALS is built with support for
|
||||
both MPI and hypre.
|
||||
In parallel we expect that SUNDIALS is built with support for MPI and hypre.
|
||||
URL: http://computation.llnl.gov/projects/sundials/sundials-software
|
||||
Options: SUNDIALS_OPT, SUNDIALS_LIB.
|
||||
|
||||
@@ -417,14 +371,6 @@ The specific libraries and their options are:
|
||||
URL: http://crd-legacy.lbl.gov/~xiaoye/SuperLU
|
||||
Options: SUPERLU_OPT, SUPERLU_LIB.
|
||||
|
||||
- STRUMPACK (optional), used when MFEM_USE_STRUMPACK = YES. Note that STRUMPACK
|
||||
requires the PT-Scotch and Scalapack libraries as well as ParMETIS, which
|
||||
includes METIS 5 in its distribution.
|
||||
The support for STRUMPACK was added in MFEM v3.3.2 and it requires STRUMPACK
|
||||
2.0.0 or later.
|
||||
URL: http://portal.nersc.gov/project/sparse/strumpack
|
||||
Options: STRUMPACK_OPT, STRUMPACK_LIB.
|
||||
|
||||
- GnuTLS (optional), used when MFEM_USE_GNUTLS = YES. On most Linux systems,
|
||||
GnuTLS is available as a development package, e.g. gnutls-devel. On Mac OS X,
|
||||
one can get the library through the Homebrew package manager (http://brew.sh).
|
||||
@@ -438,7 +384,7 @@ The specific libraries and their options are:
|
||||
URL: www.unidata.ucar.edu/software/netcdf
|
||||
Options: NETCDF_OPT, NETCDF_LIB.
|
||||
|
||||
- PETSc (optional), used when MFEM_USE_PETSC = YES. Version 3.8 or higher of
|
||||
- PETSc (optional), used when MFEM_USE_PETSC = YES. Version 3.8 or higher of
|
||||
the PETSC dev branch is required. The MFEM and PETSc builds can share common
|
||||
libraries, e.g., hypre and SUNDIALS. Here's an example configuration, assuming
|
||||
PETSc has been cloned on the same level as mfem and hypre:
|
||||
@@ -455,12 +401,6 @@ The specific libraries and their options are:
|
||||
https://support.hdfgroup.org/HDF5 (HDF5)
|
||||
Options: SIDRE_OPT, SIDRE_LIB.
|
||||
|
||||
- Conduit, used when MFEM_USE_CONDUIT = YES. Direct Conduit Mesh Blueprint
|
||||
support requires Conduit >= v0.3.1 and VisIt >= v2.13.1 to read the output.
|
||||
URL: https://github.com/LLNL/conduit (Conduit)
|
||||
https://support.hdfgroup.org/HDF5 (HDF5)
|
||||
Options: CONDUIT_OPT, CONDUIT_LIB.
|
||||
|
||||
- MPFR (optional), used when MFEM_USE_MPFR = YES.
|
||||
URL: http://mpfr.org, it depends on the GMP library: https://gmplib.org
|
||||
Options: MPFR_OPT, MPFR_LIB.
|
||||
@@ -471,11 +411,6 @@ The specific libraries and their options are:
|
||||
URL: http://www.nongnu.org/libunwind
|
||||
Options: LIBUNWIND_OPT, LIBUNWIND_LIB.
|
||||
|
||||
- ZLIB (optional), used when MFEM_USE_GZSTREAM = YES, or when MFEM_USE_NETCDF =
|
||||
YES (in the default settings for NETCDF_OPT and NETCDF_LIB).
|
||||
URL: https://zlib.net
|
||||
Options: ZLIB_OPT, ZLIB_LIB.
|
||||
|
||||
|
||||
Building with CMake
|
||||
===================
|
||||
@@ -507,12 +442,6 @@ Debug and optimization options are controlled through the CMake variable
|
||||
CMAKE_BUILD_TYPE which can be set to standard values like "Debug", and "Release"
|
||||
(default).
|
||||
|
||||
To use a specific generator use the "-G <generator>" option of cmake:
|
||||
|
||||
cmake <mfem-source-dir> -G "Xcode"
|
||||
cmake <mfem-source-dir> -G "Visual Studio 12 2013"
|
||||
cmake <mfem-source-dir> -G "MinGW Makefiles"
|
||||
|
||||
With CMake it is possible to build MFEM as a shared library using the standard
|
||||
CMake option -DBUILD_SHARED_LIBS=1.
|
||||
|
||||
@@ -520,30 +449,15 @@ Once configured, the library can be built simply with (assuming a UNIX type
|
||||
system, where the default is to generate "UNIX Makefiles")
|
||||
|
||||
make -j 4
|
||||
or
|
||||
cmake --build .
|
||||
or
|
||||
cmake --build . --config Release [Visual Studio, Xcode]
|
||||
|
||||
The build can be quick-tested by running
|
||||
|
||||
make check
|
||||
or
|
||||
cmake --build . --target check
|
||||
or
|
||||
cmake --build . --config Release --target check [Visual Studio, Xcode]
|
||||
|
||||
which will simply compile and run Example 1/1p. For more extensive tests that
|
||||
check the results from all the serial/parallel MFEM examples and miniapps use:
|
||||
|
||||
make exec -j 4
|
||||
make test
|
||||
or
|
||||
cmake --build . --target exec
|
||||
cmake --build . --target test
|
||||
or
|
||||
cmake --build . --config Release --target exec [Visual Studio, Xcode]
|
||||
cmake --build . --config Release --target RUN_TESTS [Visual Studio, Xcode]
|
||||
|
||||
Note that running all the tests may take a while.
|
||||
|
||||
@@ -551,11 +465,6 @@ Installation prefix can be configured by setting the standard CMake variable
|
||||
CMAKE_INSTALL_PREFIX. To install the library, use
|
||||
|
||||
make install
|
||||
or
|
||||
cmake --build . --target install
|
||||
or
|
||||
cmake --build . --config Release --target install [Xcode]
|
||||
cmake --build . --config Release --target INSTALL [Visual Studio]
|
||||
|
||||
The library will be installed in <PREFIX>/lib, the headers in <PREFIX>/include,
|
||||
and the configuration CMake files in <PREFIX>/lib/cmake/mfem.
|
||||
@@ -563,8 +472,6 @@ and the configuration CMake files in <PREFIX>/lib/cmake/mfem.
|
||||
|
||||
Configuration variables (CMake)
|
||||
===============================
|
||||
See the configuration file config/defaults.cmake for the default settings.
|
||||
|
||||
Non-standard CMake variables for compilers:
|
||||
CXX - If set, overwrite the auto-detected C++ compiler, serial build
|
||||
MPICXX - If set, overwrite the auto-detected MPI C++ compiler, parallel build
|
||||
@@ -578,7 +485,6 @@ The following options are equivalent to the GNU make options with the same name:
|
||||
[see "MFEM library features/options (GNU make)" above]
|
||||
|
||||
MFEM_USE_MPI
|
||||
MFEM_USE_METIS - Set to ${MFEM_USE_MPI}, can be overwritten.
|
||||
MFEM_USE_LIBUNWIND
|
||||
MFEM_USE_LAPACK
|
||||
MFEM_THREAD_SAFE
|
||||
@@ -588,7 +494,6 @@ MFEM_TIMER_TYPE - Set automatically, can be overwritten.
|
||||
MFEM_USE_MESQUITE
|
||||
MFEM_USE_SUITESPARSE
|
||||
MFEM_USE_SUPERLU
|
||||
MFEM_USE_STRUMPACK
|
||||
MFEM_USE_GNUTLS
|
||||
MFEM_USE_NETCDF
|
||||
MFEM_USE_MPFR
|
||||
@@ -631,7 +536,7 @@ The CMake build system adds auto-detection for the following packages/libraries:
|
||||
- METIS - The option MFEM_USE_METIS_5 is auto-detected.
|
||||
- MESQUITE
|
||||
- SuiteSparse
|
||||
- SuperLUDist, STRUMPACK
|
||||
- SuperLUDist
|
||||
- ParMETIS
|
||||
- GNUTLS - Extends the built-in CMake support, to search GNUTLS_DIR as well.
|
||||
- NETCDF
|
||||
@@ -653,15 +558,15 @@ Before using another build system (e.g. Visual Studio) it is necessary to create
|
||||
a proper configuration header file, config/config.hpp, using the template from
|
||||
config/config.hpp.in:
|
||||
|
||||
cp config/config.hpp.in config/_config.hpp
|
||||
cp config/config.hpp.in config/config.hpp
|
||||
|
||||
The file config/_config.hpp can then be edited to enable desired options. The
|
||||
The file config/config.hpp can then be edited to enable desired options. The
|
||||
MFEM library is simply a combination of all object files obtained by compiling
|
||||
the .cpp source files in the source directories: general, linalg, mesh, and fem.
|
||||
|
||||
|
||||
Specifying an MPI job launcher
|
||||
==============================
|
||||
Specifying a MPI job launcher
|
||||
=============================
|
||||
By default, MFEM will use 'mpirun -np #' to launch any of its parallel tests or
|
||||
miniapps, where # is the number of MPI tasks. An alternate MPI launcher can be
|
||||
provided by setting the MFEM_MPIEXEC and MFEM_MPIEXEC_NP config variables.
|
||||
|
||||
@@ -60,4 +60,3 @@ This project is released under the LGPL v2.1 license. See LICENSE file for full
|
||||
details.
|
||||
|
||||
LLNL Release Number: LLNL-CODE-443211
|
||||
DOI: 10.11578/dc.20171025.1248
|
||||
|
||||
@@ -1,31 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_ALL_HPP
|
||||
#define MFEM_BACKENDS_ALL_HPP
|
||||
|
||||
#include "../config/config.hpp"
|
||||
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "base/backend.hpp"
|
||||
|
||||
#ifdef MFEM_USE_OCCA
|
||||
#include "occa/backend.hpp"
|
||||
#endif
|
||||
|
||||
#ifdef MFEM_USE_OMP
|
||||
#include "omp/backend.hpp"
|
||||
#endif
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_ALL_HPP
|
||||
@@ -1,215 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_ARRAY_HPP
|
||||
#define MFEM_BACKENDS_BASE_ARRAY_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "layout.hpp"
|
||||
#include "utils.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Extension to the template class Array<T>
|
||||
class PArray : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Layout with shared ownership (smart pointer)
|
||||
DLayout layout;
|
||||
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
virtual void *DoGetData() const = 0;
|
||||
|
||||
/** @brief Create and return a new array (in @a *clone) of the same dynamic
|
||||
type as this array using the same layout and ItemSize().
|
||||
|
||||
Set @a *clone to NULL if allocation fails.
|
||||
|
||||
If @a copy_data is true, the contents of this array is copied to the new
|
||||
array; otherwise, the new array remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the array data of the newly created
|
||||
object (in @a *buffer) , if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
virtual PArray *DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const = 0;
|
||||
|
||||
/// Resize the array, reallocating its data if necessary.
|
||||
/** If @a buffer is not NULL, return the array data (in @a *buffer), if it
|
||||
is stored as a contiguous array on the host; otherwise, set @a *buffer to
|
||||
NULL. Returns 0 on success and non-zero otherwise, e.g. if memory
|
||||
allocation fails.
|
||||
|
||||
If the @a new_layout is not supported, a non-zero error code will be
|
||||
returned.
|
||||
|
||||
The @a new_layout has to be valid, i.e. new_layout != NULL and
|
||||
new_layout->HasEngine() == true.
|
||||
|
||||
@note If reallocation is performed, the previous content of the array is
|
||||
NOT copied to the new location. */
|
||||
virtual int DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size) = 0;
|
||||
|
||||
/** @brief Get access to the contents of the array in host memory, as a
|
||||
contiguous array. */
|
||||
/** If the array data is stored as a contiguous array in host memory, return
|
||||
a pointer to it. Otherwise, copy the data to @a buffer (if @a buffer is
|
||||
not NULL) and return @a buffer.
|
||||
@note If not NULL, @a buffer is assumed to be of size greater than or
|
||||
equal to Size(). */
|
||||
virtual void *DoPullData(void *buffer, std::size_t item_size) = 0;
|
||||
|
||||
/** @brief Set all entries of the array to the (single) value pointed to by
|
||||
@a value_ptr. */
|
||||
virtual void DoFill(const void *value_ptr, std::size_t item_size) = 0;
|
||||
|
||||
/** @brief Set all Size() entries of the array from the given contiguous
|
||||
array, @a src_buffer, on the host. */
|
||||
virtual void DoPushData(const void *src_buffer, std::size_t item_size) = 0;
|
||||
|
||||
/// Copy the data from @a src to @a *this.
|
||||
/** Both arrays must have the same dynamic type, layout, and item_size. */
|
||||
virtual void DoAssign(const PArray &src, std::size_t item_size) = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
public:
|
||||
/** @brief The @a layout parameter will be reference counted and therefore it
|
||||
should be dynamically allocated. */
|
||||
/** The @a layout must be valid in the sense that layout != NULL and
|
||||
layout->HasEngine() == true. */
|
||||
PArray(PLayout &p_layout)
|
||||
: layout(&p_layout)
|
||||
{
|
||||
MFEM_ASSERT(layout && layout->HasEngine(), "invalid layout");
|
||||
}
|
||||
|
||||
virtual ~PArray() { }
|
||||
|
||||
/// Get the current size of the array.
|
||||
std::size_t Size() const { return layout->Size(); }
|
||||
|
||||
/// Get the current layout of the array.
|
||||
PLayout &GetLayout() const { return *layout; }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return dynamic_cast<derived_t&>(*this); }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return dynamic_cast<const derived_t&>(*this); }
|
||||
|
||||
// TODO: Error handling ... handle errors at the Engine level, at the class
|
||||
// level, or at the method level?
|
||||
|
||||
// TODO: Asynchronous execution interface ...
|
||||
|
||||
|
||||
/**
|
||||
@name Public virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
template <typename T=void>
|
||||
T* GetData() const { return (T*) DoGetData(); }
|
||||
|
||||
/** @brief Create and return a new array (in @a *clone) of the same dynamic
|
||||
type as this array using the same layout and ItemSize().
|
||||
|
||||
Set @a *clone to NULL if allocation fails.
|
||||
|
||||
If @a copy_data is true, the contents of this array is copied to the new
|
||||
array; otherwise, the new array remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the array data of the newly created
|
||||
object (in @a *buffer) , if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
template <typename T>
|
||||
DArray Clone(bool copy_data, T **buffer) const
|
||||
{ return DArray(DoClone(copy_data, (void**)buffer, sizeof(T))); }
|
||||
|
||||
/// Resize the array, reallocating its data if necessary.
|
||||
/** If @a buffer is not NULL, return the array data (in @a *buffer), if it
|
||||
is stored as a contiguous array on the host; otherwise, set @a *buffer to
|
||||
NULL. Returns 0 on success and non-zero otherwise, e.g. if memory
|
||||
allocation fails.
|
||||
|
||||
If the @a new_layout is not supported, a non-zero error code will be
|
||||
returned.
|
||||
|
||||
The @a new_layout has to be valid, i.e. new_layout != NULL and
|
||||
new_layout->HasEngine() == true.
|
||||
|
||||
@note If reallocation is performed, the previous content of the array is
|
||||
NOT copied to the new location. */
|
||||
template <typename T>
|
||||
int Resize(PLayout &new_layout, T **buffer)
|
||||
{ return DoResize(new_layout, (void**)buffer, sizeof(T)); }
|
||||
|
||||
/// Shortcut for Resize(*layout, buffer).
|
||||
/** This method is useful for updating the array after its layout is changed
|
||||
externally. */
|
||||
template <typename T>
|
||||
int Update(T **buffer)
|
||||
{ return DoResize(*layout, (void**)buffer, sizeof(T)); }
|
||||
|
||||
/// Shortcut for layout->Resize(new_size) followed by Update()
|
||||
template <typename T>
|
||||
int Resize(std::size_t new_size, T **buffer)
|
||||
{ layout->Resize(new_size); return Update(buffer); }
|
||||
|
||||
/** @brief Get access to the contents of the array in host memory, as a
|
||||
contiguous array. */
|
||||
/** If the array data is stored as a contiguous array in host memory, return
|
||||
a pointer to it. Otherwise, copy the data to @a buffer (if @a buffer is
|
||||
not NULL) and return @a buffer.
|
||||
@note If not NULL, @a buffer is assumed to be of size greater than or
|
||||
equal to Size(). */
|
||||
template <typename T>
|
||||
T *PullData(T *buffer)
|
||||
{ return Size() ? (T*)DoPullData((void*)buffer, sizeof(T)) : NULL; }
|
||||
|
||||
/** @brief Set all entries of the array to the (single) value pointed to by
|
||||
@a value_ptr. */
|
||||
template <typename T>
|
||||
void Fill(const T &value) { if (Size()) { DoFill(&value, sizeof(T)); } }
|
||||
|
||||
/** @brief Set all Size() entries of the array from the given contiguous
|
||||
array, @a src_buffer, on the host. */
|
||||
template <typename T>
|
||||
void PushData(const T *src_buffer)
|
||||
{ if (Size()) { DoPushData(src_buffer, sizeof(T)); } }
|
||||
|
||||
/// Copy the data from @a src to @a *this.
|
||||
/** Both arrays must have the same dynamic type, layout, and entry type. */
|
||||
template <typename T>
|
||||
void Assign(const PArray &src) { if (Size()) { DoAssign(src, sizeof(T)); } }
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_ARRAY_HPP
|
||||
@@ -1,57 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_BACKEND_HPP
|
||||
#define MFEM_BACKENDS_BASE_BACKEND_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "memory_resource.hpp"
|
||||
#include "engine.hpp"
|
||||
#include "array.hpp"
|
||||
#include "vector.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
|
||||
#include <string>
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include <mpi.h>
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// TODO
|
||||
class Backend
|
||||
{
|
||||
public:
|
||||
/// TODO
|
||||
virtual ~Backend() { }
|
||||
|
||||
/// TODO
|
||||
virtual bool Supports(const std::string &engine_spec) const = 0;
|
||||
|
||||
/// TODO
|
||||
virtual Engine *Create(const std::string &engine_spec) = 0;
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// TODO
|
||||
virtual Engine *Create(MPI_Comm comm, const std::string &engine_spec) = 0;
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_BACKEND_HPP
|
||||
@@ -1,72 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_BILINEARFORM_HPP
|
||||
#define MFEM_BACKENDS_BASE_BILINEARFORM_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class Vector;
|
||||
class OperatorHandle;
|
||||
class BilinearForm;
|
||||
|
||||
/// TODO: doxygen
|
||||
class PBilinearForm : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Engine with shared ownership
|
||||
SharedPtr<const Engine> engine;
|
||||
/// Not owned.
|
||||
BilinearForm *bform;
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
PBilinearForm(const Engine &e, BilinearForm &bf)
|
||||
: engine(&e), bform(&bf) { }
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~PBilinearForm() { }
|
||||
|
||||
/// Get the associated Engine
|
||||
const Engine &GetEngine() const { return *engine; }
|
||||
|
||||
/// Assemble the PBilinearForm.
|
||||
/** This method is called from the method BilinearForm::Assemble() of the
|
||||
associated BilinearForm #bform.
|
||||
@returns True, if the host assembly should be skipped. */
|
||||
virtual bool Assemble() = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
OperatorHandle &A) = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
Vector &x, Vector &b,
|
||||
OperatorHandle &A, Vector &X, Vector &B,
|
||||
int copy_interior) = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void RecoverFEMSolution(const Vector &X, const Vector &b,
|
||||
Vector &x) = 0;
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_BILINEARFORM_HPP
|
||||
@@ -1,29 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
DFiniteElementSpace Engine::MakeFESpace(FiniteElementSpace &fes) const
|
||||
{
|
||||
return DFiniteElementSpace(new PFiniteElementSpace(*this, fes));
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
@@ -1,210 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_ENGINE_HPP
|
||||
#define MFEM_BACKENDS_BASE_ENGINE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "../../general/scalars.hpp"
|
||||
#include "memory_resource.hpp"
|
||||
#include "smart_pointers.hpp"
|
||||
#include "utils.hpp"
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include <mpi.h>
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// Forward declarations.
|
||||
class Backend;
|
||||
template <typename T> class Array;
|
||||
class Vector;
|
||||
class Operator;
|
||||
class FiniteElementSpace;
|
||||
class LinearForm;
|
||||
class BilinearForm;
|
||||
class MixedBilinearForm;
|
||||
class NonlinearForm;
|
||||
|
||||
|
||||
/// In parallel, each MPI rank will usually create a single engine.
|
||||
class Engine : public RefCounted
|
||||
{
|
||||
protected:
|
||||
Backend *backend; ///< Backend that created the engine. Not owned.
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
MPI_Comm comm; ///< Associated MPI communicator (may be MPI_COMM_NULL).
|
||||
#endif
|
||||
|
||||
/// Number of memory resources used by the Engine.
|
||||
int num_mem_res;
|
||||
/// Number of workers used by the Engine.
|
||||
int num_workers;
|
||||
|
||||
/// Memory resources used by the engine - array of pointers.
|
||||
/** Both the array and the entries are owned. */
|
||||
MemoryResource **memory_resources;
|
||||
|
||||
/// Relative computational speed of the workers. Owned.
|
||||
double *workers_weights;
|
||||
|
||||
/// For each worker, which memory resource it uses.
|
||||
int *workers_mem_res;
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
Engine(Backend *b, int n_mem, int n_workers)
|
||||
: backend(b),
|
||||
#ifdef MFEM_USE_MPI
|
||||
comm(MPI_COMM_NULL),
|
||||
#endif
|
||||
num_mem_res(n_mem),
|
||||
num_workers(n_workers),
|
||||
memory_resources(new MemoryResource*[num_mem_res]()),
|
||||
workers_weights(new double[num_workers]()),
|
||||
workers_mem_res(new int[num_workers]())
|
||||
{ /* Note: all arrays are value-initialized with zeros. */ }
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual ~Engine()
|
||||
{
|
||||
delete [] workers_mem_res;
|
||||
delete [] workers_weights;
|
||||
for (int i = 0; i < num_mem_res; i++)
|
||||
{
|
||||
delete memory_resources[i];
|
||||
}
|
||||
delete [] memory_resources;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
@name Machine resources interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// Get the associated MPI_Comm
|
||||
MPI_Comm GetComm() const { return comm; }
|
||||
#endif
|
||||
|
||||
/// TODO
|
||||
int GetNumMemRes() const { return num_mem_res; }
|
||||
|
||||
/// TODO
|
||||
MemoryResource &GetMemRes(int idx) const { return *memory_resources[idx]; }
|
||||
|
||||
/// TODO
|
||||
int GetNumWorkers() const { return num_workers; }
|
||||
|
||||
/// TODO
|
||||
const double *GetWorkersWeights() const { return workers_weights; }
|
||||
|
||||
/// TODO
|
||||
const int *GetWorkersMemRes() const { return workers_mem_res; }
|
||||
|
||||
///@}
|
||||
// End: Machine resources interface
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t &As() { *util::As<derived_t>(this); }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { *util::As<const derived_t>(this); }
|
||||
|
||||
|
||||
// TODO: Error handling ... handle errors at the Engine level, at the class
|
||||
// level, or at the method level?
|
||||
|
||||
|
||||
/**
|
||||
@name Virtual interface: finite element data structures and algorithms
|
||||
*/
|
||||
///@{
|
||||
|
||||
// TODO: Asynchronous execution in this class ...
|
||||
|
||||
/// Allocate and return a new layout for the given @a size.
|
||||
/** The layout decomposition is determined automatically by the Engine using
|
||||
a deterministic algorithm: calls to this method with the same @a size
|
||||
will produce the same result, as long as the Engine remains unmodified
|
||||
between the calls.
|
||||
|
||||
The returned object is allocated with operator new and must be
|
||||
deallocated by the caller.
|
||||
|
||||
TODO: Returns NULL if memory allocation fails?
|
||||
*/
|
||||
virtual DLayout MakeLayout(std::size_t size) const = 0;
|
||||
|
||||
/// Allocate and return a new layout for the given worker decomposition.
|
||||
/** The returned object is allocated with operator new and must be
|
||||
deallocated by the caller.
|
||||
|
||||
TODO: Returns NULL if memory allocation fails?
|
||||
|
||||
The @a offsets should satisfy: offsets.Size() == number of workers + 1,
|
||||
offsets[0] == 0, and offsets[i] <= offsets[i+1], for i: 0 <= i < number
|
||||
of workers. */
|
||||
virtual DLayout MakeLayout(const Array<std::size_t> &offsets) const = 0;
|
||||
|
||||
// Note: There may be other ways to construct layouts in the future, e.g.
|
||||
// block-vector layouts, or multi-vector layouts.
|
||||
|
||||
/// TODO
|
||||
virtual DArray MakeArray(PLayout &layout, std::size_t item_size) const = 0;
|
||||
|
||||
/// Allocate and return a new vector using the given @a layout.
|
||||
/** The returned object is a smart pointer that will automatically deallocate
|
||||
the vector.
|
||||
|
||||
TODO: Produce an error if memory allocation fails?
|
||||
|
||||
Only layouts returned by this Engine are guaranteed to be supported.
|
||||
Using a type that is not supported will produce an error. */
|
||||
virtual DVector MakeVector(PLayout &layout,
|
||||
int type_id = ScalarId<double>::value) const = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual DFiniteElementSpace MakeFESpace(FiniteElementSpace &fes) const;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual DBilinearForm MakeBilinearForm(BilinearForm &bf) const = 0;
|
||||
|
||||
|
||||
// Question: How do we construct coefficients?
|
||||
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual void AssembleLinearForm(LinearForm &l_form) const = 0;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual Operator *MakeOperator(const MixedBilinearForm &mbl_form) const = 0;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual Operator *MakeOperator(const NonlinearForm &nl_form) const = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_ENGINE_HPP
|
||||
@@ -1,61 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_FE_SPACE_HPP
|
||||
#define MFEM_BACKENDS_BASE_FE_SPACE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "utils.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class FiniteElementSpace;
|
||||
|
||||
/// TODO: doxygen
|
||||
class PFiniteElementSpace : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Engine with shared ownership
|
||||
SharedPtr<const Engine> engine;
|
||||
/// Not owned.
|
||||
FiniteElementSpace *fes;
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
PFiniteElementSpace(const Engine &e, FiniteElementSpace &fespace)
|
||||
: engine(&e), fes(&fespace) { }
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~PFiniteElementSpace() { }
|
||||
|
||||
/// Get the associated engine
|
||||
const Engine &GetEngine() const { return *engine; }
|
||||
|
||||
mfem::FiniteElementSpace* GetFESpace() const { return fes; }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_FE_SPACE_HPP
|
||||
@@ -1,104 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_LAYOUT_HPP
|
||||
#define MFEM_BACKENDS_BASE_LAYOUT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "smart_pointers.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Polymorphic layout (array/vector layout descriptor)
|
||||
class PLayout : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Engine with shared ownership
|
||||
SharedPtr<const Engine> engine;
|
||||
std::size_t size;
|
||||
|
||||
template <typename DObject>
|
||||
struct Maker
|
||||
{
|
||||
template <typename entry_t>
|
||||
static DObject MakeNew(PLayout &layout);
|
||||
};
|
||||
|
||||
public:
|
||||
explicit PLayout(std::size_t s = 0) : engine(NULL), size(s) { }
|
||||
|
||||
explicit PLayout(const Engine &e, std::size_t s = 0)
|
||||
: engine(&e), size(s) { }
|
||||
|
||||
virtual ~PLayout() { }
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// Resize the layout
|
||||
virtual void Resize(std::size_t new_size) { size = new_size; }
|
||||
|
||||
/// Resize the layout based on the given worker offsets
|
||||
virtual void Resize(const Array<std::size_t> &offsets)
|
||||
{ MFEM_ABORT("method not supported"); }
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
/// Layouts without engine cannot create DArray, DVector, etc.
|
||||
bool HasEngine() const { return engine != NULL; }
|
||||
|
||||
/// TODO: doxygen
|
||||
const Engine &GetEngine() const { return *engine; }
|
||||
|
||||
/// TODO: doxygen
|
||||
std::size_t Size() const { return size; }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
|
||||
/// TODO: doxygen
|
||||
template <typename DObject, typename entry_t>
|
||||
DObject Make()
|
||||
{
|
||||
MFEM_ASSERT(HasEngine(), "this method requires an Engine");
|
||||
return Maker<DObject>::template MakeNew<entry_t>(*this);
|
||||
}
|
||||
};
|
||||
|
||||
template <> struct PLayout::Maker<DArray>
|
||||
{
|
||||
template <typename entry_t> static DArray MakeNew(PLayout &layout)
|
||||
{ return layout.GetEngine().MakeArray(layout, sizeof(entry_t)); }
|
||||
};
|
||||
|
||||
template <> struct PLayout::Maker<DVector>
|
||||
{
|
||||
template <typename entry_t> static DVector MakeNew(PLayout &layout)
|
||||
{ return layout.GetEngine().MakeVector(layout, ScalarId<entry_t>::value); }
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_LAYOUT_HPP
|
||||
@@ -1,59 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "memory_resource.hpp"
|
||||
#include "../../general/error.hpp"
|
||||
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <cerrno>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
void *NewDeleteMemoryResource::DoAllocate(std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
void *p = ::operator new[](bytes);
|
||||
MFEM_VERIFY(!alignment || (std::size_t)(p) % alignment == 0,
|
||||
"invalid alignment");
|
||||
return p;
|
||||
}
|
||||
|
||||
void NewDeleteMemoryResource::DoDeallocate(void *p, std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
::operator delete[](p);
|
||||
}
|
||||
|
||||
|
||||
void *AlignedMemoryResource::DoAllocate(std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
void *p;
|
||||
if (!alignment) { alignment = sizeof(long double); }
|
||||
MFEM_VERIFY(posix_memalign(&p, alignment, bytes) == 0,
|
||||
"error in posix_memalign(): " << strerror(errno));
|
||||
return p;
|
||||
}
|
||||
|
||||
void AlignedMemoryResource::DoDeallocate(void *p, std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
free(p);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
@@ -1,70 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
|
||||
#define MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include <cstddef>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Polymorphic memory resource. Similar to C++17's std::pmr::memory_resource.
|
||||
class MemoryResource
|
||||
{
|
||||
protected:
|
||||
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment) = 0;
|
||||
virtual void DoDeallocate(void* p, std::size_t bytes,
|
||||
std::size_t alignment) = 0;
|
||||
|
||||
public:
|
||||
// Implicitly defined default & copy constructors
|
||||
|
||||
/// Virtual destructor.
|
||||
virtual ~MemoryResource() { }
|
||||
|
||||
/// If alignment == 0, use default alignment.
|
||||
void *Allocate(std::size_t bytes, std::size_t alignment = 0)
|
||||
{ return DoAllocate(bytes, alignment); }
|
||||
|
||||
/// If alignment == 0, use default alignment.
|
||||
void Deallocate(void *p, std::size_t bytes, std::size_t alignment = 0)
|
||||
{ DoDeallocate(p, bytes, alignment); }
|
||||
};
|
||||
|
||||
|
||||
/** @brief Dynamic host memory resource using operator new[](std::size_t) for
|
||||
allocation and operator delete[](void*) for deallocation. */
|
||||
class NewDeleteMemoryResource : public MemoryResource
|
||||
{
|
||||
protected:
|
||||
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment);
|
||||
virtual void DoDeallocate(void *p, std::size_t bytes, std::size_t alignment);
|
||||
};
|
||||
|
||||
|
||||
/** @brief Dynamic host memory resource using posix_memalign() for aligned
|
||||
allocation and free() for deallocation. */
|
||||
class AlignedMemoryResource : public MemoryResource
|
||||
{
|
||||
protected:
|
||||
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment);
|
||||
virtual void DoDeallocate(void *p, std::size_t bytes, std::size_t alignment);
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
|
||||
@@ -1,233 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
|
||||
#define MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "utils.hpp"
|
||||
#include "../../general/error.hpp"
|
||||
#include <cstddef>
|
||||
|
||||
// #define MFEM_TRACE_SHARED_PTR
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
#include "../../general/globals.hpp"
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Base class for classes with simple reference counting.
|
||||
/** Reference counting is performed by the class SharedPtr. */
|
||||
class RefCounted
|
||||
{
|
||||
private:
|
||||
mutable unsigned ref_count;
|
||||
|
||||
/// Only class SharedPtr can access ref_count.
|
||||
template <typename T> friend class SharedPtr;
|
||||
|
||||
public:
|
||||
RefCounted() : ref_count(0) { }
|
||||
|
||||
/** @brief Prevent SharedPtr objects from deleting this object by
|
||||
incrementing the reference counter by one. */
|
||||
void DontDelete() const { ++ref_count; }
|
||||
};
|
||||
|
||||
|
||||
/** @brief Smart pointer class that manages objects of type T derived from class
|
||||
RefCounted. */
|
||||
/** This class is generally meant to work with dynamically allocated object,
|
||||
specifically objects allocated with operator new(). It will invoke operator
|
||||
delete() to destroy the managed object when its reference counter reaches
|
||||
zero. This behavior can be overriden by calling RefCounted::DontDelete() to
|
||||
ensure that an object will not be deleted by a SharedPtr that holds a
|
||||
pointer to it.
|
||||
@note This class is NOT thread-safe and does not support circular ownership.
|
||||
*/
|
||||
template <typename T>
|
||||
class SharedPtr
|
||||
{
|
||||
public:
|
||||
typedef T stored_type;
|
||||
|
||||
private:
|
||||
T *ptr;
|
||||
|
||||
void Init(T *new_ptr)
|
||||
{
|
||||
ptr = new_ptr;
|
||||
if (ptr) { ++ptr->RefCounted::ref_count; }
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
#elif 0
|
||||
mfem::out << " [" << _MFEM_FUNC_NAME << "]: ptr = " << ptr;
|
||||
if (ptr)
|
||||
{
|
||||
mfem::out << ", new ref_count = " << ptr->RefCounted::ref_count;
|
||||
}
|
||||
mfem::out << '\n';
|
||||
#endif
|
||||
}
|
||||
void Destroy()
|
||||
{
|
||||
MFEM_ASSERT(!ptr || ptr->RefCounted::ref_count >= 1, "invalid use");
|
||||
if (ptr && --ptr->RefCounted::ref_count == 0) { delete ptr; }
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
#elif 0
|
||||
mfem::out << " [" << _MFEM_FUNC_NAME << "]: ptr = " << ptr;
|
||||
if (ptr)
|
||||
{
|
||||
mfem::out << ", new ref_count = " << ptr->RefCounted::ref_count;
|
||||
}
|
||||
mfem::out << '\n';
|
||||
#endif
|
||||
}
|
||||
|
||||
public:
|
||||
SharedPtr() : ptr(NULL)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]: ptr = " << ptr << '\n';
|
||||
#endif
|
||||
}
|
||||
|
||||
SharedPtr(const SharedPtr &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Init(other.ptr);
|
||||
}
|
||||
|
||||
template <typename U>
|
||||
SharedPtr(const SharedPtr<U> &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Init(other.Get());
|
||||
}
|
||||
|
||||
explicit SharedPtr(T *p)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Init(p);
|
||||
}
|
||||
|
||||
~SharedPtr()
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Destroy();
|
||||
}
|
||||
|
||||
SharedPtr &operator=(const SharedPtr &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Reset(other.ptr); return *this;
|
||||
}
|
||||
|
||||
template <typename U>
|
||||
SharedPtr &operator=(const SharedPtr<U> &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Reset(other.Get()); return *this;
|
||||
}
|
||||
|
||||
T &operator*() const { return *ptr; }
|
||||
T *operator->() const { return ptr; }
|
||||
|
||||
operator bool() const { return ptr; }
|
||||
bool operator!() const { return !ptr; }
|
||||
|
||||
template <typename U>
|
||||
bool operator==(const SharedPtr<U> &other) const
|
||||
{ return ptr == other.Ptr(); }
|
||||
template <typename U>
|
||||
bool operator!=(const SharedPtr<U> &other) const
|
||||
{ return ptr != other.Ptr(); }
|
||||
|
||||
template <typename U>
|
||||
bool operator==(const U &p) const { return ptr == (void*) p; }
|
||||
template <typename U>
|
||||
bool operator!=(const U &p) const { return ptr != (void*) p; }
|
||||
|
||||
T *Get() const { return ptr; }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t *As() const { return util::As<derived_t>(ptr); }
|
||||
|
||||
unsigned UseCount() const { return ptr ? ptr->RefCounted::ref_count : 0; }
|
||||
|
||||
void Reset()
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Destroy();
|
||||
ptr = NULL;
|
||||
}
|
||||
|
||||
/// The type U* needs to be implicitly convertible to T*
|
||||
template <typename U>
|
||||
void Reset(U *new_ptr)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
if (ptr != new_ptr) { Destroy(); Init(new_ptr); }
|
||||
}
|
||||
|
||||
void Swap(SharedPtr &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
std::swap(ptr, other.ptr);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template <class T>
|
||||
inline void Swap(SharedPtr<T> &a, SharedPtr<T> &b) { a.Swap(b); }
|
||||
|
||||
|
||||
class PLayout;
|
||||
typedef SharedPtr<PLayout> DLayout;
|
||||
|
||||
class PArray;
|
||||
typedef SharedPtr<PArray> DArray;
|
||||
|
||||
class PVector;
|
||||
typedef SharedPtr<PVector> DVector;
|
||||
|
||||
class PFiniteElementSpace;
|
||||
typedef SharedPtr<PFiniteElementSpace> DFiniteElementSpace;
|
||||
|
||||
class PBilinearForm;
|
||||
typedef SharedPtr<PBilinearForm> DBilinearForm;
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
|
||||
@@ -1,52 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_UTILS_HPP
|
||||
#define MFEM_BACKENDS_BASE_UTILS_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "../../general/error.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace util
|
||||
{
|
||||
|
||||
//
|
||||
// Inline methods
|
||||
//
|
||||
|
||||
/// TODO: doxygen
|
||||
template <typename derived_t, typename base_t>
|
||||
inline derived_t *As(base_t *base_obj)
|
||||
{
|
||||
MFEM_ASSERT(dynamic_cast<derived_t*>(base_obj) != NULL,
|
||||
"invalid object type");
|
||||
return static_cast<derived_t*>(base_obj);
|
||||
}
|
||||
|
||||
/// TODO: doxygen
|
||||
template <typename derived_t, typename base_t>
|
||||
inline derived_t *Is(base_t *base_obj)
|
||||
{
|
||||
return dynamic_cast<derived_t*>(base_obj);
|
||||
}
|
||||
|
||||
} // namespace mfem::util
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_UTILS_HPP
|
||||
@@ -1,153 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_VECTOR_HPP
|
||||
#define MFEM_BACKENDS_BASE_VECTOR_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "../../general/scalars.hpp"
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Polymorphic vector - array of scalars.
|
||||
class PVector : virtual public PArray
|
||||
{
|
||||
protected:
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/** @brief Create and return a new vector of the same dynamic type as this
|
||||
vector using the same layout with entries specified by @a buffer_type_id
|
||||
which should be a constant defined by the `value` field in a
|
||||
specialization of the template class mfem::ScalarId.
|
||||
|
||||
Returns NULL if allocation fails.
|
||||
|
||||
If @a copy_data is true, the contents of this vector is copied to the new
|
||||
vector; otherwise, the new vector remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the vector data of the newly created
|
||||
object (in @a *buffer), if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
virtual PVector *DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const = 0;
|
||||
|
||||
/** @brief Compute and return the dot product of @a *this and @a x. In the
|
||||
case of an MPI-parallel vector, the result must be the MPI-global dot
|
||||
product. */
|
||||
/** Both vectors must have the same dynamic type and layout. */
|
||||
virtual void DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const = 0;
|
||||
|
||||
// TODO: add reduction operations: min, max, sum
|
||||
|
||||
/// Perform the operation @a *this = @a a @a x + @a b @a y.
|
||||
/** Rules:
|
||||
- the dynamic type of both @a x and @a y is the same as that of @a *this
|
||||
- if @a a == 0, neither @a x nor its data are accessed
|
||||
- if @a b == 0, neither @a y nor its data are accessed
|
||||
- @a x's data is never the same as @a y's data, unless @a a == 0, or
|
||||
@a b == 0
|
||||
- @a x's data or @a y's data may be the same as the data of @a *this
|
||||
- all accessed vectors, @a x, @a y, and @a *this have the same layout. */
|
||||
virtual void DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id) = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
public:
|
||||
/** @brief Create a PVector. */
|
||||
/** The @a layout must be valid in the sense that layout != NULL and
|
||||
layout->HasEngine() == true. */
|
||||
PVector(PLayout &p_layout)
|
||||
: PArray(p_layout) { }
|
||||
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
|
||||
|
||||
// TODO: Error handling ... handle errors at the Engine level, at the class
|
||||
// level, or at the method level?
|
||||
|
||||
// TODO: Asynchronous execution interface ...
|
||||
|
||||
// TODO: Multi-vector interface ...
|
||||
|
||||
|
||||
/**
|
||||
@name Public virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/** @brief Create and return a new vector of the same dynamic type as this
|
||||
vector using the same layout with entries of type @a scalar_t.
|
||||
|
||||
If @a copy_data is true, the contents of this vector is copied to the new
|
||||
vector; otherwise, the new vector remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the vector data of the newly created
|
||||
object (in @a *buffer) , if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
template <typename scalar_t>
|
||||
DVector Clone(bool copy_data, scalar_t **buffer) const
|
||||
{
|
||||
return DVector(DoVectorClone(copy_data, (void**)buffer,
|
||||
ScalarId<scalar_t>::value));
|
||||
}
|
||||
|
||||
/** @brief Compute and return the dot product of @a *this and @a x. In the
|
||||
case of an MPI-parallel vector, the result must be the MPI-global dot
|
||||
product. */
|
||||
/** Both vectors must have the same dynamic type and layout. */
|
||||
template <typename scalar_t>
|
||||
scalar_t DotProduct(const PVector &x) const
|
||||
{
|
||||
scalar_t result;
|
||||
DoDotProduct(x, &result, ScalarId<scalar_t>::value);
|
||||
return result;
|
||||
}
|
||||
|
||||
// TODO: add reduction operations: min, max, sum
|
||||
|
||||
/// Perform the operation @a *this = @a a @a x + @a b @a y.
|
||||
/** Rules:
|
||||
- the dynamic type of both @a x and @a y is the same as that of @a *this
|
||||
- if @a a == 0, neither @a x nor its data are accessed
|
||||
- if @a b == 0, neither @a y nor its data are accessed
|
||||
- @a x's data is never the same as @a y's data, unless @a a == 0, or
|
||||
@a b == 0
|
||||
- @a x's data or @a y's data may be the same as the data of @a *this
|
||||
- all accessed vectors, @a x, @a y, and @a *this have the same layout. */
|
||||
template <typename scalar_t>
|
||||
void Axpby(const scalar_t &a, const PVector &x,
|
||||
const scalar_t &b, const PVector &y)
|
||||
{ if (Size()) { DoAxpby(&a, x, &b, y, ScalarId<scalar_t>::value); } }
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_VECTOR_HPP
|
||||
@@ -1,66 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
ELEMENT_BATCH : How many elements are in each
|
||||
. computation batch
|
||||
NUM_DOFS_1D : Dofs in the 1D segments
|
||||
NUM_DOFS_2D : Dofs in the 2D faces
|
||||
NUM_DOFS_3D : Dofs in the 3D domain
|
||||
NUM_QUAD_1D : Dofs in the 1D segments
|
||||
NUM_QUAD_2D : Dofs in the 2D faces
|
||||
NUM_QUAD_3D : Dofs in the 3D domain
|
||||
NUM_MAX_1D : max(NUM_QUAD_1D, NUM_DOFS_1D)
|
||||
NUM_QUAD_DOFS_1D: NUM_QUAD_1D * NUM_DOFS_1D
|
||||
COEFF_ARGS : Code that passes required arguments to the kernel
|
||||
COEFF : Code that computes the coefficient
|
||||
================================================
|
||||
|
||||
[MISSING]
|
||||
- Add support to auto-pick @dim and use @idxOrder on stack arrays
|
||||
| double a[2][2];
|
||||
| a[0][1]; <-- regular index
|
||||
| a(0,1); <-- uses @idxOrder a[0][1] or a[1][0]
|
||||
- Add support for @idxOrder to change indexing order after allocation
|
||||
| double a[2][2] @idxOrder(0,1);
|
||||
| a(0,1) -> a[1][0]
|
||||
| @set(a, idxOrder(1,0));
|
||||
| a(0,1) -> a[0][1]
|
||||
- Add support to iterate over loop depending on mode
|
||||
| for(i; @inner) {
|
||||
| for(0 < j < N) {} <-- ++j or j += block?
|
||||
| }
|
||||
*/
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# ifdef OCCA_USING_GPU
|
||||
# if USING_LOW_ORDER
|
||||
# include "mfem-occa://diffusion/tensor/gpuHighOrder.okl"
|
||||
# else
|
||||
# include "mfem-occa://diffusion/tensor/gpuHighOrder.okl"
|
||||
# endif
|
||||
# else
|
||||
# include "mfem-occa://diffusion/tensor/cpu.okl"
|
||||
# endif
|
||||
#else
|
||||
# ifdef OCCA_USING_GPU
|
||||
# if USING_LOW_ORDER
|
||||
# include "mfem-occa://diffusion/simplex/gpuHighOrder.okl"
|
||||
# else
|
||||
# include "mfem-occa://diffusion/simplex/gpuHighOrder.okl"
|
||||
# endif
|
||||
# else
|
||||
# include "mfem-occa://diffusion/simplex/cpu.okl"
|
||||
# endif
|
||||
#endif
|
||||
@@ -1,123 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
PArray *Array::DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
Array *new_array = new Array(OccaLayout(), item_size);
|
||||
if (copy_data)
|
||||
{
|
||||
new_array->slice.copyFrom(slice);
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
*buffer = new_array->GetBuffer();
|
||||
}
|
||||
return new_array;
|
||||
}
|
||||
|
||||
int Array::DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size)
|
||||
{
|
||||
MFEM_ASSERT(dynamic_cast<Layout *>(&new_layout) != NULL,
|
||||
"new_layout is not an OCCA Layout");
|
||||
Layout *lt = static_cast<Layout *>(&new_layout);
|
||||
layout.Reset(lt); // Reset() checks if the pointer is the same
|
||||
int err = ResizeData(lt, item_size);
|
||||
if (!err && buffer)
|
||||
{
|
||||
*buffer = GetBuffer();
|
||||
}
|
||||
return err;
|
||||
}
|
||||
|
||||
void *Array::DoPullData(void *buffer, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
if (!slice.getDevice().hasSeparateMemorySpace())
|
||||
{
|
||||
return slice.ptr();
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
slice.copyTo(buffer);
|
||||
}
|
||||
return buffer;
|
||||
}
|
||||
|
||||
void Array::DoFill(const void *value_ptr, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
switch (item_size)
|
||||
{
|
||||
case sizeof(int8_t):
|
||||
OccaFill((const int8_t *)value_ptr);
|
||||
break;
|
||||
case sizeof(int16_t):
|
||||
OccaFill((const int16_t *)value_ptr);
|
||||
break;
|
||||
case sizeof(int32_t):
|
||||
OccaFill((const int32_t *)value_ptr);
|
||||
break;
|
||||
// case sizeof(int64_t):
|
||||
// OccaFill((const int64_t *)value_ptr);
|
||||
// break;
|
||||
case sizeof(double):
|
||||
OccaFill((const double *)value_ptr);
|
||||
break;
|
||||
// case sizeof(::occa::double2):
|
||||
// OccaFill((const ::occa::double2 *)value_ptr);
|
||||
// break;
|
||||
default:
|
||||
MFEM_ABORT("item_size = " << item_size << " is not supported");
|
||||
}
|
||||
}
|
||||
|
||||
void Array::DoPushData(const void *src_buffer, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
if (slice.getDevice().hasSeparateMemorySpace() || slice.ptr() != src_buffer)
|
||||
{
|
||||
slice.copyFrom(src_buffer);
|
||||
}
|
||||
}
|
||||
|
||||
void Array::DoAssign(const PArray &src, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
// Note: static_cast can not be used here since PArray is a virtual base
|
||||
// class.
|
||||
const Array *source = dynamic_cast<const Array *>(&src);
|
||||
MFEM_ASSERT(source != NULL, "invalid source Array type");
|
||||
MFEM_ASSERT(Size() == source->Size(), "");
|
||||
slice.copyFrom(source->slice);
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,133 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_ARRAY_HPP
|
||||
#define MFEM_BACKENDS_OCCA_ARRAY_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "layout.hpp"
|
||||
#include "../base/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Array : public virtual PArray
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// DLayout layout;
|
||||
|
||||
// Always true: Size()*item_size == slice.size() <= data.size()
|
||||
mutable ::occa::memory data, slice;
|
||||
|
||||
//
|
||||
// Virtual interface
|
||||
//
|
||||
|
||||
virtual void *DoGetData() const { return GetBuffer(); }
|
||||
|
||||
virtual PArray *DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const;
|
||||
|
||||
virtual int DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size);
|
||||
|
||||
virtual void *DoPullData(void *buffer, std::size_t item_size);
|
||||
|
||||
virtual void DoFill(const void *value_ptr, std::size_t item_size);
|
||||
|
||||
virtual void DoPushData(const void *src_buffer, std::size_t item_size);
|
||||
|
||||
virtual void DoAssign(const PArray &src, std::size_t item_size);
|
||||
|
||||
//
|
||||
// Auxiliary methods
|
||||
//
|
||||
|
||||
inline void *GetBuffer() const;
|
||||
|
||||
inline int ResizeData(const Layout *lt, std::size_t item_size);
|
||||
|
||||
template <typename T>
|
||||
inline void OccaFill(const T *val_ptr)
|
||||
{ ::occa::linalg::operator_eq<T>(slice, *val_ptr); }
|
||||
|
||||
public:
|
||||
Array(Layout <, std::size_t item_size)
|
||||
: PArray(lt),
|
||||
data(lt.Alloc(lt.Size()*item_size)),
|
||||
slice(data)
|
||||
{ }
|
||||
|
||||
virtual ~Array() { }
|
||||
|
||||
inline void MakeRef(Array &master);
|
||||
|
||||
Layout &OccaLayout() const
|
||||
{ return *static_cast<Layout *>(layout.Get()); }
|
||||
|
||||
::occa::memory &OccaMem() { return slice; }
|
||||
const ::occa::memory &OccaMem() const { return slice; }
|
||||
};
|
||||
|
||||
|
||||
//
|
||||
// Inline methods
|
||||
//
|
||||
|
||||
inline void *Array::GetBuffer() const
|
||||
{
|
||||
if (!slice.getDevice().hasSeparateMemorySpace())
|
||||
{
|
||||
return slice.ptr();
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
inline int Array::ResizeData(const Layout *lt, std::size_t item_size)
|
||||
{
|
||||
const std::size_t new_bytes = lt->Size()*item_size;
|
||||
if (data.size() < new_bytes ||
|
||||
data.getDHandle() != lt->OccaEngine().GetDevice().getDHandle())
|
||||
{
|
||||
data = lt->Alloc(new_bytes);
|
||||
slice = data;
|
||||
// If memory allocation fails - an exception is thrown.
|
||||
}
|
||||
else if (slice.size() != new_bytes)
|
||||
{
|
||||
slice = data.slice(0, new_bytes);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
inline void Array::MakeRef(Array &master)
|
||||
{
|
||||
layout = master.layout;
|
||||
data = master.data;
|
||||
slice = master.slice;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_ARRAY_HPP
|
||||
@@ -1,47 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
bool Backend::Supports(const std::string &engine_spec) const
|
||||
{
|
||||
// TODO: check if 'engine_spec' is valid OCCA string.
|
||||
return true;
|
||||
}
|
||||
|
||||
mfem::Engine *Create(const std::string &engine_spec)
|
||||
{
|
||||
return new Engine(engine_spec);
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec)
|
||||
{
|
||||
return new Engine(comm, engine_spec);
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,49 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_BACKEND_HPP
|
||||
#define MFEM_BACKENDS_OCCA_BACKEND_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
// Only the Backend and Engine classes should be exposed through "backend.hpp"
|
||||
#include "../base/backend.hpp"
|
||||
#include "engine.hpp"
|
||||
#include <occa.hpp>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Backend : public mfem::Backend
|
||||
{
|
||||
public:
|
||||
virtual ~Backend();
|
||||
|
||||
virtual bool Supports(const std::string &engine_spec) const;
|
||||
|
||||
virtual mfem::Engine *Create(const std::string &engine_spec);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
virtual mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec);
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_BACKEND_HPP
|
||||
@@ -1,514 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "../../fem/bilinearform.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
OccaBilinearForm::OccaBilinearForm(FiniteElementSpace *ofespace_) :
|
||||
Operator(ofespace_->OccaVLayout()),
|
||||
localX((ofespace_->OccaEVLayout().DontDelete(), ofespace_->OccaEVLayout())),
|
||||
localY((ofespace_->OccaEVLayout().DontDelete(), ofespace_->OccaEVLayout()))
|
||||
{
|
||||
Init(ofespace_->OccaEngine(), ofespace_, ofespace_);
|
||||
}
|
||||
|
||||
OccaBilinearForm::OccaBilinearForm(FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_) :
|
||||
Operator(otrialFESpace_->OccaVLayout(),
|
||||
otestFESpace_->OccaVLayout()),
|
||||
localX((otrialFESpace_->OccaEVLayout().DontDelete(), otrialFESpace_->OccaEVLayout())),
|
||||
localY((otestFESpace_->OccaEVLayout().DontDelete(), otestFESpace_->OccaEVLayout()))
|
||||
{
|
||||
Init(otrialFESpace_->OccaEngine(), otrialFESpace_, otestFESpace_);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::Init(const Engine &e,
|
||||
FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_)
|
||||
{
|
||||
engine.Reset(&e);
|
||||
|
||||
otrialFESpace = otrialFESpace_;
|
||||
trialFESpace = otrialFESpace_->GetFESpace();
|
||||
|
||||
otestFESpace = otestFESpace_;
|
||||
testFESpace = otestFESpace_->GetFESpace();
|
||||
|
||||
mesh = trialFESpace->GetMesh();
|
||||
|
||||
const int elements = GetNE();
|
||||
|
||||
const int trialVDim = trialFESpace->GetVDim();
|
||||
|
||||
const int trialLocalDofs = otrialFESpace->GetLocalDofs();
|
||||
const int testLocalDofs = otestFESpace->GetLocalDofs();
|
||||
|
||||
// First-touch policy when running with OpenMP
|
||||
if (GetDevice().mode() == "OpenMP")
|
||||
{
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
const std::string &okl_defines = OccaEngine().GetOklDefines();
|
||||
::occa::kernel initLocalKernel =
|
||||
GetDevice().buildKernel(okl_path + "utils.okl",
|
||||
"InitLocalVector",
|
||||
okl_defines);
|
||||
|
||||
const std::size_t sd = sizeof(double);
|
||||
const uint64_t trialEntries = sd * (elements * trialLocalDofs);
|
||||
const uint64_t testEntries = sd * (elements * testLocalDofs);
|
||||
for (int v = 0; v < trialVDim; ++v)
|
||||
{
|
||||
const uint64_t trialOffset = v * trialEntries;
|
||||
const uint64_t testOffset = v * testEntries;
|
||||
|
||||
initLocalKernel(elements, trialLocalDofs,
|
||||
localX.OccaMem().slice(trialOffset, trialEntries));
|
||||
initLocalKernel(elements, testLocalDofs,
|
||||
localY.OccaMem().slice(testOffset, testEntries));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int OccaBilinearForm::BaseGeom() const
|
||||
{
|
||||
return mesh->GetElementBaseGeometry();
|
||||
}
|
||||
|
||||
int OccaBilinearForm::GetDim() const
|
||||
{
|
||||
return mesh->Dimension();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetNE() const
|
||||
{
|
||||
return mesh->GetNE();
|
||||
}
|
||||
|
||||
Mesh& OccaBilinearForm::GetMesh() const
|
||||
{
|
||||
return *mesh;
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaBilinearForm::GetTrialOccaFESpace() const
|
||||
{
|
||||
return *otrialFESpace;
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaBilinearForm::GetTestOccaFESpace() const
|
||||
{
|
||||
return *otestFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaBilinearForm::GetTrialFESpace() const
|
||||
{
|
||||
return *trialFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaBilinearForm::GetTestFESpace() const
|
||||
{
|
||||
return *testFESpace;
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTrialNDofs() const
|
||||
{
|
||||
return trialFESpace->GetNDofs();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTestNDofs() const
|
||||
{
|
||||
return testFESpace->GetNDofs();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTrialVDim() const
|
||||
{
|
||||
return trialFESpace->GetVDim();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTestVDim() const
|
||||
{
|
||||
return testFESpace->GetVDim();
|
||||
}
|
||||
|
||||
const FiniteElement& OccaBilinearForm::GetTrialFE(const int i) const
|
||||
{
|
||||
return *(trialFESpace->GetFE(i));
|
||||
}
|
||||
|
||||
const FiniteElement& OccaBilinearForm::GetTestFE(const int i) const
|
||||
{
|
||||
return *(testFESpace->GetFE(i));
|
||||
}
|
||||
|
||||
// Adds new Domain Integrator.
|
||||
void OccaBilinearForm::AddDomainIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, DomainIntegrator);
|
||||
}
|
||||
|
||||
// Adds new Boundary Integrator.
|
||||
void OccaBilinearForm::AddBoundaryIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, BoundaryIntegrator);
|
||||
}
|
||||
|
||||
// Adds new interior Face Integrator.
|
||||
void OccaBilinearForm::AddInteriorFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, InteriorFaceIntegrator);
|
||||
}
|
||||
|
||||
// Adds new boundary Face Integrator.
|
||||
void OccaBilinearForm::AddBoundaryFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, BoundaryFaceIntegrator);
|
||||
}
|
||||
|
||||
// Adds Integrator based on OccaIntegratorType
|
||||
void OccaBilinearForm::AddIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props,
|
||||
const OccaIntegratorType itype)
|
||||
{
|
||||
if (integrator == NULL)
|
||||
{
|
||||
std::stringstream error_ss;
|
||||
error_ss << "OccaBilinearForm::";
|
||||
switch (itype)
|
||||
{
|
||||
case DomainIntegrator : error_ss << "AddDomainIntegrator"; break;
|
||||
case BoundaryIntegrator : error_ss << "AddBoundaryIntegrator"; break;
|
||||
case InteriorFaceIntegrator: error_ss << "AddInteriorFaceIntegrator"; break;
|
||||
case BoundaryFaceIntegrator: error_ss << "AddBoundaryFaceIntegrator"; break;
|
||||
}
|
||||
error_ss << " (...):\n"
|
||||
<< " Integrator is NULL";
|
||||
const std::string error = error_ss.str();
|
||||
mfem_error(error.c_str());
|
||||
}
|
||||
integrator->SetupIntegrator(*this, baseKernelProps + props, itype);
|
||||
integrators.push_back(integrator);
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTrialProlongation() const
|
||||
{
|
||||
return otrialFESpace->GetProlongationOperator();
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTestProlongation() const
|
||||
{
|
||||
return otestFESpace->GetProlongationOperator();
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTrialRestriction() const
|
||||
{
|
||||
return otrialFESpace->GetRestrictionOperator();
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTestRestriction() const
|
||||
{
|
||||
return otestFESpace->GetRestrictionOperator();
|
||||
}
|
||||
|
||||
void OccaBilinearForm::Assemble()
|
||||
{
|
||||
// [MISSING] Find geometric information that is needed by intergrators
|
||||
// to share between integrators.
|
||||
const int integratorCount = (int) integrators.size();
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
integrators[i]->Assemble();
|
||||
}
|
||||
}
|
||||
|
||||
void OccaBilinearForm::FormLinearSystem(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *&Aout,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior)
|
||||
{
|
||||
FormOperator(constraintList, Aout);
|
||||
InitRHS(constraintList, x, b, Aout, X, B, copy_interior);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::FormOperator(const mfem::Array<int> &constraintList,
|
||||
mfem::Operator *&Aout)
|
||||
{
|
||||
const mfem::Operator *trialP = GetTrialProlongation();
|
||||
const mfem::Operator *testP = GetTestProlongation();
|
||||
mfem::Operator *rap = this;
|
||||
|
||||
if (trialP)
|
||||
{
|
||||
rap = new RAPOperator(*testP, *this, *trialP);
|
||||
}
|
||||
|
||||
Aout = new OccaConstrainedOperator(rap, constraintList,
|
||||
rap != this);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::InitRHS(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *A,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior)
|
||||
{
|
||||
const std::string okl_defines = OccaEngine().GetOklDefines();
|
||||
|
||||
// FIXME: move these kernels to the Backend?
|
||||
static ::occa::kernelBuilder get_subvector_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_get_subvector",
|
||||
|
||||
"const int dof_i = v2[i];"
|
||||
"v0[i] = dof_i >= 0 ? v1[dof_i] : -v1[-dof_i - 1];",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}" + okl_defines);
|
||||
|
||||
static ::occa::kernelBuilder set_subvector_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_set_subvector",
|
||||
"const int dof_i = v2[i];"
|
||||
"if (dof_i >= 0) { v0[dof_i] = v1[i]; }"
|
||||
"else { v0[-dof_i - 1] = -v1[i]; }",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}" + okl_defines);
|
||||
|
||||
const mfem::Operator *P = GetTrialProlongation();
|
||||
const mfem::Operator *R = GetTrialRestriction();
|
||||
|
||||
if (P)
|
||||
{
|
||||
// Variational restriction with P
|
||||
B.Resize(P->InLayout());
|
||||
P->MultTranspose(b, B);
|
||||
X.Resize(R->OutLayout());
|
||||
R->Mult(x, X);
|
||||
}
|
||||
else
|
||||
{
|
||||
// rap, X and B point to the same data as this, x and b
|
||||
X.MakeRef(x);
|
||||
B.MakeRef(b);
|
||||
}
|
||||
|
||||
if (!copy_interior && constraintList.Size() > 0)
|
||||
{
|
||||
::occa::kernel get_subvector_kernel =
|
||||
get_subvector_builder.build(GetDevice());
|
||||
::occa::kernel set_subvector_kernel =
|
||||
set_subvector_builder.build(GetDevice());
|
||||
|
||||
const Array &constrList = constraintList.Get_PArray()->As<Array>();
|
||||
Vector subvec(constrList.OccaLayout());
|
||||
|
||||
get_subvector_kernel(constraintList.Size(),
|
||||
subvec.OccaMem(),
|
||||
X.Get_PVector()->As<Vector>().OccaMem(),
|
||||
constrList.OccaMem());
|
||||
|
||||
X.Fill(0.0);
|
||||
|
||||
set_subvector_kernel(constraintList.Size(),
|
||||
X.Get_PVector()->As<Vector>().OccaMem(),
|
||||
subvec.OccaMem(),
|
||||
constrList.OccaMem());
|
||||
}
|
||||
|
||||
OccaConstrainedOperator *cA = dynamic_cast<OccaConstrainedOperator*>(A);
|
||||
if (cA)
|
||||
{
|
||||
cA->EliminateRHS(X.Get_PVector()->As<Vector>(),
|
||||
B.Get_PVector()->As<Vector>());
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem_error("OccaBilinearForm::InitRHS expects an OccaConstrainedOperator");
|
||||
}
|
||||
}
|
||||
|
||||
// Matrix vector multiplication.
|
||||
void OccaBilinearForm::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
otrialFESpace->GlobalToLocal(x, localX);
|
||||
localY.Fill<double>(0.0);
|
||||
|
||||
const int integratorCount = (int) integrators.size();
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
integrators[i]->MultAdd(localX, localY);
|
||||
}
|
||||
|
||||
otestFESpace->LocalToGlobal(localY, y);
|
||||
}
|
||||
|
||||
// Matrix transpose vector multiplication.
|
||||
void OccaBilinearForm::MultTranspose_(const Vector &x, Vector &y) const
|
||||
{
|
||||
otestFESpace->GlobalToLocal(x, localX);
|
||||
localY.Fill<double>(0.0);
|
||||
|
||||
const int integratorCount = (int) integrators.size();
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
integrators[i]->MultTransposeAdd(localX, localY);
|
||||
}
|
||||
|
||||
otrialFESpace->LocalToGlobal(localY, y);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::OccaRecoverFEMSolution(const mfem::Vector &X,
|
||||
const mfem::Vector &b,
|
||||
mfem::Vector &x)
|
||||
{
|
||||
const mfem::Operator *P = this->GetTrialProlongation();
|
||||
if (P)
|
||||
{
|
||||
// Apply conforming prolongation
|
||||
x.Resize(P->OutLayout());
|
||||
P->Mult(X, x);
|
||||
}
|
||||
// Otherwise X and x point to the same data
|
||||
}
|
||||
|
||||
// Frees memory bilinear form.
|
||||
OccaBilinearForm::~OccaBilinearForm()
|
||||
{
|
||||
// Make sure all integrators free their data
|
||||
IntegratorVector::iterator it = integrators.begin();
|
||||
while (it != integrators.end())
|
||||
{
|
||||
delete *it;
|
||||
++it;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void BilinearForm::InitOccaBilinearForm()
|
||||
{
|
||||
// Init 'obform' using 'bform'
|
||||
MFEM_ASSERT(bform != NULL, "");
|
||||
MFEM_ASSERT(obform == NULL, "");
|
||||
|
||||
FiniteElementSpace &ofes =
|
||||
bform->FESpace()->Get_PFESpace()->As<FiniteElementSpace>();
|
||||
obform = new OccaBilinearForm(&ofes);
|
||||
|
||||
// Transfer domain integrators
|
||||
mfem::Array<mfem::BilinearFormIntegrator*> &dbfi = *bform->GetDBFI();
|
||||
for (int i = 0; i < dbfi.Size(); i++)
|
||||
{
|
||||
std::string integ_name(dbfi[i]->Name());
|
||||
Coefficient *scal_coeff = dbfi[i]->GetScalarCoefficient();
|
||||
ConstantCoefficient *const_coeff =
|
||||
dynamic_cast<ConstantCoefficient*>(scal_coeff);
|
||||
// TODO: other types of coefficients ...
|
||||
double val = const_coeff ? const_coeff->constant : 1.0;
|
||||
OccaCoefficient ocoeff(obform->OccaEngine(), val);
|
||||
|
||||
OccaIntegrator *ointeg = NULL;
|
||||
|
||||
if (integ_name == "(undefined)")
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator does not define Name()");
|
||||
}
|
||||
else if (integ_name == "diffusion")
|
||||
{
|
||||
ointeg = new OccaDiffusionIntegrator(ocoeff);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator [Name() = " << integ_name
|
||||
<< "] is not supported");
|
||||
}
|
||||
|
||||
const mfem::IntegrationRule *ir = dbfi[i]->GetIntRule();
|
||||
if (ir) { ointeg->SetIntegrationRule(*ir); }
|
||||
|
||||
obform->AddDomainIntegrator(ointeg);
|
||||
}
|
||||
|
||||
// TODO: other types of integrators ...
|
||||
}
|
||||
|
||||
bool BilinearForm::Assemble()
|
||||
{
|
||||
if (obform == NULL) { InitOccaBilinearForm(); }
|
||||
|
||||
obform->Assemble();
|
||||
|
||||
return true; // --> host assembly is not needed
|
||||
}
|
||||
|
||||
void BilinearForm::FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::OperatorHandle &A)
|
||||
{
|
||||
if (A.Type() == mfem::Operator::ANY_TYPE)
|
||||
{
|
||||
mfem::Operator *Aout = NULL;
|
||||
obform->FormOperator(ess_tdof_list, Aout);
|
||||
A.Reset(Aout);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Operator::Type is not supported, type = " << A.Type());
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::OperatorHandle &A,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior)
|
||||
{
|
||||
FormSystemMatrix(ess_tdof_list, A);
|
||||
obform->InitRHS(ess_tdof_list, x, b, A.Ptr(), X, B, copy_interior);
|
||||
}
|
||||
|
||||
void BilinearForm::RecoverFEMSolution(const mfem::Vector &X,
|
||||
const mfem::Vector &b,
|
||||
mfem::Vector &x)
|
||||
{
|
||||
obform->OccaRecoverFEMSolution(X, b, x);
|
||||
}
|
||||
|
||||
BilinearForm::~BilinearForm()
|
||||
{
|
||||
delete obform;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,213 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
|
||||
#define MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "fespace.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
enum OccaIntegratorType
|
||||
{
|
||||
DomainIntegrator = 0,
|
||||
BoundaryIntegrator = 1,
|
||||
InteriorFaceIntegrator = 2,
|
||||
BoundaryFaceIntegrator = 3
|
||||
};
|
||||
|
||||
class OccaIntegrator;
|
||||
|
||||
|
||||
/** Class for bilinear form - "Matrix" with associated FE space and
|
||||
BLFIntegrators. */
|
||||
class OccaBilinearForm : public Operator
|
||||
{
|
||||
friend class OccaIntegrator;
|
||||
|
||||
protected:
|
||||
typedef std::vector<OccaIntegrator*> IntegratorVector;
|
||||
|
||||
SharedPtr<const Engine> engine;
|
||||
|
||||
// State information
|
||||
mutable mfem::Mesh *mesh;
|
||||
|
||||
mutable FiniteElementSpace *otrialFESpace;
|
||||
mutable mfem::FiniteElementSpace *trialFESpace;
|
||||
|
||||
mutable FiniteElementSpace *otestFESpace;
|
||||
mutable mfem::FiniteElementSpace *testFESpace;
|
||||
|
||||
IntegratorVector integrators;
|
||||
|
||||
// Device data
|
||||
::occa::properties baseKernelProps;
|
||||
|
||||
// The input and output vectors are mapped to local nodes for efficient
|
||||
// operations. In other words, they are E-vectors.
|
||||
// The size is: (number of elements) * (nodes in element) * (vector dim)
|
||||
mutable Vector localX, localY;
|
||||
|
||||
public:
|
||||
OccaBilinearForm(FiniteElementSpace *ofespace_);
|
||||
|
||||
OccaBilinearForm(FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_);
|
||||
|
||||
void Init(const Engine &e,
|
||||
FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_);
|
||||
|
||||
const Engine &OccaEngine() const { return *engine; }
|
||||
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return engine->GetDevice(idx); }
|
||||
|
||||
// Useful mesh Information
|
||||
int BaseGeom() const;
|
||||
int GetDim() const;
|
||||
int64_t GetNE() const;
|
||||
|
||||
mfem::Mesh& GetMesh() const;
|
||||
|
||||
FiniteElementSpace& GetTrialOccaFESpace() const;
|
||||
FiniteElementSpace& GetTestOccaFESpace() const;
|
||||
|
||||
mfem::FiniteElementSpace& GetTrialFESpace() const;
|
||||
mfem::FiniteElementSpace& GetTestFESpace() const;
|
||||
|
||||
// Useful FE information
|
||||
int64_t GetTrialNDofs() const;
|
||||
int64_t GetTestNDofs() const;
|
||||
|
||||
int64_t GetTrialVDim() const;
|
||||
int64_t GetTestVDim() const;
|
||||
|
||||
const mfem::FiniteElement& GetTrialFE(const int i) const;
|
||||
const mfem::FiniteElement& GetTestFE(const int i) const;
|
||||
|
||||
// Adds new Domain Integrator.
|
||||
void AddDomainIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds new Boundary Integrator.
|
||||
void AddBoundaryIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds new interior Face Integrator.
|
||||
void AddInteriorFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds new boundary Face Integrator.
|
||||
void AddBoundaryFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds Integrator based on OccaIntegratorType
|
||||
void AddIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props,
|
||||
const OccaIntegratorType itype);
|
||||
|
||||
virtual const mfem::Operator *GetTrialProlongation() const;
|
||||
virtual const mfem::Operator *GetTestProlongation() const;
|
||||
|
||||
virtual const mfem::Operator *GetTrialRestriction() const;
|
||||
virtual const mfem::Operator *GetTestRestriction() const;
|
||||
|
||||
// Assembles the form i.e. sums over all domain/bdr integrators.
|
||||
virtual void Assemble();
|
||||
|
||||
void FormLinearSystem(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *&Aout,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior = 0);
|
||||
|
||||
void FormOperator(const mfem::Array<int> &constraintList,
|
||||
mfem::Operator *&Aout);
|
||||
|
||||
void InitRHS(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *Aout,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior = 0);
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
|
||||
void OccaRecoverFEMSolution(const mfem::Vector &X, const mfem::Vector &b,
|
||||
mfem::Vector &x);
|
||||
|
||||
// Destroys bilinear form.
|
||||
~OccaBilinearForm();
|
||||
};
|
||||
|
||||
|
||||
/// TODO: doxygen
|
||||
class BilinearForm : public mfem::PBilinearForm
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// mfem::BilinearForm *bform;
|
||||
OccaBilinearForm *obform;
|
||||
|
||||
// Called from Assemble() if obform is NULL to initialize obform.
|
||||
void InitOccaBilinearForm();
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
BilinearForm(const Engine &e, mfem::BilinearForm &bf)
|
||||
: mfem::PBilinearForm(e, bf), obform(NULL) { }
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~BilinearForm();
|
||||
|
||||
/// Assemble the PBilinearForm.
|
||||
/** This method is called from the method mfem::BilinearForm::Assemble() of
|
||||
the associated mfem::BilinearForm, #bform.
|
||||
@returns True, if the host assembly should NOT be performed. */
|
||||
virtual bool Assemble();
|
||||
|
||||
virtual void FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::OperatorHandle &A);
|
||||
|
||||
virtual void FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::OperatorHandle &A,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior);
|
||||
|
||||
virtual void RecoverFEMSolution(const mfem::Vector &X, const mfem::Vector &b,
|
||||
mfem::Vector &x);
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
|
||||
@@ -1,956 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "bilininteg.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
std::map<std::string, OccaDofQuadMaps> OccaDofQuadMaps::AllDofQuadMaps;
|
||||
|
||||
OccaGeometry OccaGeometry::Get(::occa::device device,
|
||||
FiniteElementSpace &ofespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const int flags)
|
||||
{
|
||||
OccaGeometry geom;
|
||||
|
||||
mfem::Mesh &mesh = *(ofespace.GetMesh());
|
||||
if (!mesh.GetNodes())
|
||||
{
|
||||
mesh.SetCurvature(1, false, -1, mfem::Ordering::byVDIM);
|
||||
}
|
||||
mfem::GridFunction &nodes = *(mesh.GetNodes());
|
||||
const mfem::FiniteElementSpace &fespace = *(nodes.FESpace());
|
||||
const mfem::FiniteElement &fe = *(fespace.GetFE(0));
|
||||
|
||||
const int dims = fe.GetDim();
|
||||
const int elements = fespace.GetNE();
|
||||
const int numDofs = fe.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
MFEM_ASSERT(dims == mesh.SpaceDimension(), "");
|
||||
|
||||
geom.meshNodes.allocate(device,
|
||||
dims, numDofs, elements);
|
||||
|
||||
const mfem::Table &e2dTable = fespace.GetElementToDofTable();
|
||||
const int *elementMap = e2dTable.GetJ();
|
||||
nodes.Pull();
|
||||
for (int e = 0; e < elements; ++e)
|
||||
{
|
||||
for (int dof = 0; dof < numDofs; ++dof)
|
||||
{
|
||||
const int gid = elementMap[dof + numDofs*e];
|
||||
for (int dim = 0; dim < dims; ++dim)
|
||||
{
|
||||
geom.meshNodes(dim, dof, e) = nodes[fespace.DofToVDof(gid,dim)];
|
||||
}
|
||||
}
|
||||
}
|
||||
geom.meshNodes.keepInDevice();
|
||||
|
||||
if (flags & Jacobian)
|
||||
{
|
||||
geom.J.allocate(device,
|
||||
dims, dims, numQuad, elements);
|
||||
}
|
||||
else
|
||||
{
|
||||
geom.J.allocate(device, 1);
|
||||
}
|
||||
if (flags & JacobianInv)
|
||||
{
|
||||
geom.invJ.allocate(device,
|
||||
dims, dims, numQuad, elements);
|
||||
}
|
||||
else
|
||||
{
|
||||
geom.invJ.allocate(device, 1);
|
||||
}
|
||||
if (flags & JacobianDet)
|
||||
{
|
||||
geom.detJ.allocate(device,
|
||||
numQuad, elements);
|
||||
}
|
||||
else
|
||||
{
|
||||
geom.detJ.allocate(device, 1);
|
||||
}
|
||||
|
||||
geom.J.stopManaging();
|
||||
geom.invJ.stopManaging();
|
||||
geom.detJ.stopManaging();
|
||||
|
||||
OccaDofQuadMaps &maps = OccaDofQuadMaps::GetSimplexMaps(device, fe, ir);
|
||||
|
||||
::occa::properties props;
|
||||
props["defines/NUM_DOFS"] = numDofs;
|
||||
props["defines/NUM_QUAD"] = numQuad;
|
||||
props["defines/STORE_JACOBIAN"] = (flags & Jacobian);
|
||||
props["defines/STORE_JACOBIAN_INV"] = (flags & JacobianInv);
|
||||
props["defines/STORE_JACOBIAN_DET"] = (flags & JacobianDet);
|
||||
|
||||
const std::string &okl_path = ofespace.OccaEngine().GetOklPath();
|
||||
const std::string &okl_defines = ofespace.OccaEngine().GetOklDefines();
|
||||
::occa::kernel init = device.buildKernel(okl_path + "geometry.okl",
|
||||
stringWithDim("InitGeometryInfo",
|
||||
fe.GetDim()),
|
||||
props + okl_defines);
|
||||
init(elements,
|
||||
maps.dofToQuadD,
|
||||
geom.meshNodes,
|
||||
geom.J, geom.invJ, geom.detJ);
|
||||
|
||||
return geom;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps::OccaDofQuadMaps() :
|
||||
hash() {}
|
||||
|
||||
OccaDofQuadMaps::OccaDofQuadMaps(const OccaDofQuadMaps &maps)
|
||||
{
|
||||
*this = maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::operator = (const OccaDofQuadMaps &maps)
|
||||
{
|
||||
hash = maps.hash;
|
||||
dofToQuad = maps.dofToQuad;
|
||||
dofToQuadD = maps.dofToQuadD;
|
||||
quadToDof = maps.quadToDof;
|
||||
quadToDofD = maps.quadToDofD;
|
||||
quadWeights = maps.quadWeights;
|
||||
return *this;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return Get(device,
|
||||
*fespace.GetFE(0),
|
||||
*fespace.GetFE(0),
|
||||
ir,
|
||||
transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return Get(device, fe, fe, ir, transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const FiniteElementSpace &trialFESpace,
|
||||
const FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return Get(device,
|
||||
*trialFESpace.GetFE(0),
|
||||
*testFESpace.GetFE(0),
|
||||
ir,
|
||||
transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return (dynamic_cast<const mfem::TensorBasisElement*>(&trialFE)
|
||||
? GetTensorMaps(device, trialFE, testFE, ir, transpose)
|
||||
: GetSimplexMaps(device, trialFE, testFE, ir, transpose));
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return GetTensorMaps(device,
|
||||
fe, fe,
|
||||
ir, transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
const mfem::TensorBasisElement &trialTFE =
|
||||
dynamic_cast<const mfem::TensorBasisElement&>(trialFE);
|
||||
const mfem::TensorBasisElement &testTFE =
|
||||
dynamic_cast<const mfem::TensorBasisElement&>(testFE);
|
||||
|
||||
std::stringstream ss;
|
||||
ss << ::occa::hash(device)
|
||||
<< "Tensor"
|
||||
<< "O1:" << trialFE.GetOrder()
|
||||
<< "O2:" << testFE.GetOrder()
|
||||
<< "BT1:" << trialTFE.GetBasisType()
|
||||
<< "BT2:" << testTFE.GetBasisType()
|
||||
<< "Q:" << ir.GetNPoints();
|
||||
std::string hash = ss.str();
|
||||
|
||||
// If we've already made the dof-quad maps, reuse them
|
||||
OccaDofQuadMaps &maps = AllDofQuadMaps[hash];
|
||||
if (!maps.hash.size())
|
||||
{
|
||||
// Create the dof-quad maps
|
||||
maps.hash = hash;
|
||||
|
||||
OccaDofQuadMaps trialMaps = GetD2QTensorMaps(device, trialFE, ir);
|
||||
OccaDofQuadMaps testMaps = GetD2QTensorMaps(device, testFE , ir, true);
|
||||
|
||||
maps.dofToQuad = trialMaps.dofToQuad;
|
||||
maps.dofToQuadD = trialMaps.dofToQuadD;
|
||||
maps.quadToDof = testMaps.dofToQuad;
|
||||
maps.quadToDofD = testMaps.dofToQuadD;
|
||||
maps.quadWeights = testMaps.quadWeights;
|
||||
}
|
||||
return maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps OccaDofQuadMaps::GetD2QTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
const mfem::TensorBasisElement &tfe =
|
||||
dynamic_cast<const mfem::TensorBasisElement&>(fe);
|
||||
|
||||
const mfem::Poly_1D::Basis &basis = tfe.GetBasis1D();
|
||||
const int order = fe.GetOrder();
|
||||
// [MISSING] Get 1D dofs
|
||||
const int dofs = order + 1;
|
||||
const int dims = fe.GetDim();
|
||||
|
||||
// Create the dof -> quadrature point map
|
||||
const mfem::IntegrationRule &ir1D =
|
||||
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir.GetOrder());
|
||||
const int quadPoints = ir1D.GetNPoints();
|
||||
const int quadPoints2D = quadPoints*quadPoints;
|
||||
const int quadPoints3D = quadPoints2D*quadPoints;
|
||||
const int quadPointsND = ((dims == 1) ? quadPoints :
|
||||
((dims == 2) ? quadPoints2D : quadPoints3D));
|
||||
|
||||
OccaDofQuadMaps maps;
|
||||
// Initialize the dof -> quad mapping
|
||||
maps.dofToQuad.allocate(device,
|
||||
quadPoints, dofs);
|
||||
maps.dofToQuadD.allocate(device,
|
||||
quadPoints, dofs);
|
||||
|
||||
double *quadWeights1DData = NULL;
|
||||
|
||||
if (transpose)
|
||||
{
|
||||
maps.dofToQuad.reindex(1,0);
|
||||
maps.dofToQuadD.reindex(1,0);
|
||||
// Initialize quad weights only for transpose
|
||||
maps.quadWeights.allocate(device,
|
||||
quadPointsND);
|
||||
quadWeights1DData = new double[quadPoints];
|
||||
}
|
||||
|
||||
mfem::Vector d2q(dofs);
|
||||
mfem::Vector d2qD(dofs);
|
||||
for (int q = 0; q < quadPoints; ++q)
|
||||
{
|
||||
const mfem::IntegrationPoint &ip = ir1D.IntPoint(q);
|
||||
basis.Eval(ip.x, d2q, d2qD);
|
||||
if (transpose)
|
||||
{
|
||||
quadWeights1DData[q] = ip.weight;
|
||||
}
|
||||
for (int d = 0; d < dofs; ++d)
|
||||
{
|
||||
maps.dofToQuad(q, d) = d2q[d];
|
||||
maps.dofToQuadD(q, d) = d2qD[d];
|
||||
}
|
||||
}
|
||||
|
||||
maps.dofToQuad.keepInDevice();
|
||||
maps.dofToQuadD.keepInDevice();
|
||||
|
||||
if (transpose)
|
||||
{
|
||||
for (int q = 0; q < quadPointsND; ++q)
|
||||
{
|
||||
const int qx = q % quadPoints;
|
||||
const int qz = q / quadPoints2D;
|
||||
const int qy = (q - qz*quadPoints2D) / quadPoints;
|
||||
double w = quadWeights1DData[qx];
|
||||
if (dims > 1)
|
||||
{
|
||||
w *= quadWeights1DData[qy];
|
||||
}
|
||||
if (dims > 2)
|
||||
{
|
||||
w *= quadWeights1DData[qz];
|
||||
}
|
||||
maps.quadWeights[q] = w;
|
||||
}
|
||||
maps.quadWeights.keepInDevice();
|
||||
delete [] quadWeights1DData;
|
||||
}
|
||||
|
||||
return maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return GetSimplexMaps(device,
|
||||
fe, fe,
|
||||
ir, transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
std::stringstream ss;
|
||||
ss << ::occa::hash(device)
|
||||
<< "Simplex"
|
||||
<< "O1:" << trialFE.GetOrder()
|
||||
<< "O2:" << testFE.GetOrder()
|
||||
<< "Q:" << ir.GetNPoints();
|
||||
std::string hash = ss.str();
|
||||
|
||||
// If we've already made the dof-quad maps, reuse them
|
||||
OccaDofQuadMaps &maps = AllDofQuadMaps[hash];
|
||||
if (!maps.hash.size())
|
||||
{
|
||||
// Create the dof-quad maps
|
||||
maps.hash = hash;
|
||||
|
||||
OccaDofQuadMaps trialMaps = GetD2QSimplexMaps(device, trialFE, ir);
|
||||
OccaDofQuadMaps testMaps = GetD2QSimplexMaps(device, testFE , ir, true);
|
||||
|
||||
maps.dofToQuad = trialMaps.dofToQuad;
|
||||
maps.dofToQuadD = trialMaps.dofToQuadD;
|
||||
maps.quadToDof = testMaps.dofToQuad;
|
||||
maps.quadToDofD = testMaps.dofToQuadD;
|
||||
maps.quadWeights = testMaps.quadWeights;
|
||||
}
|
||||
return maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps OccaDofQuadMaps::GetD2QSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
const int dims = fe.GetDim();
|
||||
const int numDofs = fe.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
OccaDofQuadMaps maps;
|
||||
// Initialize the dof -> quad mapping
|
||||
maps.dofToQuad.allocate(device,
|
||||
numQuad, numDofs);
|
||||
maps.dofToQuadD.allocate(device,
|
||||
dims, numQuad, numDofs);
|
||||
|
||||
if (transpose)
|
||||
{
|
||||
maps.dofToQuad.reindex(1,0);
|
||||
maps.dofToQuadD.reindex(1,0);
|
||||
// Initialize quad weights only for transpose
|
||||
maps.quadWeights.allocate(device,
|
||||
numQuad);
|
||||
}
|
||||
|
||||
mfem::Vector d2q(numDofs);
|
||||
mfem::DenseMatrix d2qD(numDofs, dims);
|
||||
for (int q = 0; q < numQuad; ++q)
|
||||
{
|
||||
const mfem::IntegrationPoint &ip = ir.IntPoint(q);
|
||||
if (transpose)
|
||||
{
|
||||
maps.quadWeights[q] = ip.weight;
|
||||
}
|
||||
fe.CalcShape(ip, d2q);
|
||||
fe.CalcDShape(ip, d2qD);
|
||||
for (int d = 0; d < numDofs; ++d)
|
||||
{
|
||||
const double w = d2q[d];
|
||||
maps.dofToQuad(q, d) = w;
|
||||
for (int dim = 0; dim < dims; ++dim)
|
||||
{
|
||||
const double wD = d2qD(d, dim);
|
||||
maps.dofToQuadD(dim, q, d) = wD;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
maps.dofToQuad.keepInDevice();
|
||||
maps.dofToQuadD.keepInDevice();
|
||||
if (transpose)
|
||||
{
|
||||
maps.quadWeights.keepInDevice();
|
||||
}
|
||||
|
||||
return maps;
|
||||
}
|
||||
|
||||
//---[ Integrator Defines ]-----------
|
||||
std::string stringWithDim(const std::string &s, const int dim)
|
||||
{
|
||||
std::string ret = s;
|
||||
ret += ('0' + (char) dim);
|
||||
ret += 'D';
|
||||
return ret;
|
||||
}
|
||||
|
||||
int closestWarpBatchTo(const int value)
|
||||
{
|
||||
return ((value + 31) / 32) * 32;
|
||||
}
|
||||
|
||||
int closestMultipleWarpBatch(const int multiple, const int maxSize)
|
||||
{
|
||||
if (multiple > maxSize)
|
||||
{
|
||||
return maxSize;
|
||||
}
|
||||
int batch = (32 / multiple);
|
||||
int minDiff = 32 - (multiple * batch);
|
||||
for (int i = 64; i <= maxSize; i += 32)
|
||||
{
|
||||
const int newDiff = i - (multiple * (i / multiple));
|
||||
if (newDiff < minDiff)
|
||||
{
|
||||
batch = (i / multiple);
|
||||
minDiff = newDiff;
|
||||
}
|
||||
}
|
||||
return batch;
|
||||
}
|
||||
|
||||
void SetProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
SetProperties(fespace, fespace, ir, props);
|
||||
}
|
||||
|
||||
void SetProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["defines/TRIAL_VDIM"] = trialFESpace.GetVDim();
|
||||
props["defines/TEST_VDIM"] = testFESpace.GetVDim();
|
||||
props["defines/NUM_DIM"] = trialFESpace.GetDim();
|
||||
|
||||
if (trialFESpace.hasTensorBasis())
|
||||
{
|
||||
SetTensorProperties(trialFESpace, testFESpace, ir, props);
|
||||
}
|
||||
else
|
||||
{
|
||||
SetSimplexProperties(trialFESpace, testFESpace, ir, props);
|
||||
}
|
||||
}
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
SetTensorProperties(fespace, fespace, ir, props);
|
||||
}
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace.GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace.GetFE(0));
|
||||
|
||||
const mfem::IntegrationRule &ir1D =
|
||||
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir.GetOrder());
|
||||
|
||||
const int trialDofs = trialFE.GetDof();
|
||||
const int testDofs = testFE.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
const int trialDofs1D = trialFE.GetOrder() + 1;
|
||||
const int testDofs1D = testFE.GetOrder() + 1;
|
||||
const int quad1D = ir1D.GetNPoints();
|
||||
int trialDofsND = trialDofs1D;
|
||||
int testDofsND = testDofs1D;
|
||||
int quadND = quad1D;
|
||||
|
||||
const bool trialByVDIM = (trialFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
const bool testByVDIM = (testFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
|
||||
props["defines/ORDERING_BY_NODES"] = 0;
|
||||
props["defines/ORDERING_BY_VDIM"] = 1;
|
||||
props["defines/VDIM_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TRIAL_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TEST_ORDERING"] = (int) testByVDIM;
|
||||
|
||||
props["defines/USING_TENSOR_OPS"] = 1;
|
||||
props["defines/NUM_DOFS"] = trialDofs;
|
||||
props["defines/NUM_QUAD"] = numQuad;
|
||||
|
||||
props["defines/TRIAL_DOFS"] = trialDofs;
|
||||
props["defines/TEST_DOFS"] = testDofs;
|
||||
|
||||
for (int d = 1; d <= 3; ++d)
|
||||
{
|
||||
if (d > 1)
|
||||
{
|
||||
trialDofsND *= trialDofs1D;
|
||||
testDofsND *= testDofs1D;
|
||||
quadND *= quad1D;
|
||||
}
|
||||
props["defines"][stringWithDim("NUM_DOFS_", d)] = trialDofsND;
|
||||
props["defines"][stringWithDim("NUM_QUAD_", d)] = quadND;
|
||||
|
||||
props["defines"][stringWithDim("TRIAL_DOFS_", d)] = trialDofsND;
|
||||
props["defines"][stringWithDim("TEST_DOFS_" , d)] = testDofsND;
|
||||
}
|
||||
|
||||
// 1D Defines
|
||||
const int m1InnerBatch = 32 * ((quad1D + 31) / 32);
|
||||
props["defines/A1_ELEMENT_BATCH"] = closestMultipleWarpBatch(quad1D, 512);
|
||||
props["defines/M1_OUTER_ELEMENT_BATCH"] = closestMultipleWarpBatch(m1InnerBatch,
|
||||
512);
|
||||
props["defines/M1_INNER_ELEMENT_BATCH"] = m1InnerBatch;
|
||||
|
||||
// 2D Defines
|
||||
props["defines/A2_ELEMENT_BATCH"] = 1;
|
||||
props["defines/A2_QUAD_BATCH"] = 1;
|
||||
props["defines/M2_ELEMENT_BATCH"] = 32;
|
||||
|
||||
// 3D Defines
|
||||
const int a3QuadBatch = closestMultipleWarpBatch(quadND, 512);
|
||||
props["defines/A3_ELEMENT_BATCH"] = closestMultipleWarpBatch(a3QuadBatch, 512);
|
||||
props["defines/A3_QUAD_BATCH"] = a3QuadBatch;
|
||||
}
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
SetSimplexProperties(fespace, fespace, ir, props);
|
||||
}
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace.GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace.GetFE(0));
|
||||
|
||||
const int trialDofs = trialFE.GetDof();
|
||||
const int testDofs = testFE.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
const int maxDQ = std::max(std::max(trialDofs, testDofs), numQuad);
|
||||
|
||||
const bool trialByVDIM = (trialFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
const bool testByVDIM = (testFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
|
||||
props["defines/ORDERING_BY_NODES"] = 0;
|
||||
props["defines/ORDERING_BY_VDIM"] = 1;
|
||||
props["defines/VDIM_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TRIAL_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TEST_ORDERING"] = (int) testByVDIM;
|
||||
|
||||
props["defines/USING_TENSOR_OPS"] = 0;
|
||||
props["defines/NUM_DOFS"] = trialDofs;
|
||||
props["defines/NUM_QUAD"] = numQuad;
|
||||
|
||||
props["defines/TRIAL_DOFS"] = trialDofs;
|
||||
props["defines/TEST_DOFS"] = testDofs;
|
||||
|
||||
// 2D Defines
|
||||
const int quadBatch = closestWarpBatchTo(numQuad);
|
||||
props["defines/A2_ELEMENT_BATCH"] = closestMultipleWarpBatch(quadBatch, 2048);
|
||||
props["defines/A2_QUAD_BATCH"] = quadBatch;
|
||||
props["defines/M2_INNER_BATCH"] = closestWarpBatchTo(maxDQ);
|
||||
|
||||
// 3D Defines
|
||||
props["defines/A3_ELEMENT_BATCH"] = closestMultipleWarpBatch(quadBatch, 2048);
|
||||
props["defines/A3_QUAD_BATCH"] = quadBatch;
|
||||
props["defines/M3_INNER_BATCH"] = closestWarpBatchTo(maxDQ);
|
||||
}
|
||||
|
||||
|
||||
//---[ Base Integrator ]--------------
|
||||
OccaIntegrator::OccaIntegrator(const Engine &e)
|
||||
: engine(&e),
|
||||
bform(),
|
||||
mesh(),
|
||||
otrialFESpace(),
|
||||
otestFESpace(),
|
||||
trialFESpace(),
|
||||
testFESpace(),
|
||||
itype(DomainIntegrator),
|
||||
ir(NULL),
|
||||
hasTensorBasis(false) { }
|
||||
|
||||
OccaIntegrator::~OccaIntegrator() {}
|
||||
|
||||
void OccaIntegrator::SetupMaps()
|
||||
{
|
||||
maps = OccaDofQuadMaps::Get(GetDevice(),
|
||||
*otrialFESpace,
|
||||
*otestFESpace,
|
||||
*ir);
|
||||
|
||||
mapsTranspose = OccaDofQuadMaps::Get(GetDevice(),
|
||||
*otestFESpace,
|
||||
*otrialFESpace,
|
||||
*ir);
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaIntegrator::GetTrialOccaFESpace() const
|
||||
{
|
||||
return *otrialFESpace;
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaIntegrator::GetTestOccaFESpace() const
|
||||
{
|
||||
return *otestFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaIntegrator::GetTrialFESpace() const
|
||||
{
|
||||
return *trialFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaIntegrator::GetTestFESpace() const
|
||||
{
|
||||
return *testFESpace;
|
||||
}
|
||||
|
||||
void OccaIntegrator::SetIntegrationRule(const mfem::IntegrationRule &ir_)
|
||||
{
|
||||
ir = &ir_;
|
||||
}
|
||||
|
||||
const mfem::IntegrationRule& OccaIntegrator::GetIntegrationRule() const
|
||||
{
|
||||
return *ir;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaIntegrator::GetDofQuadMaps()
|
||||
{
|
||||
return maps;
|
||||
}
|
||||
|
||||
void OccaIntegrator::SetupIntegrator(OccaBilinearForm &bform_,
|
||||
const ::occa::properties &props_,
|
||||
const OccaIntegratorType itype_)
|
||||
{
|
||||
MFEM_ASSERT(engine == &bform_.OccaEngine(), "");
|
||||
bform = &bform_;
|
||||
mesh = &(bform_.GetMesh());
|
||||
|
||||
otrialFESpace = &(bform_.GetTrialOccaFESpace());
|
||||
otestFESpace = &(bform_.GetTestOccaFESpace());
|
||||
|
||||
trialFESpace = &(bform_.GetTrialFESpace());
|
||||
testFESpace = &(bform_.GetTestFESpace());
|
||||
|
||||
hasTensorBasis = otrialFESpace->hasTensorBasis();
|
||||
|
||||
props = props_;
|
||||
itype = itype_;
|
||||
|
||||
if (ir == NULL)
|
||||
{
|
||||
SetupIntegrationRule();
|
||||
}
|
||||
|
||||
SetupMaps();
|
||||
|
||||
SetProperties(*otrialFESpace,
|
||||
*otestFESpace,
|
||||
*ir,
|
||||
props);
|
||||
|
||||
Setup();
|
||||
}
|
||||
|
||||
OccaGeometry OccaIntegrator::GetGeometry(const int flags)
|
||||
{
|
||||
return OccaGeometry::Get(GetDevice(), *otrialFESpace, *ir, flags);
|
||||
}
|
||||
|
||||
::occa::kernel OccaIntegrator::GetAssembleKernel(const ::occa::properties
|
||||
&props)
|
||||
{
|
||||
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
|
||||
return GetKernel(stringWithDim("Assemble", fe.GetDim()),
|
||||
props);
|
||||
}
|
||||
|
||||
::occa::kernel OccaIntegrator::GetMultAddKernel(const ::occa::properties &props)
|
||||
{
|
||||
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
|
||||
return GetKernel(stringWithDim("MultAdd", fe.GetDim()),
|
||||
props);
|
||||
}
|
||||
|
||||
::occa::kernel OccaIntegrator::GetKernel(const std::string &kernelName,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
const std::string filename = GetName() + ".okl";
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
const std::string &okl_defines = OccaEngine().GetOklDefines();
|
||||
return GetDevice().buildKernel(okl_path + filename,
|
||||
kernelName,
|
||||
props + okl_defines);
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Diffusion Integrator ]---------
|
||||
OccaDiffusionIntegrator::OccaDiffusionIntegrator(const OccaCoefficient &coeff_)
|
||||
:
|
||||
OccaIntegrator(coeff_.OccaEngine()),
|
||||
coeff(coeff_),
|
||||
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
|
||||
{
|
||||
coeff.SetName("COEFF");
|
||||
}
|
||||
|
||||
OccaDiffusionIntegrator::~OccaDiffusionIntegrator() {}
|
||||
|
||||
|
||||
std::string OccaDiffusionIntegrator::GetName()
|
||||
{
|
||||
return "DiffusionIntegrator";
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::SetupIntegrationRule()
|
||||
{
|
||||
const FiniteElement &trialFE = *(trialFESpace->GetFE(0));
|
||||
const FiniteElement &testFE = *(testFESpace->GetFE(0));
|
||||
ir = &mfem::DiffusionIntegrator::GetRule(trialFE, testFE);
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::Setup()
|
||||
{
|
||||
::occa::properties kernelProps = props;
|
||||
|
||||
coeff.Setup(*this, kernelProps);
|
||||
|
||||
// Setup assemble and mult kernels
|
||||
assembleKernel = GetAssembleKernel(kernelProps);
|
||||
multKernel = GetMultAddKernel(kernelProps);
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::Assemble()
|
||||
{
|
||||
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
|
||||
|
||||
const int dims = fe.GetDim();
|
||||
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
|
||||
|
||||
const int elements = trialFESpace->GetNE();
|
||||
const int quadraturePoints = ir->GetNPoints();
|
||||
|
||||
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
|
||||
|
||||
assembledOperator.Resize<double>(symmDims * quadraturePoints * elements,
|
||||
NULL);
|
||||
|
||||
assembleKernel((int) mesh->GetNE(),
|
||||
maps.quadWeights,
|
||||
geom.J,
|
||||
coeff,
|
||||
assembledOperator.OccaMem());
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::MultAdd(Vector &x, Vector &y)
|
||||
{
|
||||
// Note: x and y are E-vectors
|
||||
|
||||
multKernel((int) mesh->GetNE(),
|
||||
maps.dofToQuad,
|
||||
maps.dofToQuadD,
|
||||
maps.quadToDof,
|
||||
maps.quadToDofD,
|
||||
assembledOperator.OccaMem(),
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Mass Integrator ]--------------
|
||||
OccaMassIntegrator::OccaMassIntegrator(const OccaCoefficient &coeff_) :
|
||||
OccaIntegrator(coeff_.OccaEngine()),
|
||||
coeff(coeff_),
|
||||
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
|
||||
{
|
||||
coeff.SetName("COEFF");
|
||||
}
|
||||
|
||||
OccaMassIntegrator::~OccaMassIntegrator() {}
|
||||
|
||||
std::string OccaMassIntegrator::GetName()
|
||||
{
|
||||
return "MassIntegrator";
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::SetupIntegrationRule()
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace->GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace->GetFE(0));
|
||||
mfem::ElementTransformation &T = *trialFESpace->GetElementTransformation(0);
|
||||
ir = &mfem::MassIntegrator::GetRule(trialFE, testFE, T);
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::Setup()
|
||||
{
|
||||
::occa::properties kernelProps = props;
|
||||
|
||||
coeff.Setup(*this, kernelProps);
|
||||
|
||||
// Setup assemble and mult kernels
|
||||
assembleKernel = GetAssembleKernel(kernelProps);
|
||||
multKernel = GetMultAddKernel(kernelProps);
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::Assemble()
|
||||
{
|
||||
if (assembledOperator.Size())
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
const int elements = trialFESpace->GetNE();
|
||||
const int quadraturePoints = ir->GetNPoints();
|
||||
|
||||
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
|
||||
|
||||
assembledOperator.Resize<double>(quadraturePoints * elements, NULL);
|
||||
|
||||
assembleKernel((int) mesh->GetNE(),
|
||||
maps.quadWeights,
|
||||
geom.J,
|
||||
coeff,
|
||||
assembledOperator.OccaMem());
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::SetOperator(Vector &v)
|
||||
{
|
||||
assembledOperator = v;
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::MultAdd(Vector &x, Vector &y)
|
||||
{
|
||||
multKernel((int) mesh->GetNE(),
|
||||
maps.dofToQuad,
|
||||
maps.dofToQuadD,
|
||||
maps.quadToDof,
|
||||
maps.quadToDofD,
|
||||
assembledOperator.OccaMem(),
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Vector Mass Integrator ]--------------
|
||||
OccaVectorMassIntegrator::OccaVectorMassIntegrator(const OccaCoefficient &
|
||||
coeff_)
|
||||
:
|
||||
OccaIntegrator(coeff_.OccaEngine()),
|
||||
coeff(coeff_),
|
||||
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
|
||||
{
|
||||
coeff.SetName("COEFF");
|
||||
}
|
||||
|
||||
OccaVectorMassIntegrator::~OccaVectorMassIntegrator() {}
|
||||
|
||||
std::string OccaVectorMassIntegrator::GetName()
|
||||
{
|
||||
return "VectorMassIntegrator";
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::SetupIntegrationRule()
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace->GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace->GetFE(0));
|
||||
mfem::ElementTransformation &T = *trialFESpace->GetElementTransformation(0);
|
||||
ir = &mfem::MassIntegrator::GetRule(trialFE, testFE, T);
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::Setup()
|
||||
{
|
||||
::occa::properties kernelProps = props;
|
||||
|
||||
coeff.Setup(*this, kernelProps);
|
||||
|
||||
// Setup assemble and mult kernels
|
||||
assembleKernel = GetAssembleKernel(kernelProps);
|
||||
multKernel = GetMultAddKernel(kernelProps);
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::Assemble()
|
||||
{
|
||||
const int elements = trialFESpace->GetNE();
|
||||
const int quadraturePoints = ir->GetNPoints();
|
||||
|
||||
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
|
||||
|
||||
assembledOperator.Resize<double>(quadraturePoints * elements, NULL);
|
||||
|
||||
assembleKernel((int) mesh->GetNE(),
|
||||
maps.quadWeights,
|
||||
geom.J,
|
||||
coeff,
|
||||
assembledOperator.OccaMem());
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::MultAdd(Vector &x, Vector &y)
|
||||
{
|
||||
multKernel((int) mesh->GetNE(),
|
||||
maps.dofToQuad,
|
||||
maps.dofToQuadD,
|
||||
maps.quadToDof,
|
||||
maps.quadToDofD,
|
||||
assembledOperator.OccaMem(),
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,323 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
|
||||
#define MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "coefficient.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class OccaGeometry
|
||||
{
|
||||
public:
|
||||
::occa::array<double> meshNodes;
|
||||
::occa::array<double> J, invJ, detJ;
|
||||
|
||||
// byVDIM -> [x y z x y z x y z]
|
||||
// byNodes -> [x x x y y y z z z]
|
||||
static const int Jacobian = (1 << 0);
|
||||
static const int JacobianInv = (1 << 1);
|
||||
static const int JacobianDet = (1 << 2);
|
||||
|
||||
static OccaGeometry Get(::occa::device device,
|
||||
FiniteElementSpace &ofespace,
|
||||
const IntegrationRule &ir,
|
||||
const int flags = (Jacobian |
|
||||
JacobianInv |
|
||||
JacobianDet));
|
||||
};
|
||||
|
||||
class OccaDofQuadMaps
|
||||
{
|
||||
private:
|
||||
// Reuse dof-quad maps
|
||||
static std::map<std::string, OccaDofQuadMaps> AllDofQuadMaps;
|
||||
std::string hash;
|
||||
|
||||
public:
|
||||
// Local stiffness matrices (B and B^T operators)
|
||||
::occa::array<double, ::occa::dynamic> dofToQuad, dofToQuadD; // B
|
||||
::occa::array<double, ::occa::dynamic> quadToDof, quadToDofD; // B^T
|
||||
::occa::array<double> quadWeights;
|
||||
|
||||
OccaDofQuadMaps();
|
||||
OccaDofQuadMaps(const OccaDofQuadMaps &maps);
|
||||
OccaDofQuadMaps& operator = (const OccaDofQuadMaps &maps);
|
||||
|
||||
// [[x y] [x y] [x y]]
|
||||
// [[x y z] [x y z] [x y z]]
|
||||
// mfem::GridFunction* mfem::Mesh::GetNodes() { return Nodes; }
|
||||
|
||||
// mfem::FiniteElementSpace *Nodes->FESpace()
|
||||
// 25
|
||||
// 1D [x x x x x x]
|
||||
// 2D [x y x y x y]
|
||||
// GetVdim()
|
||||
// 3D ordering == byVDIM -> [x y z x y z x y z x y z x y z x y z]
|
||||
// ordering == byNODES -> [x x x x x x y y y y y y z z z z z z]
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const FiniteElementSpace &trialFESpace,
|
||||
const FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps GetD2QTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps GetD2QSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
};
|
||||
|
||||
//---[ Define Methods ]---------------
|
||||
std::string stringWithDim(const std::string &s, const int dim);
|
||||
int closestWarpBatch(const int multiple, const int maxSize);
|
||||
|
||||
void SetProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &fespace,
|
||||
const IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
//---[ Base Integrator ]--------------
|
||||
class OccaIntegrator
|
||||
{
|
||||
protected:
|
||||
SharedPtr<const Engine> engine;
|
||||
|
||||
OccaBilinearForm *bform;
|
||||
mfem::Mesh *mesh;
|
||||
|
||||
FiniteElementSpace *otrialFESpace;
|
||||
FiniteElementSpace *otestFESpace;
|
||||
|
||||
mfem::FiniteElementSpace *trialFESpace;
|
||||
mfem::FiniteElementSpace *testFESpace;
|
||||
|
||||
::occa::properties props;
|
||||
OccaIntegratorType itype;
|
||||
|
||||
const IntegrationRule *ir;
|
||||
bool hasTensorBasis;
|
||||
OccaDofQuadMaps maps;
|
||||
OccaDofQuadMaps mapsTranspose;
|
||||
|
||||
public:
|
||||
OccaIntegrator(const Engine &e);
|
||||
virtual ~OccaIntegrator();
|
||||
|
||||
const Engine &OccaEngine() const { return *engine; }
|
||||
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return engine->GetDevice(idx); }
|
||||
|
||||
virtual std::string GetName() = 0;
|
||||
|
||||
FiniteElementSpace& GetTrialOccaFESpace() const;
|
||||
FiniteElementSpace& GetTestOccaFESpace() const;
|
||||
|
||||
mfem::FiniteElementSpace& GetTrialFESpace() const;
|
||||
mfem::FiniteElementSpace& GetTestFESpace() const;
|
||||
|
||||
void SetIntegrationRule(const mfem::IntegrationRule &ir_);
|
||||
const mfem::IntegrationRule& GetIntegrationRule() const;
|
||||
|
||||
OccaDofQuadMaps& GetDofQuadMaps();
|
||||
|
||||
void SetupMaps();
|
||||
|
||||
virtual void SetupIntegrationRule() = 0;
|
||||
|
||||
virtual void SetupIntegrator(OccaBilinearForm &bform_,
|
||||
const ::occa::properties &props_,
|
||||
const OccaIntegratorType itype_);
|
||||
|
||||
virtual void Setup() = 0;
|
||||
|
||||
virtual void Assemble() = 0;
|
||||
/// This method works on E-vectors!
|
||||
virtual void MultAdd(Vector &x, Vector &y) = 0;
|
||||
|
||||
virtual void MultTransposeAdd(Vector &x, Vector &y)
|
||||
{
|
||||
mfem_error("OccaIntegrator::MultTransposeAdd() is not overloaded!");
|
||||
}
|
||||
|
||||
OccaGeometry GetGeometry(const int flags = (OccaGeometry::Jacobian |
|
||||
OccaGeometry::JacobianInv |
|
||||
OccaGeometry::JacobianDet));
|
||||
|
||||
::occa::kernel GetAssembleKernel(const ::occa::properties &props);
|
||||
::occa::kernel GetMultAddKernel(const ::occa::properties &props);
|
||||
|
||||
::occa::kernel GetKernel(const std::string &kernelName,
|
||||
const ::occa::properties &props);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Diffusion Integrator ]---------
|
||||
class OccaDiffusionIntegrator : public OccaIntegrator
|
||||
{
|
||||
private:
|
||||
OccaCoefficient coeff;
|
||||
|
||||
::occa::kernel assembleKernel, multKernel;
|
||||
|
||||
Vector assembledOperator;
|
||||
|
||||
public:
|
||||
OccaDiffusionIntegrator(const OccaCoefficient &coeff_);
|
||||
virtual ~OccaDiffusionIntegrator();
|
||||
|
||||
virtual std::string GetName();
|
||||
|
||||
virtual void SetupIntegrationRule();
|
||||
|
||||
virtual void Setup();
|
||||
|
||||
virtual void Assemble();
|
||||
virtual void MultAdd(Vector &x, Vector &y);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Mass Integrator ]--------------
|
||||
class OccaMassIntegrator : public OccaIntegrator
|
||||
{
|
||||
private:
|
||||
OccaCoefficient coeff;
|
||||
|
||||
::occa::kernel assembleKernel, multKernel;
|
||||
|
||||
Vector assembledOperator;
|
||||
|
||||
public:
|
||||
OccaMassIntegrator(const OccaCoefficient &coeff_);
|
||||
virtual ~OccaMassIntegrator();
|
||||
|
||||
virtual std::string GetName();
|
||||
|
||||
virtual void SetupIntegrationRule();
|
||||
|
||||
virtual void Setup();
|
||||
|
||||
virtual void Assemble();
|
||||
void SetOperator(Vector &v);
|
||||
|
||||
virtual void MultAdd(Vector &x, Vector &y);
|
||||
};
|
||||
//====================================
|
||||
|
||||
//---[ Vector Mass Integrator ]--------------
|
||||
class OccaVectorMassIntegrator : public OccaIntegrator
|
||||
{
|
||||
private:
|
||||
OccaCoefficient coeff;
|
||||
|
||||
::occa::kernel assembleKernel, multKernel;
|
||||
|
||||
Vector assembledOperator;
|
||||
|
||||
public:
|
||||
OccaVectorMassIntegrator(const OccaCoefficient &coeff_);
|
||||
virtual ~OccaVectorMassIntegrator();
|
||||
|
||||
virtual std::string GetName();
|
||||
|
||||
virtual void SetupIntegrationRule();
|
||||
|
||||
virtual void Setup();
|
||||
|
||||
virtual void Assemble();
|
||||
|
||||
virtual void MultAdd(Vector &x, Vector &y);
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
|
||||
@@ -1,344 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "coefficient.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
//---[ Parameter ]------------
|
||||
OccaParameter::~OccaParameter() {}
|
||||
|
||||
void OccaParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props) {}
|
||||
|
||||
::occa::kernelArg OccaParameter::KernelArgs()
|
||||
{
|
||||
return ::occa::kernelArg();
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Include Parameter ]------------
|
||||
OccaIncludeParameter::OccaIncludeParameter(const std::string &filename_) :
|
||||
filename(filename_) {}
|
||||
|
||||
OccaParameter* OccaIncludeParameter::Clone()
|
||||
{
|
||||
return new OccaIncludeParameter(filename);
|
||||
}
|
||||
|
||||
void OccaIncludeParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["headers"].asArray() += "#include " + filename;
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Source Parameter ]------------
|
||||
OccaSourceParameter::OccaSourceParameter(const std::string &source_) :
|
||||
source(source_) {}
|
||||
|
||||
OccaParameter* OccaSourceParameter::Clone()
|
||||
{
|
||||
return new OccaSourceParameter(source);
|
||||
}
|
||||
|
||||
void OccaSourceParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["headers"].asArray() += source;
|
||||
}
|
||||
//====================================
|
||||
|
||||
//---[ Vector Parameter ]-------
|
||||
OccaVectorParameter::OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const bool useRestrict_) :
|
||||
name(name_),
|
||||
v(v_),
|
||||
useRestrict(useRestrict_),
|
||||
attr("") {}
|
||||
|
||||
OccaVectorParameter::OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const std::string &attr_,
|
||||
const bool useRestrict_) :
|
||||
name(name_),
|
||||
v(v_),
|
||||
useRestrict(useRestrict_),
|
||||
attr(attr_) {}
|
||||
|
||||
OccaParameter* OccaVectorParameter::Clone()
|
||||
{
|
||||
return new OccaVectorParameter(name, v, attr, useRestrict);
|
||||
}
|
||||
|
||||
void OccaVectorParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
std::string &args = (props["defines/COEFF_ARGS"]
|
||||
.asString()
|
||||
.string());
|
||||
args += "const double *";
|
||||
if (useRestrict)
|
||||
{
|
||||
args += " restrict ";
|
||||
}
|
||||
args += name;
|
||||
if (attr.size())
|
||||
{
|
||||
args += ' ';
|
||||
args += attr;
|
||||
}
|
||||
args += ",\n";
|
||||
}
|
||||
|
||||
::occa::kernelArg OccaVectorParameter::KernelArgs()
|
||||
{
|
||||
return ::occa::kernelArg(v.OccaMem());
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ GridFunction Parameter ]-------
|
||||
OccaGridFunctionParameter::OccaGridFunctionParameter(const std::string &name_,
|
||||
OccaGridFunction &gf_,
|
||||
const bool useRestrict_)
|
||||
: name(name_),
|
||||
gf(gf_),
|
||||
gfQuad(*(new Layout(gf_.OccaLayout().OccaEngine(), 0))),
|
||||
useRestrict(useRestrict_) {}
|
||||
|
||||
OccaParameter* OccaGridFunctionParameter::Clone()
|
||||
{
|
||||
OccaGridFunctionParameter *param =
|
||||
new OccaGridFunctionParameter(name, gf, useRestrict);
|
||||
param->gfQuad.MakeRef(gfQuad);
|
||||
return param;
|
||||
}
|
||||
|
||||
void OccaGridFunctionParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
|
||||
std::string &args = (props["defines/COEFF_ARGS"]
|
||||
.asString()
|
||||
.string());
|
||||
args += "const double *";
|
||||
if (useRestrict)
|
||||
{
|
||||
args += " restrict ";
|
||||
}
|
||||
args += name;
|
||||
args += " @dim(NUM_QUAD, numElements),\n";
|
||||
|
||||
gf.ToQuad(integ.GetIntegrationRule(), gfQuad);
|
||||
}
|
||||
|
||||
::occa::kernelArg OccaGridFunctionParameter::KernelArgs()
|
||||
{
|
||||
return gfQuad.OccaMem();
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Coefficient ]------------------
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, const double value) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = value;
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, const std::string &source) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = source;
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, const char *source) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = source;
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const OccaCoefficient &coeff) :
|
||||
engine(coeff.engine),
|
||||
integ(NULL),
|
||||
name(coeff.name),
|
||||
coeffValue(coeff.coeffValue)
|
||||
{
|
||||
|
||||
const int paramCount = (int) coeff.params.size();
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
params.push_back(coeff.params[i]->Clone());
|
||||
}
|
||||
}
|
||||
|
||||
OccaCoefficient::~OccaCoefficient()
|
||||
{
|
||||
const int paramCount = (int) params.size();
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
delete params[i];
|
||||
}
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::SetName(const std::string &name_)
|
||||
{
|
||||
name = name_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
void OccaCoefficient::Setup(OccaIntegrator &integ_,
|
||||
::occa::properties &props_)
|
||||
{
|
||||
integ = &integ_;
|
||||
|
||||
const int paramCount = (int) params.size();
|
||||
props_["defines"][name + "_ARGS"] = "";
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
params[i]->Setup(integ_, props_);
|
||||
}
|
||||
props_["defines"][name] = coeffValue;
|
||||
|
||||
props = props_;
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::Add(OccaParameter *param)
|
||||
{
|
||||
params.push_back(param);
|
||||
return *this;
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::IncludeHeader(const std::string &filename)
|
||||
{
|
||||
return Add(new OccaIncludeParameter(filename));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::IncludeSource(const std::string &source)
|
||||
{
|
||||
return Add(new OccaSourceParameter(source));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const bool useRestrict)
|
||||
{
|
||||
return Add(new OccaVectorParameter(name_, v, useRestrict));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const std::string &attr,
|
||||
const bool useRestrict)
|
||||
{
|
||||
return Add(new OccaVectorParameter(name_, v, attr, useRestrict));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::AddGridFunction(const std::string &name_,
|
||||
OccaGridFunction &gf,
|
||||
const bool useRestrict)
|
||||
{
|
||||
return Add(new OccaGridFunctionParameter(name_, gf, useRestrict));
|
||||
}
|
||||
|
||||
bool OccaCoefficient::IsConstant()
|
||||
{
|
||||
return coeffValue.isNumber();
|
||||
}
|
||||
|
||||
double OccaCoefficient::GetConstantValue()
|
||||
{
|
||||
if (!IsConstant())
|
||||
{
|
||||
mfem_error("OccaCoefficient is not constant");
|
||||
}
|
||||
return coeffValue.number();
|
||||
}
|
||||
|
||||
Vector OccaCoefficient::Eval()
|
||||
{
|
||||
if (integ == NULL)
|
||||
{
|
||||
mfem_error("OccaCoefficient requires a Setup() call before Eval()");
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace &fespace = integ->GetTrialFESpace();
|
||||
const mfem::IntegrationRule &ir = integ->GetIntegrationRule();
|
||||
|
||||
const int elements = fespace.GetNE();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
Vector quadCoeff(*(new Layout(OccaEngine(), numQuad * elements)));
|
||||
Eval(quadCoeff);
|
||||
return quadCoeff;
|
||||
}
|
||||
|
||||
void OccaCoefficient::Eval(Vector &quadCoeff)
|
||||
{
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
const std::string &okl_defines = OccaEngine().GetOklDefines();
|
||||
static ::occa::kernelBuilder builder =
|
||||
::occa::kernelBuilder::fromFile(okl_path + "coefficient.okl",
|
||||
"CoefficientEval", okl_defines);
|
||||
|
||||
if (integ == NULL)
|
||||
{
|
||||
mfem_error("OccaCoefficient requires a Setup() call before Eval()");
|
||||
}
|
||||
|
||||
const int elements = integ->GetTrialFESpace().GetNE();
|
||||
|
||||
::occa::properties kernelProps = props;
|
||||
if (name != "COEFF")
|
||||
{
|
||||
kernelProps["defines/COEFF"] = name;
|
||||
kernelProps["defines/COEFF_ARGS"] = name + "_ARGS";
|
||||
}
|
||||
kernelProps += okl_defines;
|
||||
|
||||
::occa::kernel evalKernel = builder.build(GetDevice(), kernelProps);
|
||||
evalKernel(elements, *this, quadCoeff.OccaMem());
|
||||
}
|
||||
|
||||
OccaCoefficient::operator ::occa::kernelArg ()
|
||||
{
|
||||
::occa::kernelArg kArg;
|
||||
const int paramCount = (int) params.size();
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
kArg.add(params[i]->KernelArgs());
|
||||
}
|
||||
return kArg;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,284 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
|
||||
#define MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class OccaIntegrator;
|
||||
|
||||
|
||||
class OccaParameter
|
||||
{
|
||||
public:
|
||||
virtual ~OccaParameter();
|
||||
|
||||
virtual OccaParameter* Clone() = 0;
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs();
|
||||
};
|
||||
|
||||
|
||||
//---[ Include Parameter ]------------
|
||||
class OccaIncludeParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
std::string filename;
|
||||
|
||||
public:
|
||||
OccaIncludeParameter(const std::string &filename_);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Source Parameter ]------------
|
||||
class OccaSourceParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
std::string source;
|
||||
|
||||
public:
|
||||
OccaSourceParameter(const std::string &filename_);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Define Parameter ]------------
|
||||
template <class TM>
|
||||
class OccaDefineParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
TM value;
|
||||
|
||||
public:
|
||||
OccaDefineParameter(const std::string &name_,
|
||||
const TM &value_) :
|
||||
name(name_),
|
||||
value(value_) {}
|
||||
|
||||
virtual OccaParameter* Clone()
|
||||
{
|
||||
return new OccaDefineParameter(name, value);
|
||||
}
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["defines"][name] = value;
|
||||
}
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Variable Parameter ]-----------
|
||||
template <class TM>
|
||||
class OccaVariableParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
const TM &value;
|
||||
|
||||
public:
|
||||
OccaVariableParameter(const std::string &name_,
|
||||
const TM &value_) :
|
||||
name(name_),
|
||||
value(value_) {}
|
||||
|
||||
virtual OccaParameter* Clone()
|
||||
{
|
||||
return new OccaVariableParameter(name, value);
|
||||
}
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
std::string &args = (props["defines/COEFF_ARGS"]
|
||||
.asString()
|
||||
.string());
|
||||
// const TM name,\n"
|
||||
args += "const ";
|
||||
args += ::occa::primitiveinfo<TM>::name;
|
||||
args += ' ';
|
||||
args += name;
|
||||
args += ",\n";
|
||||
}
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs()
|
||||
{
|
||||
return ::occa::kernelArg(value);
|
||||
}
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Vector Parameter ]-------
|
||||
class OccaVectorParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
Vector v;
|
||||
bool useRestrict;
|
||||
std::string attr;
|
||||
|
||||
public:
|
||||
OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const bool useRestrict_ = false);
|
||||
|
||||
OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const std::string &attr_,
|
||||
const bool useRestrict_ = false);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs();
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ GridFunction Parameter ]-------
|
||||
class OccaGridFunctionParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
OccaGridFunction &gf;
|
||||
Vector gfQuad;
|
||||
bool useRestrict;
|
||||
|
||||
public:
|
||||
OccaGridFunctionParameter(const std::string &name_,
|
||||
OccaGridFunction &gf_,
|
||||
const bool useRestrict_ = false);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs();
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Coefficient ]------------------
|
||||
// [MISSING]
|
||||
// Needs to know about the integrator's
|
||||
// - fespace
|
||||
// - ir
|
||||
// Step where parameters that need the ir get called for setup
|
||||
// For example, GridFunction (d, e) -> (q, e)
|
||||
class OccaCoefficient
|
||||
{
|
||||
private:
|
||||
SharedPtr<const Engine> engine;
|
||||
|
||||
OccaIntegrator *integ;
|
||||
|
||||
std::string name;
|
||||
::occa::json coeffValue;
|
||||
|
||||
::occa::properties props;
|
||||
std::vector<OccaParameter*> params;
|
||||
|
||||
public:
|
||||
OccaCoefficient(const Engine &e, const double value = 1.0);
|
||||
OccaCoefficient(const Engine &e, const std::string &source);
|
||||
OccaCoefficient(const Engine &e, const char *source);
|
||||
~OccaCoefficient();
|
||||
|
||||
OccaCoefficient(const OccaCoefficient &coeff);
|
||||
|
||||
const Engine &OccaEngine() const { return *engine; }
|
||||
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return engine->GetDevice(idx); }
|
||||
|
||||
OccaCoefficient& SetName(const std::string &name_);
|
||||
|
||||
void Setup(OccaIntegrator &integ_,
|
||||
::occa::properties &props_);
|
||||
|
||||
OccaCoefficient& Add(OccaParameter *param);
|
||||
|
||||
OccaCoefficient& IncludeHeader(const std::string &filename);
|
||||
OccaCoefficient& IncludeSource(const std::string &source);
|
||||
|
||||
template <class TM>
|
||||
OccaCoefficient& AddDefine(const std::string &name_, const TM &value)
|
||||
{
|
||||
return Add(new OccaDefineParameter<TM>(name_, value));
|
||||
}
|
||||
|
||||
template <class TM>
|
||||
OccaCoefficient& AddVariable(const std::string &name_, const TM &value)
|
||||
{
|
||||
return Add(new OccaVariableParameter<TM>(name_, value));
|
||||
}
|
||||
|
||||
OccaCoefficient& AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const bool useRestrict = false);
|
||||
|
||||
|
||||
OccaCoefficient& AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const std::string &attr,
|
||||
const bool useRestrict = false);
|
||||
|
||||
OccaCoefficient& AddGridFunction(const std::string &name_,
|
||||
OccaGridFunction &gf,
|
||||
const bool useRestrict = false);
|
||||
|
||||
bool IsConstant();
|
||||
double GetConstantValue();
|
||||
|
||||
Vector Eval();
|
||||
void Eval(Vector &quadCoeff);
|
||||
|
||||
operator ::occa::kernelArg ();
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
|
||||
@@ -1,40 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_OCCA_DEFINES
|
||||
#define MFEM_OCCA_DEFINES
|
||||
|
||||
#ifndef USING_TENSOR_OPS
|
||||
# define USING_TENSOR_OPS 0
|
||||
#endif
|
||||
|
||||
#ifdef OCCA_USING_GPU
|
||||
# define GPU_ORDER_2(I0, I1) @dimOrder(I0, I1)
|
||||
# define GPU_ORDER_3(I0, I1, I2) @dimOrder(I0, I1, I2)
|
||||
# define GPU_ORDER_4(I0, I1, I2, I3) @dimOrder(I0, I1, I2, I3)
|
||||
#else
|
||||
# define GPU_ORDER_2(I0, I1) @dimOrder(0, 1)
|
||||
# define GPU_ORDER_3(I0, I1, I2) @dimOrder(0, 1, 2)
|
||||
# define GPU_ORDER_4(I0, I1, I2, I3) @dimOrder(0, 1, 2, 3)
|
||||
#endif
|
||||
|
||||
#ifndef COEFF
|
||||
# define COEFF 1.0
|
||||
# define COEFF_ARGS
|
||||
#endif
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# include "mfem-occa://defines/tensor.okl"
|
||||
#else
|
||||
# include "mfem-occa://defines/simplex.okl"
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -1,40 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#define USING_LOW_ORDER 1
|
||||
#define USING_HI_ORDER 0
|
||||
|
||||
typedef double* DofToQuad_t @dim(NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD2D_t @dim(2, NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD3D_t @dim(3, NUM_QUAD, NUM_DOFS);
|
||||
|
||||
typedef double* QuadToDof_t @dim(NUM_DOFS, NUM_QUAD);
|
||||
typedef double* QuadToDofD2D_t @dim(2, NUM_DOFS, NUM_QUAD);
|
||||
typedef double* QuadToDofD3D_t @dim(3, NUM_DOFS, NUM_QUAD);
|
||||
|
||||
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD, numElements);
|
||||
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD, numElements);
|
||||
|
||||
typedef double* SymmOperator2D_t @dim(3, NUM_QUAD, numElements);
|
||||
typedef double* SymmOperator3D_t @dim(6, NUM_QUAD, numElements);
|
||||
|
||||
typedef double* DLocal_t @dim(NUM_DOFS, numElements);
|
||||
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
|
||||
|
||||
#if VDIM_ORDERING == ORDERING_BY_VDIM
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements);
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements);
|
||||
#else
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements) @dimOrder(2,0,1);
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements) @dimOrder(2,0,1);
|
||||
#endif
|
||||
|
||||
typedef int* DLocalMap_t @dim(NUM_DOFS, numElements);
|
||||
@@ -1,85 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#if NUM_QUAD_1D < NUM_DOFS_1D
|
||||
# define NUM_MAX_1D NUM_DOFS_1D
|
||||
#else
|
||||
# define NUM_MAX_1D NUM_QUAD_1D
|
||||
#endif
|
||||
|
||||
#define NUM_MAX_2D (NUM_MAX_1D * NUM_MAX_1D)
|
||||
|
||||
#define NUM_QUAD_DOFS_1D (NUM_QUAD_1D * NUM_DOFS_1D)
|
||||
|
||||
#define QUAD_2D_ID(X, Y) (X + ((Y) * NUM_QUAD_1D))
|
||||
#define DOFS_2D_ID(X, Y) (X + ((Y) * NUM_DOFS_1D))
|
||||
|
||||
#define QUAD_3D_ID(X, Y, Z) (X + ((Y) * NUM_QUAD_1D) + ((Z) * NUM_QUAD_2D))
|
||||
#define DOFS_3D_ID(X, Y, Z) (X + ((Y) * NUM_DOFS_1D) + ((Z) * NUM_DOFS_2D))
|
||||
|
||||
#if NUM_MAX_1D < 8
|
||||
# define USING_LOW_ORDER 1
|
||||
# define USING_HI_ORDER 0
|
||||
#else
|
||||
# define USING_LOW_ORDER 0
|
||||
# define USING_HI_ORDER 1
|
||||
#endif
|
||||
|
||||
#define M1_ELEMENT_BATCHES (M1_OUTER_ELEMENT_BATCH * M1_INNER_ELEMENT_BATCH)
|
||||
|
||||
typedef double* DofToQuad_t @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
typedef double* QuadToDof_t @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
typedef double* Jacobian_t @dim(NUM_DIM, NUM_DIM, numElements);
|
||||
typedef double* Jacobian1D_t @dim(NUM_QUAD_1D, numElements);
|
||||
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD_2D, numElements);
|
||||
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD_3D, numElements);
|
||||
|
||||
typedef double* SymmOperator1D_t @dim(NUM_QUAD_1D, numElements);
|
||||
typedef double* SymmOperator2D_t @dim(3, NUM_QUAD_2D, numElements);
|
||||
typedef double* SymmOperator3D_t @dim(6, NUM_QUAD_3D, numElements);
|
||||
|
||||
typedef double* DLocal_t @dim(NUM_DOFS, numElements);
|
||||
typedef double* DLocal1D_t @dim(NUM_DOFS_1D, numElements);
|
||||
typedef double* DLocal2D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
typedef double* DLocal3D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
|
||||
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
|
||||
typedef double* QLocal1D_t @dim(NUM_QUAD_1D, numElements);
|
||||
typedef double* QLocal2D_t @dim(NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
typedef double* QLocal3D_t @dim(NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
|
||||
#if VDIM_ORDERING == ORDERING_BY_VDIM
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements);
|
||||
typedef double* DVLocal1D_t @dim(NUM_VDIM, NUM_DOFS_1D, numElements);
|
||||
typedef double* DVLocal2D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
typedef double* DVLocal3D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements);
|
||||
typedef double* QVLocal1D_t @dim(NUM_VDIM, NUM_QUAD_1D, numElements);
|
||||
typedef double* QVLocal2D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
typedef double* QVLocal3D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
#else
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements) @dimOrder(2,0,1);
|
||||
typedef double* DVLocal1D_t @dim(NUM_VDIM, NUM_DOFS_1D, numElements) @dimOrder(2,0,1);
|
||||
typedef double* DVLocal2D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, numElements) @dimOrder(3,0,1,2);
|
||||
typedef double* DVLocal3D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements) @dimOrder(4,0,1,2,3);
|
||||
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements) @dimOrder(2,0,1);
|
||||
typedef double* QVLocal1D_t @dim(NUM_VDIM, NUM_QUAD_1D, numElements) @dimOrder(2,0,1);
|
||||
typedef double* QVLocal2D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, numElements) @dimOrder(3,0,1,2);
|
||||
typedef double* QVLocal3D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements) @dimOrder(4,0,1,2,3);
|
||||
#endif
|
||||
|
||||
typedef int* DLocalMap_t @dim(NUM_DOFS, numElements);
|
||||
typedef int* DLocalMap1D_t @dim(NUM_DOFS_1D, numElements);
|
||||
typedef int* DLocalMap2D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
typedef int* DLocalMap3D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
@@ -1,168 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
const double * restrict quadWeights,
|
||||
const Jacobian2D_t restrict J,
|
||||
COEFF_ARGS
|
||||
SymmOperator2D_t restrict oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J12*J12 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J12*J11 + J22*J21); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J21*J21); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DofToQuadD2D_t restrict dofToQuadD,
|
||||
const QuadToDof_t restrict quadToDof,
|
||||
const QuadToDofD2D_t restrict quadToDofD,
|
||||
const SymmOperator2D_t restrict oper,
|
||||
const DLocal_t restrict solIn,
|
||||
DLocal_t restrict solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_sol[NUM_DOFS];
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] = 0;
|
||||
}
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
double gradX = 0, gradY = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
const double gradX2 = (O11 * gradX) + (O12 * gradY);
|
||||
const double gradY2 = (O12 * gradX) + (O22 * gradY);
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] += ((gradX2 * quadToDofD(0, d, q)) +
|
||||
(gradY2 * quadToDofD(1, d, q)));
|
||||
}
|
||||
}
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
solOut(d, e) += r_sol[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
const double * restrict quadWeights,
|
||||
const Jacobian3D_t restrict J,
|
||||
COEFF_ARGS
|
||||
SymmOperator3D_t restrict oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3) + (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3) + (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DofToQuadD3D_t restrict dofToQuadD,
|
||||
const QuadToDof_t restrict quadToDof,
|
||||
const QuadToDofD3D_t restrict quadToDofD,
|
||||
const SymmOperator3D_t restrict oper,
|
||||
const DLocal_t restrict solIn,
|
||||
DLocal_t restrict solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_sol[NUM_DOFS];
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] = 0;
|
||||
}
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
double gradX = 0, gradY = 0, gradZ = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
gradZ += s * quadToDofD(2, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
const double gradX2 = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
|
||||
const double gradY2 = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
|
||||
const double gradZ2 = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] += ((gradX2 * quadToDofD(0, d, q)) +
|
||||
(gradY2 * quadToDofD(1, d, q)) +
|
||||
(gradZ2 * quadToDofD(2, d, q)));
|
||||
}
|
||||
}
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
solOut(d, e) += r_sol[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,182 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator2D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += A2_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J12*J12 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J12*J11 + J22*J21); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J21*J21); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DofToQuadD2D_t restrict dofToQuadD,
|
||||
const QuadToDof_t restrict quadToDof,
|
||||
const QuadToDofD2D_t restrict quadToDofD,
|
||||
const SymmOperator2D_t restrict oper,
|
||||
const DLocal_t restrict solIn,
|
||||
DLocal_t restrict solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gradX[NUM_QUAD];
|
||||
@shared double s_gradY[NUM_QUAD];
|
||||
|
||||
for (int qOff = 0; qOff < M2_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M2_INNER_BATCH) {
|
||||
double gradX = 0, gradY = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
s_gradX[q] = (O11 * gradX) + (O12 * gradY);
|
||||
s_gradY[q] = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
|
||||
for (int dOff = 0; dOff < M2_INNER_BATCH; ++dOff) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M2_INNER_BATCH) {
|
||||
double r_sol = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
// FIXME: s_gradX and s_gradY are @shared used outside of @inner
|
||||
r_sol += ((s_gradX[q] * quadToDofD(0, d, q)) +
|
||||
(s_gradY[q] * quadToDofD(1, d, q)));
|
||||
}
|
||||
solOut(d, e) += r_sol;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator3D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += A3_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3) + (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3) + (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DofToQuadD3D_t restrict dofToQuadD,
|
||||
const QuadToDof_t restrict quadToDof,
|
||||
const QuadToDofD3D_t restrict quadToDofD,
|
||||
const SymmOperator3D_t restrict oper,
|
||||
const DLocal_t restrict solIn,
|
||||
DLocal_t restrict solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gradX[NUM_QUAD];
|
||||
@shared double s_gradY[NUM_QUAD];
|
||||
@shared double s_gradZ[NUM_QUAD];
|
||||
|
||||
for (int qOff = 0; qOff < M3_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M3_INNER_BATCH) {
|
||||
double gradX = 0, gradY = 0, gradZ = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
gradZ += s * quadToDofD(2, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
s_gradX[q] = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
|
||||
s_gradY[q] = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
|
||||
s_gradZ[q] = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
|
||||
}
|
||||
}
|
||||
|
||||
for (int dOff = 0; dOff < M3_INNER_BATCH; ++dOff) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M3_INNER_BATCH) {
|
||||
double r_sol = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_sol += ((s_gradX[q] * quadToDofD(0, d, q)) +
|
||||
(s_gradY[q] * quadToDofD(1, d, q)) +
|
||||
(s_gradZ[q] * quadToDofD(2, d, q)));
|
||||
}
|
||||
solOut(d, e) += r_sol;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,370 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
const double * restrict quadWeights,
|
||||
const Jacobian1D_t restrict J,
|
||||
COEFF_ARGS
|
||||
SymmOperator1D_t restrict oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
oper(q, e) = quadWeights[q] * COEFF / J(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DofToQuad_t restrict dofToQuadD,
|
||||
const QuadToDof_t restrict quadToDof,
|
||||
const QuadToDof_t restrict quadToDofD,
|
||||
const SymmOperator1D_t restrict oper,
|
||||
const DLocal1D_t restrict solIn,
|
||||
DLocal1D_t restrict solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double grad[NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] += s * dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] *= oper(qx, e);
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double gradX = grad[qx];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, e) += gradX * quadToDofD(dx, qx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
const double * restrict quadWeights,
|
||||
const Jacobian2D_t restrict J,
|
||||
COEFF_ARGS
|
||||
SymmOperator2D_t restrict oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_2D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J21*J21 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J21*J11 + J22*J12); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J12*J12); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DofToQuad_t restrict dofToQuadD,
|
||||
const QuadToDof_t restrict quadToDof,
|
||||
const QuadToDof_t restrict quadToDofD,
|
||||
const SymmOperator2D_t restrict oper,
|
||||
const DLocal2D_t restrict solIn,
|
||||
DLocal2D_t restrict solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double grad[NUM_QUAD_1D][NUM_QUAD_1D][2];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qy][qx][0] = 0;
|
||||
grad[qy][qx][1] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double gradX[NUM_QUAD_1D][2];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] = 0;
|
||||
gradX[qx][1] = 0;
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, dy, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] += s * dofToQuad(qx, dx);
|
||||
gradX[qx][1] += s * dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
const double wDy = dofToQuadD(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qy][qx][0] += gradX[qx][1] * wy;
|
||||
grad[qy][qx][1] += gradX[qx][0] * wDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate Dxy, xDy in plane
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const int q = QUAD_2D_ID(qx, qy);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
const double gradX = grad[qy][qx][0];
|
||||
const double gradY = grad[qy][qx][1];
|
||||
|
||||
grad[qy][qx][0] = (O11 * gradX) + (O12 * gradY);
|
||||
grad[qy][qx][1] = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double gradX[NUM_DOFS_1D][2];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradX[dx][0] = 0;
|
||||
gradX[dx][1] = 0;
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double gX = grad[qy][qx][0];
|
||||
const double gY = grad[qy][qx][1];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = quadToDof(dx, qx);
|
||||
const double wDx = quadToDofD(dx, qx);
|
||||
gradX[dx][0] += gX * wDx;
|
||||
gradX[dx][1] += gY * wx;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = quadToDof(dy, qy);
|
||||
const double wDy = quadToDofD(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, dy, e) += ((gradX[dx][0] * wy) +
|
||||
(gradX[dx][1] * wDy));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
const double * restrict quadWeights,
|
||||
const Jacobian3D_t restrict J,
|
||||
COEFF_ARGS
|
||||
SymmOperator3D_t restrict oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_3D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3), (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3), (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DofToQuad_t restrict dofToQuadD,
|
||||
const QuadToDof_t restrict quadToDof,
|
||||
const QuadToDof_t restrict quadToDofD,
|
||||
const SymmOperator3D_t restrict oper,
|
||||
const DLocal3D_t restrict solIn,
|
||||
DLocal3D_t restrict solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double grad[NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D][4];
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qz][qy][qx][0] = 0;
|
||||
grad[qz][qy][qx][1] = 0;
|
||||
grad[qz][qy][qx][2] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
double gradXY[NUM_QUAD_1D][NUM_QUAD_1D][4];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradXY[qy][qx][0] = 0;
|
||||
gradXY[qy][qx][1] = 0;
|
||||
gradXY[qy][qx][2] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double gradX[NUM_QUAD_1D][2];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] = 0;
|
||||
gradX[qx][1] = 0;
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, dy, dz, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] += s * dofToQuad(qx, dx);
|
||||
gradX[qx][1] += s * dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
const double wDy = dofToQuadD(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double wx = gradX[qx][0];
|
||||
const double wDx = gradX[qx][1];
|
||||
gradXY[qy][qx][0] += wDx * wy;
|
||||
gradXY[qy][qx][1] += wx * wDy;
|
||||
gradXY[qy][qx][2] += wx * wy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
const double wz = dofToQuad(qz, dz);
|
||||
const double wDz = dofToQuadD(qz, dz);
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qz][qy][qx][0] += gradXY[qy][qx][0] * wz;
|
||||
grad[qz][qy][qx][1] += gradXY[qy][qx][1] * wz;
|
||||
grad[qz][qy][qx][2] += gradXY[qy][qx][2] * wDz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const int q = QUAD_3D_ID(qx, qy, qz);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
const double gradX = grad[qz][qy][qx][0];
|
||||
const double gradY = grad[qz][qy][qx][1];
|
||||
const double gradZ = grad[qz][qy][qx][2];
|
||||
|
||||
grad[qz][qy][qx][0] = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
|
||||
grad[qz][qy][qx][1] = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
|
||||
grad[qz][qy][qx][2] = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
double gradXY[NUM_DOFS_1D][NUM_DOFS_1D][4];
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradXY[dy][dx][0] = 0;
|
||||
gradXY[dy][dx][1] = 0;
|
||||
gradXY[dy][dx][2] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double gradX[NUM_DOFS_1D][4];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradX[dx][0] = 0;
|
||||
gradX[dx][1] = 0;
|
||||
gradX[dx][2] = 0;
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double gX = grad[qz][qy][qx][0];
|
||||
const double gY = grad[qz][qy][qx][1];
|
||||
const double gZ = grad[qz][qy][qx][2];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = quadToDof(dx, qx);
|
||||
const double wDx = quadToDofD(dx, qx);
|
||||
gradX[dx][0] += gX * wDx;
|
||||
gradX[dx][1] += gY * wx;
|
||||
gradX[dx][2] += gZ * wx;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = quadToDof(dy, qy);
|
||||
const double wDy = quadToDofD(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradXY[dy][dx][0] += gradX[dx][0] * wy;
|
||||
gradXY[dy][dx][1] += gradX[dx][1] * wDy;
|
||||
gradXY[dy][dx][2] += gradX[dx][2] * wy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = quadToDof(dz, qz);
|
||||
const double wDz = quadToDofD(dz, qz);
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, dy, dz, e) += ((gradXY[dy][dx][0] * wz) +
|
||||
(gradXY[dy][dx][1] * wz) +
|
||||
(gradXY[dy][dx][2] * wDz));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,433 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator1D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A1_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A1_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
oper(q, e) = quadWeights[q] * COEFF / J(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DofToQuad_t restrict dofToQuadD,
|
||||
const QuadToDof_t restrict quadToDof,
|
||||
const QuadToDof_t restrict quadToDofD,
|
||||
const SymmOperator1D_t restrict oper,
|
||||
const DLocal1D_t restrict solIn,
|
||||
DLocal1D_t restrict solOut) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M1_ELEMENT_BATCHES; @outer) {
|
||||
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
@exclusive double grad[NUM_QUAD_1D];
|
||||
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
for (int i = el; i < NUM_QUAD_DOFS_1D; i += M1_INNER_ELEMENT_BATCH) {
|
||||
s_dofToQuadD[i] = dofToQuadD[i];
|
||||
s_quadToDofD[i] = quadToDofD[i];
|
||||
}
|
||||
}
|
||||
|
||||
for (int b = 0; b < M1_OUTER_ELEMENT_BATCH; ++b) {
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
const int e = eOff + b*M1_INNER_ELEMENT_BATCH + el;
|
||||
if (e < numElements) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] += s * s_dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] *= oper(qx, e);
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
double s = 0;
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
s += grad[qx] * s_quadToDofD(dx, qx);
|
||||
}
|
||||
solOut(dx, e) += s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator2D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD_2D; q += A2_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J21*J21 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J21*J11 + J22*J12); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J12*J12); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DofToQuad_t restrict dofToQuadD,
|
||||
const QuadToDof_t restrict quadToDof,
|
||||
const QuadToDof_t restrict quadToDofD,
|
||||
const SymmOperator2D_t restrict oper,
|
||||
const DLocal2D_t restrict solIn,
|
||||
DLocal2D_t restrict solOut) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M2_ELEMENT_BATCH; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_xy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_xDy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_grad[2 * NUM_QUAD_2D] @dim(2, NUM_QUAD_1D, NUM_QUAD_1D);
|
||||
|
||||
@exclusive double r_x[NUM_MAX_1D];
|
||||
@exclusive double r_y[NUM_QUAD_1D];
|
||||
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
for (int id = x; id < NUM_QUAD_DOFS_1D; id += NUM_MAX_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
s_dofToQuadD[id] = dofToQuadD[id];
|
||||
s_quadToDof[id] = quadToDof[id];
|
||||
s_quadToDofD[id] = quadToDofD[id];
|
||||
}
|
||||
}
|
||||
|
||||
for (int e = eOff; e < (eOff + M2_ELEMENT_BATCH); ++e) {
|
||||
if (e < numElements) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
s_xy(dx, qy) = 0;
|
||||
s_xDy(dx, qy) = 0;
|
||||
}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
r_x[dy] = solIn(dx, dy, e);
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double xy = 0;
|
||||
double xDy = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
xy += r_x[dy] * s_dofToQuad(qy, dy);
|
||||
xDy += r_x[dy] * s_dofToQuadD(qy, dy);
|
||||
}
|
||||
s_xy(dx, qy) = xy;
|
||||
s_xDy(dx, qy) = xDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
if (qy < NUM_QUAD_1D) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
double gradX = 0, gradY = 0;
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradX += s_xy(dx, qy) * s_dofToQuadD(qx, dx);
|
||||
gradY += s_xDy(dx, qy) * s_dofToQuad(qx, dx);
|
||||
}
|
||||
|
||||
const int q = QUAD_2D_ID(qx, qy);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
s_grad(0, qx, qy) = (O11 * gradX) + (O12 * gradY);
|
||||
s_grad(1, qx, qy) = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx; @inner) {
|
||||
if (qx < NUM_QUAD_1D) {
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
s_xy(dy, qx) = 0;
|
||||
s_xDy(dy, qx) = 0;
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
r_x[qy] = s_grad(0, qx, qy);
|
||||
r_y[qy] = s_grad(1, qx, qy);
|
||||
}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double xy = 0;
|
||||
double xDy = 0;
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
xy += r_x[qy] * s_quadToDof(dy, qy);
|
||||
xDy += r_y[qy] * s_quadToDofD(dy, qy);
|
||||
}
|
||||
s_xy(dy, qx) = xy;
|
||||
s_xDy(dy, qx) = xDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double s = 0;
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
s += ((s_xy(dy, qx) * s_quadToDofD(dx, qx)) +
|
||||
(s_xDy(dy, qx) * s_quadToDof(dx, qx)));
|
||||
}
|
||||
solOut(dx, dy, e) += s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator3D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD_3D; q += A3_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3), (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3), (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DofToQuad_t restrict dofToQuadD,
|
||||
const QuadToDof_t restrict quadToDof,
|
||||
const QuadToDof_t restrict quadToDofD,
|
||||
const SymmOperator3D_t restrict oper,
|
||||
const DLocal3D_t restrict solIn,
|
||||
DLocal3D_t restrict solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_z[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
@shared double s_Dz[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
@shared double s_xyDz[NUM_QUAD_2D] @dim(NUM_QUAD_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store z axis as registers
|
||||
@exclusive double r_qz[NUM_QUAD_1D];
|
||||
@exclusive double r_qDz[NUM_QUAD_1D];
|
||||
@exclusive double r_dDxyz[NUM_DOFS_1D];
|
||||
@exclusive double r_dxDyz[NUM_DOFS_1D];
|
||||
@exclusive double r_dxyDz[NUM_DOFS_1D];
|
||||
|
||||
for (int y = 0; y < NUM_MAX_1D; ++y; @inner) {
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
const int id = (y * NUM_MAX_1D) + x;
|
||||
// Fetch Q <--> D maps
|
||||
if (id < NUM_QUAD_DOFS_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
s_dofToQuadD[id] = dofToQuadD[id];
|
||||
s_quadToDof[id] = quadToDof[id];
|
||||
s_quadToDofD[id] = quadToDofD[id];
|
||||
}
|
||||
// Initialize our Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] = 0;
|
||||
r_qDz[qz] = 0;
|
||||
}
|
||||
// Initialize our solution updates in the Z axis
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
r_dDxyz[dz] = 0;
|
||||
r_dxDyz[dz] = 0;
|
||||
r_dxyDz[dz] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double s = solIn(dx, dy, dz, e);
|
||||
// Calculate D -> Q in the Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] += s * s_dofToQuad(qz, dz);
|
||||
r_qDz[qz] += s * s_dofToQuadD(qz, dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// For each xy plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
// Fill xy plane at given z position
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
s_z(dx, dy) = r_qz[qz];
|
||||
s_Dz(dx, dy) = r_qDz[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
double Dxyz = 0;
|
||||
double xDyz = 0;
|
||||
double xyDz = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = s_dofToQuad(qy, dy);
|
||||
const double wDy = s_dofToQuadD(qy, dy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = s_dofToQuad(qx, dx);
|
||||
const double wDx = s_dofToQuadD(qx, dx);
|
||||
const double z = s_z(dx, dy);
|
||||
const double Dz = s_Dz(dx, dy);
|
||||
Dxyz += wDx * wy * z;
|
||||
xDyz += wx * wDy * z;
|
||||
xyDz += wx * wy * Dz;
|
||||
}
|
||||
}
|
||||
|
||||
const int q = QUAD_3D_ID(qx, qy, qz);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
const double qDxyz = (O11 * Dxyz) + (O12 * xDyz) + (O13 * xyDz);
|
||||
const double qxDyz = (O12 * Dxyz) + (O22 * xDyz) + (O23 * xyDz);
|
||||
const double qxyDz = (O13 * Dxyz) + (O23 * xDyz) + (O33 * xyDz);
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = s_quadToDof(dz, qz);
|
||||
const double wDz = s_quadToDofD(dz, qz);
|
||||
r_dDxyz[dz] += wz * qDxyz;
|
||||
r_dxDyz[dz] += wz * qxDyz;
|
||||
r_dxyDz[dz] += wDz * qxyDz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Iterate over xy planes to compute solution
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
// Place xy plane in shared memory
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
s_z(qx, qy) = r_dDxyz[dz];
|
||||
s_Dz(qx, qy) = r_dxDyz[dz];
|
||||
s_xyDz(qx, qy) = r_dxyDz[dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Finalize solution in xy plane
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
double solZ = 0;
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = s_quadToDof(dy, qy);
|
||||
const double wDy = s_quadToDofD(dy, qy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double wx = s_quadToDof(dx, qx);
|
||||
const double wDx = s_quadToDofD(dx, qx);
|
||||
const double Dxyz = s_z(qx, qy);
|
||||
const double xDyz = s_Dz(qx, qy);
|
||||
const double xyDz = s_xyDz(qx, qy);
|
||||
solZ += ((wDx * wy * Dxyz) +
|
||||
(wx * wDy * xDyz) +
|
||||
(wx * wy * xyDz));
|
||||
}
|
||||
}
|
||||
solOut(dx, dy, dz, e) += solZ;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,140 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "url_handler.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
bool Engine::fileOpenerRegistered = false;
|
||||
|
||||
void Engine::Init(const std::string &engine_spec)
|
||||
{
|
||||
//
|
||||
// Initialize inherited fields
|
||||
//
|
||||
memory_resources[0] = NULL;
|
||||
workers_weights[0]= 1.0;
|
||||
workers_mem_res[0] = 0;
|
||||
|
||||
//
|
||||
// Initialize the OCCA engine
|
||||
//
|
||||
::occa::properties props(engine_spec);
|
||||
device = new ::occa::device[1];
|
||||
device[0].setup(props);
|
||||
|
||||
okl_path = "mfem-occa://";
|
||||
// okl_defines = "...";
|
||||
if (!fileOpenerRegistered)
|
||||
{
|
||||
// The directories from "MFEM_OCCA_OKL_PATH", if any, have the highest
|
||||
// priority.
|
||||
FileOpener *fo = new FileOpener("mfem-occa://", "MFEM_OCCA_OKL_PATH");
|
||||
// Next in priority is the source path, if it exists.
|
||||
std::string mfem_src_prefix = mfem::GetSourcePath();
|
||||
fo->AddDir(mfem_src_prefix + "/backends/occa");
|
||||
// And last in priority is the install path, if it exists.
|
||||
std::string mfem_install_prefix = mfem::GetInstallPath();
|
||||
fo->AddDir(mfem_install_prefix + "/lib/mfem/occa");
|
||||
::occa::io::fileOpener::add(fo);
|
||||
fileOpenerRegistered = true;
|
||||
}
|
||||
}
|
||||
|
||||
Engine::Engine(const std::string &engine_spec)
|
||||
: mfem::Engine(NULL, 1, 1)
|
||||
{
|
||||
Init(engine_spec);
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
Engine::Engine(MPI_Comm _comm, const std::string &engine_spec)
|
||||
: mfem::Engine(NULL, 1, 1)
|
||||
{
|
||||
comm = _comm;
|
||||
Init(engine_spec);
|
||||
}
|
||||
#endif
|
||||
|
||||
DLayout Engine::MakeLayout(std::size_t size) const
|
||||
{
|
||||
return DLayout(new Layout(*this, size));
|
||||
}
|
||||
|
||||
DLayout Engine::MakeLayout(const mfem::Array<std::size_t> &offsets) const
|
||||
{
|
||||
MFEM_ASSERT(offsets.Size() == 2,
|
||||
"multiple workers are not supported yet");
|
||||
return DLayout(new Layout(*this, offsets.Last()));
|
||||
}
|
||||
|
||||
DArray Engine::MakeArray(PLayout &layout, std::size_t item_size) const
|
||||
{
|
||||
MFEM_ASSERT(dynamic_cast<Layout *>(&layout) != NULL,
|
||||
"invalid input layout");
|
||||
Layout *lt = static_cast<Layout *>(&layout);
|
||||
return DArray(new Array(*lt, item_size));
|
||||
}
|
||||
|
||||
DVector Engine::MakeVector(PLayout &layout, int type_id) const
|
||||
{
|
||||
MFEM_ASSERT(type_id == ScalarId<double>::value, "invalid type_id");
|
||||
MFEM_ASSERT(dynamic_cast<Layout *>(&layout) != NULL,
|
||||
"invalid input layout");
|
||||
Layout *lt = static_cast<Layout *>(&layout);
|
||||
return DVector(new Vector(*lt));
|
||||
}
|
||||
|
||||
DFiniteElementSpace Engine::MakeFESpace(mfem::FiniteElementSpace &fespace) const
|
||||
{
|
||||
return DFiniteElementSpace(new FiniteElementSpace(*this, fespace));
|
||||
}
|
||||
|
||||
DBilinearForm Engine::MakeBilinearForm(mfem::BilinearForm &bf) const
|
||||
{
|
||||
return DBilinearForm(new BilinearForm(*this, bf));
|
||||
}
|
||||
|
||||
void Engine::AssembleLinearForm(LinearForm &l_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
}
|
||||
|
||||
mfem::Operator *Engine::MakeOperator(const MixedBilinearForm &mbl_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
mfem::Operator *Engine::MakeOperator(const NonlinearForm &nl_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,111 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_ENGINE_HPP
|
||||
#define MFEM_BACKENDS_OCCA_ENGINE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "../base/backend.hpp"
|
||||
#include <occa.hpp>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Engine : public mfem::Engine
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// mfem::Backend *backend;
|
||||
#ifdef MFEM_USE_MPI
|
||||
// MPI_Comm comm;
|
||||
#endif
|
||||
// int num_mem_res;
|
||||
// int num_workers;
|
||||
// MemoryResource **memory_resources;
|
||||
// double *workers_weights;
|
||||
// int *workers_mem_res;
|
||||
|
||||
static bool fileOpenerRegistered;
|
||||
::occa::device *device; // An array of OCCA devices
|
||||
std::string okl_path, okl_defines;
|
||||
|
||||
void Init(const std::string &engine_spec);
|
||||
|
||||
public:
|
||||
Engine(const std::string &engine_spec);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
Engine(MPI_Comm comm, const std::string &engine_spec);
|
||||
#endif
|
||||
|
||||
virtual ~Engine() { delete [] device; }
|
||||
|
||||
/**
|
||||
@name OCCA specific interface, used by other objects in the OCCA backend
|
||||
*/
|
||||
///@{
|
||||
|
||||
::occa::device GetDevice(int idx = 0) const { return device[idx]; }
|
||||
|
||||
/// TODO: doxygen
|
||||
const std::string &GetOklPath() const { return okl_path; }
|
||||
|
||||
/// TODO: doxygen
|
||||
const std::string &GetOklDefines() const { return okl_defines; }
|
||||
|
||||
///@}
|
||||
// End: OCCA specific interface
|
||||
|
||||
/**
|
||||
@name Virtual interface: finite element data structures and algorithms
|
||||
*/
|
||||
///@{
|
||||
|
||||
virtual DLayout MakeLayout(std::size_t size) const;
|
||||
virtual DLayout MakeLayout(const mfem::Array<std::size_t> &offsets) const;
|
||||
|
||||
virtual DArray MakeArray(PLayout &layout, std::size_t item_size) const;
|
||||
|
||||
virtual DVector MakeVector(PLayout &layout,
|
||||
int type_id = ScalarId<double>::value) const;
|
||||
|
||||
virtual DFiniteElementSpace MakeFESpace(mfem::FiniteElementSpace &
|
||||
fespace) const;
|
||||
|
||||
virtual DBilinearForm MakeBilinearForm(mfem::BilinearForm &bf) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual void AssembleLinearForm(LinearForm &l_form) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual mfem::Operator *MakeOperator(const MixedBilinearForm &mbl_form) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual mfem::Operator *MakeOperator(const NonlinearForm &nl_form) const;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_ENGINE_HPP
|
||||
@@ -1,174 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "interpolation.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
FiniteElementSpace::FiniteElementSpace(const Engine &e,
|
||||
mfem::FiniteElementSpace &fespace)
|
||||
: PFiniteElementSpace(e, fespace),
|
||||
e_layout(e, 0) // resized in SetupLocalGlobalMaps()
|
||||
{
|
||||
vdim = fespace.GetVDim();
|
||||
ordering = fespace.GetOrdering();
|
||||
|
||||
SetupLocalGlobalMaps();
|
||||
SetupOperators();
|
||||
SetupKernels();
|
||||
}
|
||||
|
||||
FiniteElementSpace::~FiniteElementSpace()
|
||||
{
|
||||
delete [] elementDofMap;
|
||||
delete [] elementDofMapInverse;
|
||||
delete restrictionOp;
|
||||
delete prolongationOp;
|
||||
}
|
||||
|
||||
void FiniteElementSpace::SetupLocalGlobalMaps()
|
||||
{
|
||||
const mfem::FiniteElement &fe = *(fes->GetFE(0));
|
||||
const mfem::TensorBasisElement *el =
|
||||
dynamic_cast<const mfem::TensorBasisElement*>(&fe);
|
||||
|
||||
const mfem::Table &e2dTable = fes->GetElementToDofTable();
|
||||
const int *elementMap = e2dTable.GetJ();
|
||||
const int elements = fes->GetNE();
|
||||
|
||||
globalDofs = fes->GetNDofs();
|
||||
localDofs = fe.GetDof();
|
||||
|
||||
e_layout.Resize(localDofs * elements * fes->GetVDim());
|
||||
|
||||
elementDofMap = new int[localDofs];
|
||||
elementDofMapInverse = new int[localDofs];
|
||||
if (el)
|
||||
{
|
||||
::memcpy(elementDofMap,
|
||||
el->GetDofMap().GetData(),
|
||||
localDofs * sizeof(int));
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i < localDofs; ++i)
|
||||
{
|
||||
elementDofMap[i] = i;
|
||||
}
|
||||
}
|
||||
for (int i = 0; i < localDofs; ++i)
|
||||
{
|
||||
elementDofMapInverse[elementDofMap[i]] = i;
|
||||
}
|
||||
|
||||
// Allocate device offsets and indices
|
||||
globalToLocalOffsets.allocate(GetDevice(),
|
||||
globalDofs + 1);
|
||||
globalToLocalIndices.allocate(GetDevice(),
|
||||
localDofs, elements);
|
||||
localToGlobalMap.allocate(GetDevice(),
|
||||
localDofs, elements);
|
||||
|
||||
int *offsets = globalToLocalOffsets.ptr();
|
||||
int *indices = globalToLocalIndices.ptr();
|
||||
int *l2gMap = localToGlobalMap.ptr();
|
||||
|
||||
// We'll be keeping a count of how many local nodes point
|
||||
// to its global dof
|
||||
for (int i = 0; i <= globalDofs; ++i)
|
||||
{
|
||||
offsets[i] = 0;
|
||||
}
|
||||
|
||||
for (int e = 0; e < elements; ++e)
|
||||
{
|
||||
for (int d = 0; d < localDofs; ++d)
|
||||
{
|
||||
const int gid = elementMap[localDofs*e + d];
|
||||
++offsets[gid + 1];
|
||||
}
|
||||
}
|
||||
// Aggregate to find offsets for each global dof
|
||||
for (int i = 1; i <= globalDofs; ++i)
|
||||
{
|
||||
offsets[i] += offsets[i - 1];
|
||||
}
|
||||
// For each global dof, fill in all local nodes that point
|
||||
// to it
|
||||
for (int e = 0; e < elements; ++e)
|
||||
{
|
||||
for (int d = 0; d < localDofs; ++d)
|
||||
{
|
||||
const int gid = elementMap[localDofs*e + elementDofMap[d]];
|
||||
const int lid = localDofs*e + d;
|
||||
indices[offsets[gid]++] = lid;
|
||||
l2gMap[lid] = gid;
|
||||
}
|
||||
}
|
||||
// We shifted the offsets vector by 1 by using it
|
||||
// as a counter. Now we shift it back.
|
||||
for (int i = globalDofs; i > 0; --i)
|
||||
{
|
||||
offsets[i] = offsets[i - 1];
|
||||
}
|
||||
offsets[0] = 0;
|
||||
|
||||
globalToLocalOffsets.keepInDevice();
|
||||
globalToLocalIndices.keepInDevice();
|
||||
localToGlobalMap.keepInDevice();
|
||||
}
|
||||
|
||||
void FiniteElementSpace::SetupOperators()
|
||||
{
|
||||
const mfem::SparseMatrix *R = fes->GetRestrictionMatrix();
|
||||
const mfem::Operator *P = fes->GetProlongationMatrix();
|
||||
CreateRPOperators(OccaVLayout(), OccaTrueVLayout(),
|
||||
R, P,
|
||||
restrictionOp,
|
||||
prolongationOp);
|
||||
}
|
||||
|
||||
void FiniteElementSpace::SetupKernels()
|
||||
{
|
||||
::occa::properties props("defines: {"
|
||||
" TILESIZE: 256,"
|
||||
"}");
|
||||
props["defines/NUM_VDIM"] = vdim;
|
||||
|
||||
props["defines/ORDERING_BY_NODES"] = 0;
|
||||
props["defines/ORDERING_BY_VDIM"] = 1;
|
||||
props["defines/VDIM_ORDERING"] = (int) (ordering == Ordering::byVDIM);
|
||||
|
||||
::occa::device device = GetDevice();
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
const std::string &okl_defines = OccaEngine().GetOklDefines();
|
||||
globalToLocalKernel = device.buildKernel(okl_path + "fespace.okl",
|
||||
"GlobalToLocal",
|
||||
props + okl_defines);
|
||||
localToGlobalKernel = device.buildKernel(okl_path + "fespace.okl",
|
||||
"LocalToGlobal",
|
||||
props + okl_defines);
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,146 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_FE_SPACE_HPP
|
||||
#define MFEM_BACKENDS_OCCA_FE_SPACE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "operator.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
/// TODO: doxygen
|
||||
class FiniteElementSpace : public mfem::PFiniteElementSpace
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// mfem::FiniteElementSpace *fes;
|
||||
|
||||
Layout e_layout;
|
||||
|
||||
int *elementDofMap;
|
||||
int *elementDofMapInverse;
|
||||
|
||||
::occa::array<int> globalToLocalOffsets;
|
||||
::occa::array<int> globalToLocalIndices;
|
||||
::occa::array<int> localToGlobalMap;
|
||||
::occa::kernel globalToLocalKernel, localToGlobalKernel;
|
||||
|
||||
mfem::Ordering::Type ordering;
|
||||
|
||||
int globalDofs, localDofs;
|
||||
int vdim;
|
||||
|
||||
mfem::Operator *restrictionOp, *prolongationOp;
|
||||
|
||||
void SetupLocalGlobalMaps();
|
||||
void SetupOperators();
|
||||
void SetupKernels();
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
FiniteElementSpace(const Engine &e, mfem::FiniteElementSpace &fespace);
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~FiniteElementSpace();
|
||||
|
||||
/// TODO: doxygen
|
||||
const Engine &OccaEngine() const
|
||||
{ return *static_cast<const Engine *>(engine.Get()); }
|
||||
|
||||
/// TODO: doxygen
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return OccaEngine().GetDevice(idx); }
|
||||
|
||||
mfem::Mesh* GetMesh() const { return fes->GetMesh(); }
|
||||
|
||||
Layout &OccaVLayout() const
|
||||
{ return *fes->GetVLayout().As<Layout>(); }
|
||||
|
||||
Layout &OccaTrueVLayout() const
|
||||
{ return *fes->GetTrueVLayout().As<Layout>(); }
|
||||
|
||||
Layout &OccaEVLayout() { return e_layout; }
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
bool isDistributed() const { return (OccaEngine().GetComm() != MPI_COMM_NULL); }
|
||||
#else
|
||||
bool isDistributed() const { return false; }
|
||||
#endif
|
||||
|
||||
bool hasTensorBasis() const
|
||||
{ return dynamic_cast<const mfem::TensorBasisElement*>(fes->GetFE(0)); }
|
||||
|
||||
mfem::Ordering::Type GetOrdering() const { return ordering; }
|
||||
|
||||
int GetGlobalDofs() const { return globalDofs; }
|
||||
int GetLocalDofs() const { return localDofs; }
|
||||
|
||||
int GetDim() const { return fes->GetMesh()->Dimension(); }
|
||||
int GetVDim() const { return vdim; }
|
||||
|
||||
int GetVSize() const { return globalDofs * vdim; }
|
||||
int GetTrueVSize() const { return fes->GetTrueVSize(); }
|
||||
int GetGlobalVSize() const { return globalDofs*vdim; /* FIXME: MPI */ }
|
||||
int GetGlobalTrueVSize() const { return fes->GetTrueVSize(); }
|
||||
|
||||
int GetNE() const { return fes->GetNE(); }
|
||||
|
||||
const mfem::FiniteElementCollection* FEColl() const
|
||||
{ return fes->FEColl(); }
|
||||
const mfem::FiniteElement* GetFE(const int idx) const
|
||||
{ return fes->GetFE(idx); }
|
||||
|
||||
const int* GetElementDofMap() const { return elementDofMap; }
|
||||
const int* GetElementDofMapInverse() const { return elementDofMapInverse; }
|
||||
|
||||
const mfem::Operator* GetRestrictionOperator() { return restrictionOp; }
|
||||
const mfem::Operator* GetProlongationOperator() { return prolongationOp; }
|
||||
|
||||
const ::occa::array<int> GetLocalToGlobalMap() const
|
||||
{ return localToGlobalMap; }
|
||||
|
||||
void GlobalToLocal(const Vector &globalVec, Vector &localVec) const
|
||||
{
|
||||
globalToLocalKernel(globalDofs,
|
||||
localDofs * fes->GetNE(),
|
||||
globalToLocalOffsets,
|
||||
globalToLocalIndices,
|
||||
globalVec.OccaMem(), localVec.OccaMem());
|
||||
}
|
||||
void LocalToGlobal(const Vector &localVec, Vector &globalVec) const
|
||||
{
|
||||
localToGlobalKernel(globalDofs,
|
||||
localDofs * fes->GetNE(),
|
||||
globalToLocalOffsets,
|
||||
globalToLocalIndices,
|
||||
localVec.OccaMem(), globalVec.OccaMem());
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_FE_SPACE_HPP
|
||||
@@ -1,67 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
TILESIZE : Tilesize for iterating over entries
|
||||
================================================
|
||||
*/
|
||||
|
||||
#if VDIM_ORDERING == ORDERING_BY_VDIM
|
||||
typedef double *Global_t @dim(NUM_VDIM, globalEntries);
|
||||
typedef double *Local_t @dim(NUM_VDIM, localEntries);
|
||||
#else
|
||||
typedef double *Global_t @dim(NUM_VDIM, globalEntries) @dimOrder(1, 0);
|
||||
typedef double *Local_t @dim(NUM_VDIM, localEntries) @dimOrder(1, 0);
|
||||
#endif
|
||||
|
||||
@kernel void GlobalToLocal(const int globalEntries,
|
||||
const int localEntries,
|
||||
const int * restrict offsets,
|
||||
const int * restrict indices,
|
||||
const Global_t restrict globalX,
|
||||
Local_t restrict localX) {
|
||||
|
||||
for (int i = 0; i < globalEntries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < globalEntries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double dofValue = globalX(v, i);
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
localX(v, indices[j]) = dofValue;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void LocalToGlobal(const int globalEntries,
|
||||
const int localEntries,
|
||||
const int * restrict offsets,
|
||||
const int * restrict indices,
|
||||
const Local_t restrict localX,
|
||||
Global_t restrict globalX) {
|
||||
|
||||
for (int i = 0; i < globalEntries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < globalEntries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
double dofValue = 0;
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
dofValue += localX(v, indices[j]);
|
||||
}
|
||||
globalX(v, i) = dofValue;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,181 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef STORE_JACOBIAN
|
||||
# define STORE_JACOBIAN 1
|
||||
#endif
|
||||
|
||||
#ifndef STORE_JACOBIAN_INV
|
||||
# define STORE_JACOBIAN_INV 1
|
||||
#endif
|
||||
|
||||
#ifndef STORE_JACOBIAN_DET
|
||||
# define STORE_JACOBIAN_DET 1
|
||||
#endif
|
||||
|
||||
typedef double* Local1D_t @dim(1, NUM_DOFS, numElements);
|
||||
typedef double* Local2D_t @dim(2, NUM_DOFS, numElements);
|
||||
typedef double* Local3D_t @dim(3, NUM_DOFS, numElements);
|
||||
|
||||
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
|
||||
|
||||
typedef double* DofToQuadD1D_t @dim(NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD2D_t @dim(2, NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD3D_t @dim(3, NUM_QUAD, NUM_DOFS);
|
||||
|
||||
typedef double* Jacobian1D_t @dim(NUM_QUAD, numElements);
|
||||
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD, numElements);
|
||||
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD, numElements);
|
||||
|
||||
@kernel void InitGeometryInfo1D(const int numElements,
|
||||
const DofToQuadD1D_t restrict dofToQuadD,
|
||||
const Local1D_t restrict nodes,
|
||||
Jacobian1D_t restrict J,
|
||||
Jacobian1D_t restrict invJ,
|
||||
QLocal_t restrict detJ) {
|
||||
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_nodes[NUM_DOFS];
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
|
||||
s_nodes[d] = nodes(0, d, e);
|
||||
}
|
||||
}
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
double J11 = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double wx = dofToQuadD(q, d);
|
||||
J11 += wx * s_nodes[d];
|
||||
}
|
||||
#if STORE_JACOBIAN
|
||||
J(q, e) = J11;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_INV
|
||||
invJ(q, e) = 1.0 / J11;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_DET
|
||||
detJ(q, e) = J11;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void InitGeometryInfo2D(const int numElements,
|
||||
const DofToQuadD2D_t restrict dofToQuadD,
|
||||
const Local2D_t restrict nodes,
|
||||
Jacobian2D_t restrict J,
|
||||
Jacobian2D_t restrict invJ,
|
||||
QLocal_t restrict detJ) {
|
||||
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_nodes[2 * NUM_DOFS] @dim(2, NUM_DOFS);
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
|
||||
s_nodes(0, d) = nodes(0, d, e);
|
||||
s_nodes(1, d) = nodes(1, d, e);
|
||||
}
|
||||
}
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
double J11 = 0, J12 = 0;
|
||||
double J21 = 0, J22 = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double wx = dofToQuadD(0, q, d);
|
||||
const double wy = dofToQuadD(1, q, d);
|
||||
const double x = s_nodes(0, d);
|
||||
const double y = s_nodes(1, d);
|
||||
J11 += (wx * x); J12 += (wx * y);
|
||||
J21 += (wy * x); J22 += (wy * y);
|
||||
}
|
||||
#if STORE_JACOBIAN_INV || STORE_JACOBIAN_DET
|
||||
const double r_detJ = (J11 * J22) - (J12 * J21);
|
||||
#endif
|
||||
#if STORE_JACOBIAN
|
||||
J(0, 0, q, e) = J11; J(1, 0, q, e) = J12;
|
||||
J(0, 1, q, e) = J21; J(1, 1, q, e) = J22;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_INV
|
||||
const double r_idetJ = 1.0 / r_detJ;
|
||||
invJ(0, 0, q, e) = J22 * r_idetJ;
|
||||
invJ(1, 0, q, e) = -J12 * r_idetJ;
|
||||
|
||||
invJ(0, 1, q, e) = -J21 * r_idetJ;
|
||||
invJ(1, 1, q, e) = J11 * r_idetJ;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_DET
|
||||
detJ(q, e) = r_detJ;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void InitGeometryInfo3D(const int numElements,
|
||||
const DofToQuadD3D_t restrict dofToQuadD,
|
||||
const Local3D_t restrict nodes,
|
||||
Jacobian3D_t restrict J,
|
||||
Jacobian3D_t restrict invJ,
|
||||
QLocal_t restrict detJ) {
|
||||
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_nodes[3 * NUM_DOFS] @dim(3, NUM_DOFS);
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
|
||||
s_nodes(0, d) = nodes(0, d, e);
|
||||
s_nodes(1, d) = nodes(1, d, e);
|
||||
s_nodes(2, d) = nodes(2, d, e);
|
||||
}
|
||||
}
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
double J11 = 0, J12 = 0, J13 = 0;
|
||||
double J21 = 0, J22 = 0, J23 = 0;
|
||||
double J31 = 0, J32 = 0, J33 = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double wx = dofToQuadD(0, q, d);
|
||||
const double wy = dofToQuadD(1, q, d);
|
||||
const double wz = dofToQuadD(2, q, d);
|
||||
const double x = s_nodes(0, d);
|
||||
const double y = s_nodes(1, d);
|
||||
const double z = s_nodes(2, d);
|
||||
J11 += (wx * x); J12 += (wx * y); J13 += (wx * z);
|
||||
J21 += (wy * x); J22 += (wy * y); J23 += (wy * z);
|
||||
J31 += (wz * x); J32 += (wz * y); J33 += (wz * z);
|
||||
}
|
||||
#if STORE_JACOBIAN_INV || STORE_JACOBIAN_DET
|
||||
const double r_detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
#endif
|
||||
#if STORE_JACOBIAN
|
||||
J(0, 0, q, e) = J11; J(1, 0, q, e) = J12; J(2, 0, q, e) = J13;
|
||||
J(0, 1, q, e) = J21; J(1, 1, q, e) = J22; J(2, 1, q, e) = J23;
|
||||
J(0, 2, q, e) = J31; J(1, 2, q, e) = J32; J(2, 2, q, e) = J33;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_INV
|
||||
const double r_idetJ = 1.0 / r_detJ;
|
||||
invJ(0, 0, q, e) = r_idetJ * ((J22 * J33) - (J23 * J32));
|
||||
invJ(1, 0, q, e) = r_idetJ * ((J32 * J13) - (J33 * J12));
|
||||
invJ(2, 0, q, e) = r_idetJ * ((J12 * J23) - (J13 * J22));
|
||||
|
||||
invJ(0, 1, q, e) = r_idetJ * ((J23 * J31) - (J21 * J33));
|
||||
invJ(1, 1, q, e) = r_idetJ * ((J33 * J11) - (J31 * J13));
|
||||
invJ(2, 1, q, e) = r_idetJ * ((J13 * J21) - (J11 * J23));
|
||||
|
||||
invJ(0, 2, q, e) = r_idetJ * ((J21 * J32) - (J22 * J31));
|
||||
invJ(1, 2, q, e) = r_idetJ * ((J31 * J12) - (J32 * J11));
|
||||
invJ(2, 2, q, e) = r_idetJ * ((J11 * J22) - (J12 * J21));
|
||||
#endif
|
||||
#if STORE_JACOBIAN_DET
|
||||
detJ(q, e) = r_detJ;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,195 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "gridfunc.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "../../fem/gridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
std::map<std::string, ::occa::kernel> gridFunctionKernels;
|
||||
|
||||
::occa::kernel GetGridFunctionKernel(::occa::device device,
|
||||
FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir)
|
||||
{
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
const FiniteElement &fe = *(fespace.GetFE(0));
|
||||
const int dim = fe.GetDim();
|
||||
const int vdim = fespace.GetVDim();
|
||||
|
||||
std::stringstream ss;
|
||||
ss << ::occa::hash(device)
|
||||
<< "FEColl : " << fespace.FEColl()->Name()
|
||||
<< "Quad: " << numQuad
|
||||
<< "Dim: " << dim
|
||||
<< "VDim: " << vdim;
|
||||
std::string hash = ss.str();
|
||||
|
||||
// Kernel defines
|
||||
::occa::properties props;
|
||||
props["defines/NUM_VDIM"] = vdim;
|
||||
|
||||
SetProperties(fespace, ir, props);
|
||||
|
||||
::occa::kernel kernel = gridFunctionKernels[hash];
|
||||
if (!kernel.isInitialized())
|
||||
{
|
||||
const std::string &okl_path = fespace.OccaEngine().GetOklPath();
|
||||
kernel = device.buildKernel(okl_path + "gridfunc.okl",
|
||||
stringWithDim("GridFuncToQuad", dim),
|
||||
props);
|
||||
}
|
||||
return kernel;
|
||||
}
|
||||
|
||||
// OccaGridFunction::OccaGridFunction() :
|
||||
// Vector(),
|
||||
// ofespace(NULL),
|
||||
// sequence(0) {}
|
||||
|
||||
OccaGridFunction::OccaGridFunction(FiniteElementSpace *ofespace_)
|
||||
: PArray(ofespace_->OccaVLayout()),
|
||||
Array(ofespace_->OccaVLayout(), sizeof(double)),
|
||||
Vector(ofespace_->OccaVLayout()),
|
||||
ofespace(ofespace_),
|
||||
sequence(0) {}
|
||||
|
||||
// OccaGridFunction::OccaGridFunction(OccaFiniteElementSpace *ofespace_,
|
||||
// OccaVectorRef ref) :
|
||||
// OccaVector(ref),
|
||||
// ofespace(ofespace_),
|
||||
// sequence(0) {}
|
||||
|
||||
OccaGridFunction::OccaGridFunction(const OccaGridFunction &v)
|
||||
: PArray(v),
|
||||
Array(v),
|
||||
Vector(v),
|
||||
ofespace(v.ofespace),
|
||||
sequence(v.sequence) {}
|
||||
|
||||
OccaGridFunction& OccaGridFunction::operator = (double value)
|
||||
{
|
||||
Fill(value);
|
||||
return *this;
|
||||
}
|
||||
|
||||
OccaGridFunction& OccaGridFunction::operator = (const Vector &v)
|
||||
{
|
||||
Assign<double>(v);
|
||||
return *this;
|
||||
}
|
||||
|
||||
// OccaGridFunction& OccaGridFunction::operator = (const OccaVectorRef &v)
|
||||
// {
|
||||
// OccaVector::operator = (v);
|
||||
// return *this;
|
||||
// }
|
||||
|
||||
OccaGridFunction& OccaGridFunction::operator = (const OccaGridFunction &v)
|
||||
{
|
||||
Assign<double>(v);
|
||||
return *this;
|
||||
}
|
||||
|
||||
// void OccaGridFunction::SetGridFunction(mfem::GridFunction &gf)
|
||||
// {
|
||||
// Vector v = *this;
|
||||
// gf.MakeRef(ofespace->GetFESpace(), v, 0);
|
||||
// // Make gf the owner of the data
|
||||
// v.Swap(gf);
|
||||
// }
|
||||
|
||||
void OccaGridFunction::GetTrueDofs(Vector &v)
|
||||
{
|
||||
const mfem::Operator *R = ofespace->GetRestrictionOperator();
|
||||
if (!R)
|
||||
{
|
||||
v.MakeRef(*this);
|
||||
}
|
||||
else
|
||||
{
|
||||
v.Resize<double>(R->OutLayout(), NULL);
|
||||
mfem::Vector mfem_v(v);
|
||||
R->Mult(this->Wrap(), mfem_v);
|
||||
}
|
||||
}
|
||||
|
||||
void OccaGridFunction::SetFromTrueDofs(Vector &v)
|
||||
{
|
||||
const mfem::Operator *P = ofespace->GetProlongationOperator();
|
||||
if (!P)
|
||||
{
|
||||
MakeRef(v);
|
||||
}
|
||||
else
|
||||
{
|
||||
Resize<double>(P->OutLayout(), NULL);
|
||||
mfem::Vector mfem_this(*this);
|
||||
P->Mult(v.Wrap(), mfem_this);
|
||||
}
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace* OccaGridFunction::GetFESpace()
|
||||
{
|
||||
return ofespace->GetFESpace();
|
||||
}
|
||||
|
||||
const mfem::FiniteElementSpace* OccaGridFunction::GetFESpace() const
|
||||
{
|
||||
return ofespace->GetFESpace();
|
||||
}
|
||||
|
||||
void OccaGridFunction::ToQuad(const IntegrationRule &ir, Vector &quadValues)
|
||||
{
|
||||
const Engine &engine = OccaLayout().OccaEngine();
|
||||
::occa::device device = engine.GetDevice();
|
||||
|
||||
OccaDofQuadMaps &maps = OccaDofQuadMaps::Get(device, *ofespace, ir);
|
||||
|
||||
const int elements = ofespace->GetNE();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
quadValues.Resize<double>(*(new Layout(engine, numQuad * elements)), NULL);
|
||||
|
||||
::occa::kernel g2qKernel = GetGridFunctionKernel(device, *ofespace, ir);
|
||||
g2qKernel(elements,
|
||||
maps.dofToQuad,
|
||||
ofespace->GetLocalToGlobalMap(),
|
||||
this->OccaMem(),
|
||||
quadValues.OccaMem());
|
||||
}
|
||||
|
||||
void OccaGridFunction::Distribute(const Vector &v)
|
||||
{
|
||||
if (ofespace->isDistributed())
|
||||
{
|
||||
mfem::Vector mfem_this(*this);
|
||||
ofespace->GetProlongationOperator()->Mult(v.Wrap(), mfem_this);
|
||||
}
|
||||
else
|
||||
{
|
||||
*this = v;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,83 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
|
||||
#define MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "fespace.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class IntegrationRule;
|
||||
class GridFunction;
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class OccaIntegrator;
|
||||
class OccaDofQuadMaps;
|
||||
|
||||
// TODO: make this object part of the backend or the engine.
|
||||
extern std::map<std::string, ::occa::kernel> gridFunctionKernels;
|
||||
|
||||
// TODO: make this a method of the backend or the engine.
|
||||
::occa::kernel GetGridFunctionKernel(::occa::device device,
|
||||
FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir);
|
||||
|
||||
class OccaGridFunction : public Vector
|
||||
{
|
||||
protected:
|
||||
FiniteElementSpace *ofespace;
|
||||
long sequence;
|
||||
|
||||
::occa::kernel gridFuncToQuad[3];
|
||||
|
||||
public:
|
||||
// OccaGridFunction();
|
||||
|
||||
OccaGridFunction(FiniteElementSpace *ofespace_);
|
||||
|
||||
// OccaGridFunction(FiniteElementSpace *ofespace_,
|
||||
// OccaVectorRef ref);
|
||||
|
||||
OccaGridFunction(const OccaGridFunction &gf);
|
||||
|
||||
OccaGridFunction& operator = (double value);
|
||||
OccaGridFunction& operator = (const Vector &v);
|
||||
// OccaGridFunction& operator = (const OccaVectorRef &v);
|
||||
OccaGridFunction& operator = (const OccaGridFunction &gf);
|
||||
|
||||
// void SetGridFunction(mfem::GridFunction &gf);
|
||||
|
||||
void GetTrueDofs(Vector &v);
|
||||
void SetFromTrueDofs(Vector &v);
|
||||
|
||||
mfem::FiniteElementSpace* GetFESpace();
|
||||
const mfem::FiniteElementSpace* GetFESpace() const;
|
||||
|
||||
void ToQuad(const mfem::IntegrationRule &ir, Vector &quadValues);
|
||||
|
||||
void Distribute(const Vector &v);
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
|
||||
@@ -1,26 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# if OCCA_USING_CPU
|
||||
# include "mfem-occa://gridfunc/tensor/cpu.okl"
|
||||
# else
|
||||
# include "mfem-occa://gridfunc/tensor/gpuHighOrder.okl"
|
||||
# endif
|
||||
#else
|
||||
# if OCCA_USING_CPU
|
||||
# include "mfem-occa://gridfunc/simplex/cpu.okl"
|
||||
# else
|
||||
# include "mfem-occa://gridfunc/simplex/gpuHighOrder.okl"
|
||||
# endif
|
||||
#endif
|
||||
@@ -1,63 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DLocalMap_t restrict l2gMap,
|
||||
const double * restrict gf,
|
||||
QVLocal_t restrict out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
double r_out = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_out += r_gf * dofToQuad(d, q);
|
||||
}
|
||||
out(v, d, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DLocalMap_t restrict l2gMap,
|
||||
const double * restrict gf,
|
||||
QVLocal_t restrict out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
double r_out = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_out += r_gf * dofToQuad(d, q);
|
||||
}
|
||||
out(v, d, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,79 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DLocalMap_t restrict l2gMap,
|
||||
const double * restrict gf,
|
||||
QVLocal_t restrict out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gf[NUM_VDIM][NUM_DOFS];
|
||||
|
||||
for (int dOff = 0; dOff < M2_INNER_BATCH; ++dOff; @inner) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M2_INNER_BATCH) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
s_gf[v][d] = gf[v + gid*NUM_VDIM]];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qOff = 0; qOff < M2_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M2_INNER_BATCH) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
double r_out = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_out += s_gf[v][d] * dofToQuad(d, q);
|
||||
}
|
||||
out(v, q, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DLocalMap_t restrict l2gMap,
|
||||
const double * restrict gf,
|
||||
QVLocal_t restrict out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gf[NUM_VDIM][NUM_DOFS];
|
||||
|
||||
for (int dOff = 0; dOff < M3_INNER_BATCH; ++dOff; @inner) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M3_INNER_BATCH) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
s_gf[v][d] = gf[v + gid*NUM_VDIM]];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qOff = 0; qOff < M3_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M3_INNER_BATCH) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
double r_out = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_out += s_gf[v][d] * dofToQuad(d, q);
|
||||
}
|
||||
out(v, q, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,188 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void GridFuncToQuad1D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DLocalMap1D_t restrict l2gMap,
|
||||
const double * restrict gf,
|
||||
QVLocal1D_t restrict out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_out[NUM_VDIM][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[v][qx] = 0;
|
||||
}
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const int gid = l2gMap(dx, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[v][qx] += r_gf * dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
out(v, qx, e) = r_out[v][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DLocalMap2D_t restrict l2gMap,
|
||||
const double * restrict gf,
|
||||
QVLocal2D_t restrict out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double out_xy[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double out_x[NUM_VDIM][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
out_x[v][qy] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const int gid = l2gMap(dx, dy, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
out_x[v][qy] += r_gf * dofToQuad(qy, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double d2q = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] += d2q * out_x[v][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
out(v, qx, qy, e) = out_xy[v][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DLocalMap3D_t restrict l2gMap,
|
||||
const double * restrict gf,
|
||||
QVLocal3D_t restrict out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double out_xyz[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xyz[v][qz][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
double out_xy[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double out_x[NUM_VDIM][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_x[v][qx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const int gid = l2gMap(dx, dy, dz, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_x[v][qx] += r_gf * dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] += wy * out_x[v][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
const double wz = dofToQuad(qz, dz);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xyz[v][qz][qy][qx] += wz * out_xy[v][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
out(v, qx, qy, qz, e) = out_xyz[v][qz][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,183 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void GridFuncToQuad1D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DLocalMap1D_t restrict l2gMap,
|
||||
const double * restrict gf,
|
||||
QLocal1D_t restrict out) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M1_ELEMENT_BATCHES; @outer) {
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@exclusive double r_out[NUM_QUAD_1D];
|
||||
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
for (int i = el; i < NUM_QUAD_DOFS_1D; i += M1_INNER_ELEMENT_BATCH) {
|
||||
s_dofToQuad[i] = dofToQuad[i];
|
||||
}
|
||||
}
|
||||
|
||||
for (int b = 0; b < M1_OUTER_ELEMENT_BATCH; ++b) {
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
const int e = eOff + b*M1_INNER_ELEMENT_BATCH + el;
|
||||
if (e < numElements) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double r_gf = gf[l2gMap(dx, e)];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[qx] += r_gf * s_dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out(qx, e) = r_out[qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DLocalMap2D_t restrict l2gMap,
|
||||
const double * restrict gf,
|
||||
QLocal2D_t restrict out) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M2_ELEMENT_BATCH; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_xy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
for (int id = x; id < NUM_QUAD_DOFS_1D; id += NUM_MAX_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
}
|
||||
}
|
||||
|
||||
for (int e = eOff; e < (eOff + M2_ELEMENT_BATCH); ++e) {
|
||||
if (e < numElements) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
double r_x[NUM_DOFS_1D];
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
r_x[dy] = gf[l2gMap(dx, dy, e)];
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double xy = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
xy += r_x[dy] * s_dofToQuad(qy, dy);
|
||||
}
|
||||
s_xy(dx, qy) = xy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
if (qy < NUM_QUAD_1D) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
double val = 0;
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
val += s_xy(dx, qy) * s_dofToQuad(qx, dx);
|
||||
}
|
||||
out(qx, qy, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
const DofToQuad_t restrict dofToQuad,
|
||||
const DLocalMap3D_t restrict l2gMap,
|
||||
const double * restrict gf,
|
||||
QLocal3D_t restrict out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_z[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
|
||||
// Store z axis as registers
|
||||
@exclusive double r_qz[NUM_QUAD_1D];
|
||||
|
||||
for (int y = 0; y < NUM_MAX_1D; ++y; @inner) {
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
const int id = (y * NUM_MAX_1D) + x;
|
||||
// Fetch Q <--> D maps
|
||||
if (id < NUM_QUAD_DOFS_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
}
|
||||
// Initialize our Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double val = gf[l2gMap(dx, dy, dz, e)];
|
||||
// Calculate D -> Q in the Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] += val * s_dofToQuad(qz, dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// For each xy plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
// Fill xy plane at given z position
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
s_z(dx, dy) = r_qz[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
double val = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = s_dofToQuad(qy, dy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = s_dofToQuad(qx, dx);
|
||||
val += wx * wy * s_z(dx, dy);
|
||||
}
|
||||
}
|
||||
out(qx, qy, qz, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,162 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "interpolation.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
void CreateRPOperators(Layout &v_layout, Layout &t_layout,
|
||||
const mfem::SparseMatrix *R, const mfem::Operator *P,
|
||||
mfem::Operator *&OccaR, mfem::Operator *&OccaP)
|
||||
{
|
||||
if (!P)
|
||||
{
|
||||
OccaR = new IdentityOperator(t_layout);
|
||||
OccaP = new IdentityOperator(t_layout);
|
||||
return;
|
||||
}
|
||||
|
||||
const mfem::SparseMatrix *pmat = dynamic_cast<const mfem::SparseMatrix*>(P);
|
||||
::occa::device device = v_layout.OccaEngine().GetDevice();
|
||||
|
||||
if (R)
|
||||
{
|
||||
OccaSparseMatrix *occaR =
|
||||
CreateMappedSparseMatrix(v_layout, t_layout, *R);
|
||||
::occa::array<int> reorderIndices = occaR->reorderIndices;
|
||||
delete occaR;
|
||||
|
||||
OccaR = new RestrictionOperator(v_layout, t_layout, reorderIndices);
|
||||
}
|
||||
|
||||
if (pmat)
|
||||
{
|
||||
const mfem::SparseMatrix *pmatT = Transpose(*pmat);
|
||||
|
||||
OccaSparseMatrix *occaP =
|
||||
CreateMappedSparseMatrix(t_layout, v_layout, *pmat);
|
||||
OccaSparseMatrix *occaPT =
|
||||
CreateMappedSparseMatrix(v_layout, t_layout, *pmatT);
|
||||
|
||||
OccaP = new ProlongationOperator(*occaP, *occaPT);
|
||||
}
|
||||
else
|
||||
{
|
||||
OccaP = new ProlongationOperator(t_layout, v_layout, P);
|
||||
}
|
||||
}
|
||||
|
||||
RestrictionOperator::RestrictionOperator(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> indices) :
|
||||
Operator(in_layout, out_layout)
|
||||
{
|
||||
|
||||
entries = indices.size() / 2;
|
||||
trueIndices = indices;
|
||||
|
||||
// FIXME: paths ...
|
||||
::occa::device device = in_layout.OccaEngine().GetDevice();
|
||||
const std::string &okl_path = in_layout.OccaEngine().GetOklPath();
|
||||
const std::string &okl_defines = in_layout.OccaEngine().GetOklDefines();
|
||||
multOp = device.buildKernel(okl_path + "mappings.okl",
|
||||
"ExtractSubVector",
|
||||
"defines: { TILESIZE: 256 }" + okl_defines);
|
||||
|
||||
multTransposeOp = device.buildKernel(okl_path + "mappings.okl",
|
||||
"SetSubVector",
|
||||
"defines: { TILESIZE: 256 }" +
|
||||
okl_defines);
|
||||
}
|
||||
|
||||
void RestrictionOperator::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
multOp(entries, trueIndices, x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
|
||||
void RestrictionOperator::MultTranspose_(const Vector &x, Vector &y) const
|
||||
{
|
||||
y.Fill<double>(0.0);
|
||||
multTransposeOp(entries, trueIndices, x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
|
||||
ProlongationOperator::ProlongationOperator(OccaSparseMatrix &multOp_,
|
||||
OccaSparseMatrix &multTransposeOp_) :
|
||||
Operator(multOp_),
|
||||
pmat(NULL),
|
||||
multOp(multOp_),
|
||||
multTransposeOp(multTransposeOp_) {}
|
||||
|
||||
ProlongationOperator::ProlongationOperator(Layout &in_layout,
|
||||
Layout &out_layout,
|
||||
const mfem::Operator *pmat_) :
|
||||
Operator(in_layout, out_layout),
|
||||
pmat(pmat_),
|
||||
multOp(*this),
|
||||
multTransposeOp(*this)
|
||||
{ }
|
||||
|
||||
void ProlongationOperator::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
MFEM_VERIFY(pmat == NULL, "");
|
||||
multOp.Mult_(x, y);
|
||||
}
|
||||
|
||||
void ProlongationOperator::MultTranspose_(const Vector &x, Vector &y) const
|
||||
{
|
||||
MFEM_VERIFY(pmat == NULL, "");
|
||||
multTransposeOp.Mult_(x, y);
|
||||
}
|
||||
|
||||
void ProlongationOperator::Mult(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{
|
||||
if (pmat)
|
||||
{
|
||||
// FIXME: create an OCCA version of 'pmat'
|
||||
x.Pull();
|
||||
y.Pull(false);
|
||||
pmat->Mult(x, y);
|
||||
y.Push();
|
||||
}
|
||||
else
|
||||
{
|
||||
multOp.Mult(x, y);
|
||||
}
|
||||
}
|
||||
|
||||
void ProlongationOperator::MultTranspose(const mfem::Vector &x,
|
||||
mfem::Vector &y) const
|
||||
{
|
||||
if (pmat)
|
||||
{
|
||||
// FIXME: create an OCCA version of 'pmat'
|
||||
x.Pull();
|
||||
y.Pull(false);
|
||||
pmat->MultTranspose(x, y);
|
||||
y.Push();
|
||||
}
|
||||
else
|
||||
{
|
||||
multTransposeOp.Mult(x, y);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,79 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
|
||||
#define MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "vector.hpp"
|
||||
#include "engine.hpp"
|
||||
#include "sparsemat.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
// [MISSING] Proper destructors
|
||||
void CreateRPOperators(Layout &v_layout, Layout &t_layout,
|
||||
const mfem::SparseMatrix *R, const mfem::Operator *P,
|
||||
mfem::Operator *&OccaR, mfem::Operator *&OccaP);
|
||||
|
||||
class RestrictionOperator : public Operator
|
||||
{
|
||||
protected:
|
||||
int entries;
|
||||
::occa::array<int> trueIndices;
|
||||
::occa::kernel multOp, multTransposeOp;
|
||||
|
||||
public:
|
||||
RestrictionOperator(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> indices);
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
class ProlongationOperator : public Operator
|
||||
{
|
||||
protected:
|
||||
const mfem::Operator *pmat;
|
||||
OccaSparseMatrix multOp, multTransposeOp;
|
||||
|
||||
public:
|
||||
ProlongationOperator(OccaSparseMatrix &multOp_,
|
||||
OccaSparseMatrix &multTransposeOp_);
|
||||
|
||||
ProlongationOperator(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::Operator *pmat_);
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
|
||||
// overrides
|
||||
virtual void Mult(const mfem::Vector &x, mfem::Vector &y) const;
|
||||
virtual void MultTranspose(const mfem::Vector &x, mfem::Vector &y) const;
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
|
||||
@@ -1,40 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "layout.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
void Layout::Resize(std::size_t new_size)
|
||||
{
|
||||
size = new_size;
|
||||
}
|
||||
|
||||
void Layout::Resize(const Array<std::size_t> &offsets)
|
||||
{
|
||||
MFEM_ASSERT(offsets.Size() == 2,
|
||||
"multiple workers are not supported yet");
|
||||
size = offsets.Last();
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,68 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_LAYOUT_HPP
|
||||
#define MFEM_BACKENDS_OCCA_LAYOUT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "../base/layout.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Layout : public PLayout
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// std::size_t size;
|
||||
|
||||
public:
|
||||
Layout(const Engine &e, std::size_t s = 0) : PLayout(e, s) { }
|
||||
|
||||
const Engine &OccaEngine() const
|
||||
{ return *static_cast<const Engine *>(engine.Get()); }
|
||||
|
||||
::occa::memory Alloc(std::size_t bytes) const
|
||||
{ return OccaEngine().GetDevice().malloc(bytes); }
|
||||
|
||||
virtual ~Layout() { }
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// Resize the layout
|
||||
virtual void Resize(std::size_t new_size);
|
||||
|
||||
/// Resize the layout based on the given worker offsets
|
||||
virtual void Resize(const Array<std::size_t> &offsets);
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_LAYOUT_HPP
|
||||
@@ -1,54 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
TILESIZE : Tilesize for iterating over entries
|
||||
================================================
|
||||
*/
|
||||
|
||||
@kernel void ExtractSubVector(const int entries,
|
||||
const int * restrict indices,
|
||||
const double * restrict in,
|
||||
double * restrict out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
out[i] = in[indices[i]];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void SetSubVector(const int entries,
|
||||
const int * restrict indices,
|
||||
const double * restrict in,
|
||||
double * restrict out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
out[indices[i]] = in[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MapSubVector(const int entries,
|
||||
const int * restrict indices,
|
||||
const double * restrict in,
|
||||
double * restrict out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
const int fromIdx = indices[2*i + 0];
|
||||
const int toIdx = indices[2*i + 1];
|
||||
out[toIdx] = in[fromIdx];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,135 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "operator.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
// FIXME: move this object to the Backend?
|
||||
::occa::kernelBuilder OccaConstrainedOperator::mapDofBuilder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_map_dofs",
|
||||
|
||||
"const int idx = v2[i];"
|
||||
"v0[idx] = v1[idx];",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}");
|
||||
|
||||
// FIXME: move this object to the Backend?
|
||||
::occa::kernelBuilder OccaConstrainedOperator::clearDofBuilder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_clear_dofs",
|
||||
|
||||
"v0[v1[i]] = 0.0;",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}");
|
||||
|
||||
OccaConstrainedOperator::OccaConstrainedOperator(
|
||||
mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_)
|
||||
|
||||
: Operator(A_->InLayout()->As<Layout>()),
|
||||
z(OutLayout_()),
|
||||
w(OutLayout_()),
|
||||
mfem_z((z.DontDelete(), z)),
|
||||
mfem_w((w.DontDelete(), w))
|
||||
{
|
||||
Setup(OutLayout_().OccaEngine().GetDevice(), A_, constraintList_, own_A_);
|
||||
}
|
||||
|
||||
void OccaConstrainedOperator::Setup(::occa::device device_,
|
||||
mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_)
|
||||
{
|
||||
device = device_;
|
||||
|
||||
A = A_;
|
||||
own_A = own_A_;
|
||||
|
||||
constraintIndices = constraintList_.Size();
|
||||
constraintList = constraintList_.Get_PArray()->As<Array>().OccaMem();
|
||||
}
|
||||
|
||||
void OccaConstrainedOperator::EliminateRHS(const Vector &x, Vector &b) const
|
||||
{
|
||||
const std::string &okl_defines = InLayout_().OccaEngine().GetOklDefines();
|
||||
::occa::kernel mapDofs = mapDofBuilder.build(device, okl_defines);
|
||||
|
||||
w.Fill<double>(0.0);
|
||||
|
||||
if (constraintIndices)
|
||||
{
|
||||
mapDofs(constraintIndices, w.OccaMem(), x.OccaMem(), constraintList);
|
||||
}
|
||||
|
||||
A->Mult(mfem_w, mfem_z);
|
||||
|
||||
b.Axpby<double>(1.0, b, -1.0, z);
|
||||
|
||||
if (constraintIndices)
|
||||
{
|
||||
mapDofs(constraintIndices, b.OccaMem(), x.OccaMem(), constraintList);
|
||||
}
|
||||
}
|
||||
|
||||
void OccaConstrainedOperator::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
mfem::Vector mfem_y(y);
|
||||
if (constraintIndices == 0)
|
||||
{
|
||||
A->Mult(x.Wrap(), mfem_y);
|
||||
return;
|
||||
}
|
||||
|
||||
const std::string &okl_defines = InLayout_().OccaEngine().GetOklDefines();
|
||||
::occa::kernel mapDofs = mapDofBuilder.build(device, okl_defines);
|
||||
::occa::kernel clearDofs = clearDofBuilder.build(device, okl_defines);
|
||||
|
||||
z.Assign<double>(x); // z = x
|
||||
|
||||
clearDofs(constraintIndices, z.OccaMem(), constraintList);
|
||||
|
||||
A->Mult(mfem_z, mfem_y);
|
||||
|
||||
mapDofs(constraintIndices, y.OccaMem(), x.OccaMem(), constraintList);
|
||||
}
|
||||
|
||||
OccaConstrainedOperator::~OccaConstrainedOperator()
|
||||
{
|
||||
if (own_A)
|
||||
{
|
||||
delete A;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,129 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_OPERATOR_HPP
|
||||
#define MFEM_BACKENDS_OCCA_OPERATOR_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "../../linalg/operator.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Operator : public mfem::Operator
|
||||
{
|
||||
public:
|
||||
/// Creare an operator with the same dimensions as @a orig.
|
||||
Operator(const Operator &orig)
|
||||
: mfem::Operator(orig) { }
|
||||
|
||||
Operator(Layout &layout)
|
||||
: mfem::Operator(layout) { }
|
||||
|
||||
Operator(Layout &in_layout, Layout &out_layout)
|
||||
: mfem::Operator(in_layout, out_layout) { }
|
||||
|
||||
Layout &InLayout_() const
|
||||
{ return *static_cast<Layout*>(in_layout.Get()); }
|
||||
|
||||
Layout &OutLayout_() const
|
||||
{ return *static_cast<Layout*>(out_layout.Get()); }
|
||||
|
||||
virtual void Mult_(const Vector &x, Vector &y) const = 0;
|
||||
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const
|
||||
{ MFEM_ABORT("method is not supported"); }
|
||||
|
||||
// override
|
||||
virtual void Mult(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{
|
||||
Mult_(x.Get_PVector()->As<Vector>(),
|
||||
y.Get_PVector()->As<Vector>());
|
||||
}
|
||||
|
||||
// override
|
||||
virtual void MultTranspose(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{
|
||||
MultTranspose_(x.Get_PVector()->As<Vector>(),
|
||||
y.Get_PVector()->As<Vector>());
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
class OccaConstrainedOperator : public Operator
|
||||
{
|
||||
protected:
|
||||
::occa::device device;
|
||||
|
||||
mfem::Operator *A; //< The unconstrained Operator.
|
||||
bool own_A; //< Ownership flag for A.
|
||||
::occa::memory constraintList; //< List of constrained indices/dofs.
|
||||
int constraintIndices;
|
||||
mutable Vector z, w; //< Auxiliary vectors.
|
||||
mutable mfem::Vector mfem_z, mfem_w; // Wrap z, w
|
||||
|
||||
static ::occa::kernelBuilder mapDofBuilder, clearDofBuilder;
|
||||
|
||||
public:
|
||||
/** @brief Constructor from a general Operator and a list of essential
|
||||
indices/dofs.
|
||||
|
||||
Specify the unconstrained operator @a *A and a @a list of indices to
|
||||
constrain, i.e. each entry @a list[i] represents an essential-dof. If the
|
||||
ownership flag @a own_A is true, the operator @a *A will be destroyed
|
||||
when this object is destroyed. */
|
||||
OccaConstrainedOperator(mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_ = false);
|
||||
|
||||
void Setup(::occa::device device_,
|
||||
mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_ = false);
|
||||
|
||||
/** @brief Eliminate "essential boundary condition" values specified in @a x
|
||||
from the given right-hand side @a b.
|
||||
|
||||
Performs the following steps:
|
||||
|
||||
z = A((0,x_b)); b_i -= z_i; b_b = x_b;
|
||||
|
||||
where the "_b" subscripts denote the essential (boundary) indices/dofs of
|
||||
the vectors, and "_i" -- the rest of the entries. */
|
||||
void EliminateRHS(const Vector &x, Vector &b) const;
|
||||
|
||||
/** @brief Constrained operator action.
|
||||
|
||||
Performs the following steps:
|
||||
|
||||
z = A((x_i,0)); y_i = z_i; y_b = x_b;
|
||||
|
||||
where the "_b" subscripts denote the essential (boundary) indices/dofs of
|
||||
the vectors, and "_i" -- the rest of the entries. */
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
|
||||
// Destructor: destroys the unconstrained Operator @a A if @a own_A is true.
|
||||
virtual ~OccaConstrainedOperator();
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_OPERATOR_HPP
|
||||
@@ -1,57 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
TILESIZE : Tilesize for iterating over dofs
|
||||
================================================
|
||||
*/
|
||||
|
||||
@kernel void Mult(const int entries,
|
||||
const int * restrict offsets,
|
||||
const int * restrict indices,
|
||||
const double * restrict weights,
|
||||
const double * restrict in,
|
||||
double * restrict out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
double value = 0;
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
value += weights[j] * in[indices[j]];
|
||||
}
|
||||
out[i] = value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MappedMult(const int entries,
|
||||
const int * restrict offsets,
|
||||
const int * restrict indices,
|
||||
const double * restrict weights,
|
||||
const int * restrict outIndices,
|
||||
const double * restrict in,
|
||||
double * restrict out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
double value = 0;
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
value += weights[j] * in[indices[j]];
|
||||
}
|
||||
out[outIndices[i]] = value;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,248 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "sparsemat.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props) :
|
||||
Operator(in_layout, out_layout)
|
||||
{
|
||||
|
||||
Setup(in_layout.OccaEngine().GetDevice(), m, props);
|
||||
}
|
||||
|
||||
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props) :
|
||||
Operator(in_layout, out_layout)
|
||||
{
|
||||
|
||||
Setup(in_layout.OccaEngine().GetDevice(), m,
|
||||
reorderIndices, mappedIndices_, props);
|
||||
}
|
||||
|
||||
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> offsets_,
|
||||
::occa::array<int> indices_,
|
||||
::occa::array<double> weights_,
|
||||
const ::occa::properties &props) :
|
||||
Operator(in_layout, out_layout),
|
||||
offsets(offsets_),
|
||||
indices(indices_),
|
||||
weights(weights_)
|
||||
{
|
||||
|
||||
SetupKernel(in_layout.OccaEngine().GetDevice(), props);
|
||||
}
|
||||
|
||||
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> offsets_,
|
||||
::occa::array<int> indices_,
|
||||
::occa::array<double> weights_,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props) :
|
||||
Operator(in_layout, out_layout),
|
||||
offsets(offsets_),
|
||||
indices(indices_),
|
||||
weights(weights_),
|
||||
reorderIndices(reorderIndices_),
|
||||
mappedIndices(mappedIndices_)
|
||||
{
|
||||
|
||||
SetupKernel(in_layout.OccaEngine().GetDevice(), props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::Setup(::occa::device device, const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
Setup(device, m, ::occa::array<int>(), ::occa::array<int>(), props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::Setup(::occa::device device, const SparseMatrix &m,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
|
||||
const int nnz = m.GetI()[height];
|
||||
offsets.allocate(device,
|
||||
height + 1, m.GetI());
|
||||
indices.allocate(device,
|
||||
nnz, m.GetJ());
|
||||
weights.allocate(device,
|
||||
nnz, m.GetData());
|
||||
|
||||
offsets.keepInDevice();
|
||||
indices.keepInDevice();
|
||||
weights.keepInDevice();
|
||||
|
||||
reorderIndices = reorderIndices_;
|
||||
mappedIndices = mappedIndices_;
|
||||
|
||||
SetupKernel(device, props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::SetupKernel(::occa::device device,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
|
||||
const bool hasOutIndices = mappedIndices.isInitialized();
|
||||
|
||||
const ::occa::properties defaultProps("defines: {"
|
||||
" TILESIZE: 256,"
|
||||
"}");
|
||||
|
||||
const std::string &okl_path = InLayout_().OccaEngine().GetOklPath();
|
||||
const std::string &okl_defines = InLayout_().OccaEngine().GetOklDefines();
|
||||
mapKernel = device.buildKernel(okl_path + "mappings.okl",
|
||||
"MapSubVector",
|
||||
defaultProps + props + okl_defines);
|
||||
|
||||
multKernel = device.buildKernel(okl_path + "sparse.okl",
|
||||
hasOutIndices ? "MappedMult" : "Mult",
|
||||
defaultProps + props + okl_defines);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (reorderIndices.isInitialized() ||
|
||||
mappedIndices.isInitialized())
|
||||
{
|
||||
if (reorderIndices.isInitialized())
|
||||
{
|
||||
mapKernel((int) (reorderIndices.size() / 2),
|
||||
reorderIndices,
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
if (mappedIndices.isInitialized())
|
||||
{
|
||||
multKernel((int) (mappedIndices.size()),
|
||||
offsets, indices, weights,
|
||||
mappedIndices,
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
multKernel((int) height,
|
||||
offsets, indices, weights,
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
OccaSparseMatrix* CreateMappedSparseMatrix(Layout &in_layout,
|
||||
Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
const int mHeight = m.Height();
|
||||
// const int mWidth = m.Width();
|
||||
|
||||
// Count indices that are only reordered (true dofs)
|
||||
const int *I = m.GetI();
|
||||
const int *J = m.GetJ();
|
||||
const double *D = m.GetData();
|
||||
|
||||
int trueCount = 0;
|
||||
for (int i = 0; i < mHeight; ++i)
|
||||
{
|
||||
trueCount += ((I[i + 1] - I[i]) == 1);
|
||||
}
|
||||
const int dupCount = (mHeight - trueCount);
|
||||
|
||||
// Create the reordering map for entries that aren't modified (true dofs)
|
||||
::occa::device device(in_layout.OccaEngine().GetDevice());
|
||||
::occa::array<int> reorderIndices(device,
|
||||
2 * trueCount);
|
||||
::occa::array<int> mappedIndices, offsets, indices;
|
||||
::occa::array<double> weights;
|
||||
|
||||
if (dupCount)
|
||||
{
|
||||
mappedIndices.allocate(device,
|
||||
dupCount);
|
||||
}
|
||||
int trueIdx = 0, dupIdx = 0;
|
||||
for (int i = 0; i < mHeight; ++i)
|
||||
{
|
||||
const int i1 = I[i];
|
||||
if ((I[i + 1] - i1) == 1)
|
||||
{
|
||||
reorderIndices[trueIdx++] = J[i1];
|
||||
reorderIndices[trueIdx++] = i;
|
||||
}
|
||||
else
|
||||
{
|
||||
mappedIndices[dupIdx++] = i;
|
||||
}
|
||||
}
|
||||
reorderIndices.keepInDevice();
|
||||
|
||||
if (dupCount)
|
||||
{
|
||||
mappedIndices.keepInDevice();
|
||||
|
||||
// Extract sparse matrix without reordered identity
|
||||
const int dupNnz = I[mHeight] - trueCount;
|
||||
|
||||
offsets.allocate(device,
|
||||
dupCount + 1);
|
||||
indices.allocate(device,
|
||||
dupNnz);
|
||||
weights.allocate(device,
|
||||
dupNnz);
|
||||
|
||||
int nnz = 0;
|
||||
offsets[0] = 0;
|
||||
for (int i = 0; i < dupCount; ++i)
|
||||
{
|
||||
const int idx = mappedIndices[i];
|
||||
const int offStart = I[idx];
|
||||
const int offEnd = I[idx + 1];
|
||||
offsets[i + 1] = offsets[i] + (offEnd - offStart);
|
||||
for (int j = offStart; j < offEnd; ++j)
|
||||
{
|
||||
indices[nnz] = J[j];
|
||||
weights[nnz] = D[j];
|
||||
++nnz;
|
||||
}
|
||||
}
|
||||
|
||||
offsets.keepInDevice();
|
||||
indices.keepInDevice();
|
||||
weights.keepInDevice();
|
||||
}
|
||||
|
||||
return new OccaSparseMatrix(in_layout, out_layout,
|
||||
offsets, indices, weights,
|
||||
reorderIndices, mappedIndices,
|
||||
props);
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,95 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
|
||||
#define MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "vector.hpp"
|
||||
#include "engine.hpp"
|
||||
#include "operator.hpp"
|
||||
#include "../../linalg/sparsemat.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
/// TODO: doxygen
|
||||
class OccaSparseMatrix : public Operator
|
||||
{
|
||||
public:
|
||||
::occa::array<int> offsets, indices;
|
||||
::occa::array<double> weights;
|
||||
::occa::array<int> reorderIndices, mappedIndices;
|
||||
::occa::kernel mapKernel, multKernel;
|
||||
|
||||
/// Construct an empty OccaSparseMatrix.
|
||||
OccaSparseMatrix(const Operator &orig)
|
||||
: Operator(orig) { }
|
||||
|
||||
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> offsets_,
|
||||
::occa::array<int> indices_,
|
||||
::occa::array<double> weights_,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> offsets_,
|
||||
::occa::array<int> indices_,
|
||||
::occa::array<double> weights_,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
void Setup(::occa::device device, const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props);
|
||||
|
||||
void Setup(::occa::device device, const mfem::SparseMatrix &m,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props);
|
||||
|
||||
void SetupKernel(::occa::device device,
|
||||
const ::occa::properties &props);
|
||||
|
||||
// override
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
|
||||
/// TODO: doxygen
|
||||
OccaSparseMatrix* CreateMappedSparseMatrix(
|
||||
Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
|
||||
@@ -1,81 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "url_handler.hpp"
|
||||
#include "../../general/error.hpp"
|
||||
#include <cstdlib>
|
||||
#include <sys/stat.h>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
FileOpener::FileOpener(const std::string &prefix,
|
||||
const std::string &env_variable)
|
||||
: pfx(prefix)
|
||||
{
|
||||
const char *env_path = getenv(env_variable.c_str());
|
||||
if (!env_path) { return; }
|
||||
std::string path(env_path);
|
||||
for (std::size_t start = 0, end; start < path.size(); start = end + 1)
|
||||
{
|
||||
end = path.find(':', start);
|
||||
if (end == std::string::npos)
|
||||
{
|
||||
AddDir(path.substr(start, end));
|
||||
break;
|
||||
}
|
||||
AddDir(path.substr(start, end - start));
|
||||
}
|
||||
}
|
||||
|
||||
bool FileOpener::AddDir(const std::string &dir)
|
||||
{
|
||||
if (dir.size() == 0 || dir[0] != '/') { return false; }
|
||||
struct stat dir_stat;
|
||||
if (stat(dir.c_str(), &dir_stat)) { return false; }
|
||||
if (!S_ISDIR(dir_stat.st_mode)) { return false; }
|
||||
paths.push_back(dir + (*dir.rbegin() == '/' ? "" : "/"));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool FileOpener::handles(const std::string &filename)
|
||||
{
|
||||
return filename.size() >= pfx.size() &&
|
||||
filename.compare(0, pfx.size(), pfx) == 0;
|
||||
}
|
||||
|
||||
std::string FileOpener::expand(const std::string &filename)
|
||||
{
|
||||
std::string sfx(filename.substr(pfx.size()));
|
||||
for (std::size_t i = 0; i < paths.size(); i++)
|
||||
{
|
||||
std::string file = paths[i] + sfx;
|
||||
struct stat file_stat;
|
||||
if (stat(file.c_str(), &file_stat) == 0 && S_ISREG(file_stat.st_mode))
|
||||
{
|
||||
return file;
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("invalid url: " << filename);
|
||||
return sfx;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,47 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
|
||||
#define MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class FileOpener : public ::occa::io::fileOpener
|
||||
{
|
||||
protected:
|
||||
std::string pfx; // prefix, e.g. "mfem://"
|
||||
std::vector<std::string> paths; // paths to search for prefix replacement
|
||||
|
||||
public:
|
||||
FileOpener(const std::string &prefix, const std::string &env_variable);
|
||||
|
||||
bool AddDir(const std::string &dir);
|
||||
|
||||
virtual bool handles(const std::string &filename);
|
||||
virtual std::string expand(const std::string &filename);
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
|
||||
@@ -1,22 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
typedef double* Local_t @dim(numDofs, numElements);
|
||||
|
||||
@kernel void InitLocalVector(const int numElements,
|
||||
const int numDofs,
|
||||
Local_t restrict sol) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int d = 0; d < numDofs; ++d; @inner) {
|
||||
sol(d, e) = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,204 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "../../linalg/vector.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
PVector *Vector::DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const
|
||||
{
|
||||
MFEM_ASSERT(buffer_type_id == ScalarId<double>::value, "");
|
||||
Vector *new_vector = new Vector(OccaLayout());
|
||||
if (copy_data)
|
||||
{
|
||||
new_vector->slice.copyFrom(slice);
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
*buffer = new_vector->GetBuffer();
|
||||
}
|
||||
return new_vector;
|
||||
}
|
||||
|
||||
void Vector::DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const
|
||||
{
|
||||
// Can be called when Size() == 0, e.g. when an MPI-parallel vector has a
|
||||
// local size of 0.
|
||||
|
||||
MFEM_ASSERT(result_type_id == ScalarId<double>::value, "");
|
||||
double *res = (double *)result;
|
||||
MFEM_ASSERT(dynamic_cast<const Vector *>(&x) != NULL, "invalid Vector type");
|
||||
const Vector *xp = static_cast<const Vector *>(&x);
|
||||
MFEM_ASSERT(this->Size() == xp->Size(), "");
|
||||
*res = ::occa::linalg::dot<double, double, double>(this->slice, xp->slice);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
double local_dot = *res;
|
||||
if (IsParallel())
|
||||
{
|
||||
MPI_Allreduce(&local_dot, res, 1, MPI_DOUBLE, MPI_SUM,
|
||||
OccaLayout().OccaEngine().GetComm());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void Vector::DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id)
|
||||
{
|
||||
const std::string &okl_defines = OccaLayout().OccaEngine().GetOklDefines();
|
||||
|
||||
//
|
||||
// TODO: move all kernel builders to class mfem::occa::Backend
|
||||
//
|
||||
static ::occa::kernelBuilder axpby1_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"mfem_occa_axpby1",
|
||||
"v0[i] = c0 * v1[i];",
|
||||
"defines: {"
|
||||
" CTYPE0: 'double',"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" TILESIZE: '128',"
|
||||
"}");
|
||||
|
||||
static ::occa::kernelBuilder axpby2_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"mfem_occa_axpby2",
|
||||
"v0[i] = c0 * v0[i] + c1 * v1[i];",
|
||||
"defines: {"
|
||||
" CTYPE0: 'double',"
|
||||
" CTYPE1: 'double',"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" TILESIZE: '128',"
|
||||
"}");
|
||||
|
||||
static ::occa::kernelBuilder axpby3_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"mfem_occa_axpby3",
|
||||
"v0[i] = c0 * v1[i] + c1 * v2[i];",
|
||||
"defines: {"
|
||||
" CTYPE0: 'double',"
|
||||
" CTYPE1: 'double',"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'double',"
|
||||
" TILESIZE: '128',"
|
||||
"}");
|
||||
|
||||
// called only when Size() != 0
|
||||
|
||||
MFEM_ASSERT(ab_type_id == ScalarId<double>::value, "");
|
||||
const double da = *static_cast<const double *>(a);
|
||||
const double db = *static_cast<const double *>(b);
|
||||
MFEM_ASSERT(da == 0.0 || dynamic_cast<const Vector *>(&x) != NULL,
|
||||
"invalid Vector x");
|
||||
MFEM_ASSERT(db == 0.0 || dynamic_cast<const Vector *>(&y) != NULL,
|
||||
"invalid Vector y");
|
||||
const Vector *xp = static_cast<const Vector *>(&x);
|
||||
const Vector *yp = static_cast<const Vector *>(&y);
|
||||
|
||||
MFEM_ASSERT(da == 0.0 || this->Size() == xp->Size(), "");
|
||||
MFEM_ASSERT(db == 0.0 || this->Size() == yp->Size(), "");
|
||||
|
||||
if (da == 0.0)
|
||||
{
|
||||
if (db == 0.0)
|
||||
{
|
||||
OccaFill(&da);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (this->slice == yp->slice)
|
||||
{
|
||||
// *this *= db
|
||||
::occa::linalg::operator_mult_eq(slice, db);
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = db * y
|
||||
::occa::kernel kernel = axpby1_builder.build(slice.getDevice(),
|
||||
okl_defines);
|
||||
kernel((int)Size(), db, slice, yp->slice);
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (db == 0.0)
|
||||
{
|
||||
if (this->slice == xp->slice)
|
||||
{
|
||||
// *this *= da
|
||||
::occa::linalg::operator_mult_eq(slice, da);
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = da * x
|
||||
::occa::kernel kernel = axpby1_builder.build(slice.getDevice(),
|
||||
okl_defines);
|
||||
kernel((int)Size(), da, slice, xp->slice);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ASSERT(xp->slice != yp->slice, "invalid input");
|
||||
if (this->slice == xp->slice)
|
||||
{
|
||||
// *this = da * (*this) + db * y
|
||||
::occa::kernel kernel = axpby2_builder.build(slice.getDevice(),
|
||||
okl_defines);
|
||||
kernel((int)Size(), da, db, slice, yp->slice);
|
||||
}
|
||||
else if (this->slice == yp->slice)
|
||||
{
|
||||
// *this = da * x + db * (*this)
|
||||
::occa::kernel kernel = axpby2_builder.build(slice.getDevice(),
|
||||
okl_defines);
|
||||
kernel((int)Size(), db, da, slice, xp->slice);
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = da * x + db * y
|
||||
::occa::kernel kernel = axpby3_builder.build(slice.getDevice(),
|
||||
okl_defines);
|
||||
kernel((int)Size(), da, db, slice, xp->slice, yp->slice);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mfem::Vector Vector::Wrap()
|
||||
{
|
||||
return mfem::Vector(*this);
|
||||
}
|
||||
|
||||
const mfem::Vector Vector::Wrap() const
|
||||
{
|
||||
return mfem::Vector(*const_cast<Vector*>(this));
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,74 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_VECTOR_HPP
|
||||
#define MFEM_BACKENDS_OCCA_VECTOR_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "../base/vector.hpp"
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Vector : virtual public Array, public PVector
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// DLayout layout;
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
virtual PVector *DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const;
|
||||
|
||||
virtual void DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const;
|
||||
|
||||
virtual void DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id);
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
public:
|
||||
Vector(Layout <)
|
||||
: PArray(lt), Array(lt, sizeof(double)), PVector(lt)
|
||||
{ }
|
||||
|
||||
mfem::Vector Wrap();
|
||||
|
||||
const mfem::Vector Wrap() const;
|
||||
|
||||
#if defined(MFEM_USE_MPI)
|
||||
bool IsParallel() const { return (OccaLayout().OccaEngine().GetComm() != MPI_COMM_NULL); }
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_VECTOR_HPP
|
||||
@@ -1,675 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && \
|
||||
defined(MFEM_USE_OMP) && \
|
||||
defined(MFEM_USE_ACROTENSOR)
|
||||
|
||||
#include "adiffusioninteg.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
PAIntegrator::PAIntegrator(Coefficient &q, FiniteElementSpace &f)
|
||||
{
|
||||
Q = &q;
|
||||
ofes = &f;
|
||||
fes = ofes->GetFESpace();
|
||||
onGPU = (ofes->OmpEngine().ExecTarget() == Device);
|
||||
fe = fes->GetFE(0);
|
||||
tfe = dynamic_cast<const TensorBasisElement*>(fe);
|
||||
if (tfe)
|
||||
{
|
||||
tDofMap = tfe->GetDofMap();
|
||||
}
|
||||
else
|
||||
{
|
||||
tDofMap.SetSize(nDof);
|
||||
for (int i = 0; i < nDof; ++i)
|
||||
{
|
||||
tDofMap[i] = i;
|
||||
}
|
||||
}
|
||||
|
||||
nElem = fes->GetNE();
|
||||
GeomType = fe->GetGeomType();
|
||||
FEOrder = fe->GetOrder();
|
||||
nDim = fe->GetDim();
|
||||
nDof = fe->GetDof();
|
||||
|
||||
ElementTransformation *Trans = fes->GetElementTransformation(0);
|
||||
int irorder = 2*fe->GetOrder() + Trans->OrderW();
|
||||
ir = &IntRules.Get(GeomType, irorder);
|
||||
nQuad = ir->GetNPoints();
|
||||
hasTensorBasis = tfe ? true : false;
|
||||
|
||||
if (nDim > 3)
|
||||
{
|
||||
mfem_error("AcroIntegrator tensor computations don't support dim > 3.");
|
||||
}
|
||||
}
|
||||
|
||||
PAIntegrator::~PAIntegrator()
|
||||
{
|
||||
|
||||
}
|
||||
|
||||
AcroDiffusionIntegrator::AcroDiffusionIntegrator(Coefficient &q, FiniteElementSpace &f) :
|
||||
PAIntegrator(q,f)
|
||||
{
|
||||
if (onGPU)
|
||||
{
|
||||
//TE.SetExecutorType("OneOutPerThread");
|
||||
TE.SetExecutorType("Cuda");
|
||||
//TODO: Set to an existing cuda context if one exists
|
||||
}
|
||||
else
|
||||
{
|
||||
TE.SetExecutorType("CPUInterpreted");
|
||||
}
|
||||
|
||||
const IntegrationRule *ir1D = &IntRules.Get(Geometry::SEGMENT, ir->GetOrder());
|
||||
nDof1D = FEOrder + 1;
|
||||
nQuad1D = ir1D->GetNPoints();
|
||||
|
||||
if (hasTensorBasis)
|
||||
{
|
||||
H1_FECollection fec(FEOrder,1);
|
||||
const FiniteElement *fe1D = fec.FiniteElementForGeometry(Geometry::SEGMENT);
|
||||
mfem::Vector eval(nDof1D);
|
||||
DenseMatrix deval(nDof1D,1);
|
||||
B.Init(nQuad1D, nDof1D);
|
||||
G.Init(nQuad1D, nDof1D);
|
||||
std::vector<int> wdims(nDim, nQuad1D);
|
||||
W.Init(wdims);
|
||||
|
||||
mfem::Vector w(nQuad1D);
|
||||
for (int k = 0; k < nQuad1D; ++k)
|
||||
{
|
||||
const IntegrationPoint &ip = ir1D->IntPoint(k);
|
||||
fe1D->CalcShape(ip, eval);
|
||||
fe1D->CalcDShape(ip, deval);
|
||||
|
||||
B(k,0) = eval(0);
|
||||
B(k,nDof1D-1) = eval(1);
|
||||
G(k,0) = deval(0,0);
|
||||
G(k,nDof1D-1) = deval(1,0);
|
||||
for (int i = 1; i < nDof1D-1; ++i)
|
||||
{
|
||||
B(k,i) = eval(i+1);
|
||||
G(k,i) = deval(i+1,0);
|
||||
}
|
||||
w(k) = ip.weight;
|
||||
}
|
||||
|
||||
if (nDim == 1)
|
||||
{
|
||||
for (int k1 = 0; k1 < nQuad1D; ++k1)
|
||||
{
|
||||
W(k1) = w(k1);
|
||||
}
|
||||
}
|
||||
else if (nDim == 2)
|
||||
{
|
||||
for (int k1 = 0; k1 < nQuad1D; ++k1)
|
||||
{
|
||||
for (int k2 = 0; k2 < nQuad1D; ++k2)
|
||||
{
|
||||
W(k1,k2) = w(k1)*w(k2);
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (nDim == 3)
|
||||
{
|
||||
for (int k1 = 0; k1 < nQuad1D; ++k1)
|
||||
{
|
||||
for (int k2 = 0; k2 < nQuad1D; ++k2)
|
||||
{
|
||||
for (int k3 = 0; k3 < nQuad1D; ++k3)
|
||||
{
|
||||
W(k1,k2,k3) = w(k1)*w(k2)*w(k3);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem::Vector eval(nDof);
|
||||
DenseMatrix deval(nDof,nDim);
|
||||
G.Init(nQuad, nDof,nDim);
|
||||
W.Init(nQuad);
|
||||
for (int k = 0; k < nQuad; ++k)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(k);
|
||||
fe->CalcDShape(ip, deval);
|
||||
for (int i = 0; i < nDof; ++i)
|
||||
{
|
||||
for (int d = 0; d < nDim; ++d)
|
||||
{
|
||||
G(k,i,d) = deval(i,d);
|
||||
}
|
||||
}
|
||||
W(k) = ip.weight;
|
||||
}
|
||||
}
|
||||
|
||||
if (onGPU)
|
||||
{
|
||||
B.MapToGPU();
|
||||
G.MapToGPU();
|
||||
W.MapToGPU();
|
||||
}
|
||||
|
||||
// Assemble in the constructor!
|
||||
BatchedPartialAssemble();
|
||||
}
|
||||
|
||||
|
||||
AcroDiffusionIntegrator::~AcroDiffusionIntegrator()
|
||||
{
|
||||
for (int i = 0; i < Btil.Size(); i++) delete Btil[i];
|
||||
}
|
||||
|
||||
|
||||
void AcroDiffusionIntegrator::ComputeBTilde()
|
||||
{
|
||||
Btil.SetSize(nDim);
|
||||
for (int d = 0; d < nDim; ++d)
|
||||
{
|
||||
Btil[d] = new acro::Tensor(nDim, nDim, nQuad1D, nDof1D, nDof1D);
|
||||
for (int m = 0; m < nDim; ++m)
|
||||
{
|
||||
for (int n = 0; n < nDim; ++n)
|
||||
{
|
||||
acro::Tensor &BGM = (m == d) ? G : B;
|
||||
acro::Tensor &BGN = (n == d) ? G : B;
|
||||
for (int k = 0; k < nQuad1D; ++k)
|
||||
{
|
||||
for (int i = 0; i < nDof1D; ++i)
|
||||
{
|
||||
for (int j = 0; j < nDof1D; ++j)
|
||||
{
|
||||
(*Btil[d])(m, n, k, i, j) = BGM(k,i)*BGN(k,j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void AcroDiffusionIntegrator::BatchedPartialAssemble()
|
||||
{
|
||||
//Initilze the tensors
|
||||
acro::Tensor J,Jinv,Jdet,C;
|
||||
if (hasTensorBasis)
|
||||
{
|
||||
const IntegrationRule *ir1D = &IntRules.Get(Geometry::SEGMENT, ir->GetOrder());
|
||||
IntegrationPoint ip;
|
||||
if (nDim == 1)
|
||||
{
|
||||
D.Init(nElem, nDim, nDim, nQuad1D);
|
||||
J.Init(nElem, nQuad1D, nDim, nDim);
|
||||
Jinv.Init(nElem, nQuad1D, nDim, nDim);
|
||||
Jdet.Init(nElem, nQuad1D);
|
||||
C.Init(nElem, nQuad1D);
|
||||
|
||||
for (int e = 0; e < nElem; ++e)
|
||||
{
|
||||
ElementTransformation *Trans = fes->GetElementTransformation(e);
|
||||
for (int k1 = 0; k1 < nQuad1D; ++k1)
|
||||
{
|
||||
ip.x = ir1D->IntPoint(k1).x;
|
||||
ip.y = 0.0;
|
||||
ip.z = 0.0;
|
||||
Trans->SetIntPoint(&ip);
|
||||
C(e,k1) = Q->Eval(*Trans, ip);
|
||||
const DenseMatrix &JMat = Trans->Jacobian();
|
||||
for (int m = 0; m < nDim; ++m)
|
||||
{
|
||||
for (int n = 0; n < nDim; ++n)
|
||||
{
|
||||
J(e,k1,m,n) = JMat.Elem(m,n);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (nDim == 2)
|
||||
{
|
||||
D.Init(nElem, nDim, nDim, nQuad1D, nQuad1D);
|
||||
J.Init(nElem, nQuad1D, nQuad1D, nDim, nDim);
|
||||
Jinv.Init(nElem, nQuad1D, nQuad1D, nDim, nDim);
|
||||
Jdet.Init(nElem, nQuad1D, nQuad1D);
|
||||
C.Init(nElem, nQuad1D, nQuad1D);
|
||||
|
||||
for (int e = 0; e < nElem; ++e)
|
||||
{
|
||||
ElementTransformation *Trans = fes->GetElementTransformation(e);
|
||||
for (int k1 = 0; k1 < nQuad1D; ++k1)
|
||||
{
|
||||
for (int k2 = 0; k2 < nQuad1D; ++k2)
|
||||
{
|
||||
ip.x = ir1D->IntPoint(k1).x;
|
||||
ip.y = ir1D->IntPoint(k2).y;
|
||||
ip.z = 0.0;
|
||||
Trans->SetIntPoint(&ip);
|
||||
C(e,k1,k2) = Q->Eval(*Trans, ip);
|
||||
const DenseMatrix &JMat = Trans->Jacobian();
|
||||
for (int m = 0; m < nDim; ++m)
|
||||
{
|
||||
for (int n = 0; n < nDim; ++n)
|
||||
{
|
||||
J(e,k1,k2,m,n) = JMat.Elem(m,n);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (nDim == 3)
|
||||
{
|
||||
D.Init(nElem, nDim, nDim, nQuad1D, nQuad1D, nQuad1D);
|
||||
J.Init(nElem, nQuad1D, nQuad1D, nQuad1D, nDim, nDim);
|
||||
Jinv.Init(nElem, nQuad1D, nQuad1D, nQuad1D, nDim, nDim);
|
||||
Jdet.Init(nElem, nQuad1D, nQuad1D, nQuad1D);
|
||||
C.Init(nElem, nQuad1D, nQuad1D, nQuad1D);
|
||||
|
||||
for (int e = 0; e < nElem; ++e)
|
||||
{
|
||||
ElementTransformation *Trans = fes->GetElementTransformation(e);
|
||||
for (int k1 = 0; k1 < nQuad1D; ++k1)
|
||||
{
|
||||
for (int k2 = 0; k2 < nQuad1D; ++k2)
|
||||
{
|
||||
for (int k3 = 0; k3 < nQuad1D; ++k3)
|
||||
{
|
||||
ip.x = ir1D->IntPoint(k1).x;
|
||||
ip.y = ir1D->IntPoint(k2).y;
|
||||
ip.z = ir1D->IntPoint(k3).z;
|
||||
Trans->SetIntPoint(&ip);
|
||||
C(e,k1,k2,k3) = Q->Eval(*Trans, ip);
|
||||
const DenseMatrix &JMat = Trans->Jacobian();
|
||||
for (int m = 0; m < nDim; ++m)
|
||||
{
|
||||
for (int n = 0; n < nDim; ++n)
|
||||
{
|
||||
J(e,k1,k2,k3,m,n) = JMat.Elem(m,n);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
D.Init(nElem, nDim, nDim, nQuad);
|
||||
J.Init(nElem, nQuad, nDim, nDim);
|
||||
Jinv.Init(nElem, nQuad, nDim, nDim);
|
||||
Jdet.Init(nElem, nQuad);
|
||||
C.Init(nElem, nQuad);
|
||||
|
||||
for (int e = 0; e < nElem; ++e)
|
||||
{
|
||||
ElementTransformation *Trans = fes->GetElementTransformation(e);
|
||||
for (int k = 0; k < nQuad; ++k)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(k);
|
||||
Trans->SetIntPoint(&ip);
|
||||
C(e,k) = Q->Eval(*Trans, ip);
|
||||
const DenseMatrix &JMat = Trans->Jacobian();
|
||||
for (int m = 0; m < nDim; ++m)
|
||||
{
|
||||
for (int n = 0; n < nDim; ++n)
|
||||
{
|
||||
J(e,k,m,n) = JMat.Elem(m,n);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
TE.BatchMatrixInvDet(Jinv, Jdet, J);
|
||||
|
||||
if (hasTensorBasis)
|
||||
{
|
||||
if (nDim == 1)
|
||||
{
|
||||
TE("D_e_m_n_k = W_k C_e_k Jdet_e_k Jinv_e_k_m_j Jinv_e_k_n_j",
|
||||
D, W, C, Jdet, Jinv, Jinv);
|
||||
}
|
||||
else if (nDim == 2)
|
||||
{
|
||||
TE("D_e_m_n_k1_k2 = W_k1_k2 C_e_k1_k2 Jdet_e_k1_k2 Jinv_e_k1_k2_m_j Jinv_e_k1_k2_n_j",
|
||||
D, W, C, Jdet, Jinv, Jinv);
|
||||
}
|
||||
else if (nDim == 3)
|
||||
{
|
||||
TE("D_e_m_n_k1_k2_k3 = W_k1_k2_k3 C_e_k1_k2_k3 Jdet_e_k1_k2_k3 Jinv_e_k1_k2_k3_n_j Jinv_e_k1_k2_k3_m_j",
|
||||
D, W, C, Jdet, Jinv, Jinv);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
TE("D_e_m_n_k = W_k C_e_k Jdet_e_k Jinv_e_k_m_j Jinv_e_k_n_j",
|
||||
D, W, C, Jdet, Jinv, Jinv);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void AcroDiffusionIntegrator::BatchedAssembleElementMatrices(DenseTensor &elmats)
|
||||
{
|
||||
if (hasTensorBasis && Btil.Size() == 0)
|
||||
{
|
||||
ComputeBTilde();
|
||||
}
|
||||
|
||||
if (!D.IsInitialized())
|
||||
{
|
||||
BatchedPartialAssemble();
|
||||
}
|
||||
|
||||
if (!S.IsInitialized())
|
||||
{
|
||||
if (hasTensorBasis)
|
||||
{
|
||||
if (nDim == 1)
|
||||
{
|
||||
S.Init(nElem, nDof1D, nDof1D);
|
||||
}
|
||||
else if (nDim == 2)
|
||||
{
|
||||
S.Init(nElem, nDof1D, nDof1D, nDof1D, nDof1D);
|
||||
}
|
||||
else if (nDim == 3)
|
||||
{
|
||||
S.Init(nElem, nDof1D, nDof1D, nDof1D, nDof1D, nDof1D, nDof1D);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
S.Init(nElem, nDof, nDof);
|
||||
}
|
||||
if (onGPU) {S.SwitchToGPU();}
|
||||
}
|
||||
|
||||
|
||||
if (hasTensorBasis) {
|
||||
if (nDim == 1) {
|
||||
TE("S_e_i1_j1 = Btil_m_n_k1_i1_j1 D_e_m_n_k1",
|
||||
S, *Btil[0], D);
|
||||
}
|
||||
else if (nDim == 2)
|
||||
{
|
||||
TE("S_e_i1_i2_j1_j2 = Btil1_m_n_k1_i1_j1 Btil2_m_n_k2_i2_j2 D_e_m_n_k1_k2",
|
||||
S, *Btil[0], *Btil[1], D);
|
||||
}
|
||||
else if (nDim == 3)
|
||||
{
|
||||
TE("S_e_i1_i2_i3_j1_j2_j3 = Btil1_m_n_k1_i1_j1 Btil2_m_n_k2_i2_j2 Btil3_m_n_k3_i3_j3 D_e_m_n_k1_k2_k3",
|
||||
S, *Btil[0], *Btil[1], *Btil[2], D);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
TE("S_e_i_j = G_k_i_m G_k_i_n D_e_m_n_k",
|
||||
S, G, G, D);
|
||||
}
|
||||
|
||||
S.MoveFromGPU();
|
||||
for (int e = 0; e < nElem; ++e)
|
||||
{
|
||||
for (int ei = 0; ei < nDof; ++ei)
|
||||
{
|
||||
for (int ej = 0; ej < nDof; ++ej)
|
||||
{
|
||||
elmats(tDofMap[ei], tDofMap[ej], e) = S[e*nDof*nDof + ei*nDof + ej];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void AcroDiffusionIntegrator::ComputeElementMatrices(Vector &elmats)
|
||||
{
|
||||
if (hasTensorBasis && Btil.Size() == 0)
|
||||
{
|
||||
ComputeBTilde();
|
||||
}
|
||||
|
||||
if (!D.IsInitialized())
|
||||
{
|
||||
BatchedPartialAssemble();
|
||||
}
|
||||
|
||||
if (!S.IsInitialized())
|
||||
{
|
||||
if (hasTensorBasis)
|
||||
{
|
||||
if (nDim == 1)
|
||||
{
|
||||
S.Init(nElem, nDof1D, nDof1D);
|
||||
}
|
||||
else if (nDim == 2)
|
||||
{
|
||||
S.Init(nElem, nDof1D, nDof1D, nDof1D, nDof1D);
|
||||
}
|
||||
else if (nDim == 3)
|
||||
{
|
||||
S.Init(nElem, nDof1D, nDof1D, nDof1D, nDof1D, nDof1D, nDof1D);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
S.Init(nElem, nDof, nDof);
|
||||
}
|
||||
if (onGPU) {S.SwitchToGPU();}
|
||||
}
|
||||
|
||||
|
||||
if (hasTensorBasis) {
|
||||
if (nDim == 1) {
|
||||
TE("S_e_i1_j1 += Btil_m_n_k1_i1_j1 D_e_m_n_k1",
|
||||
S, *Btil[0], D);
|
||||
}
|
||||
else if (nDim == 2)
|
||||
{
|
||||
TE("S_e_i1_i2_j1_j2 += Btil1_m_n_k1_i1_j1 Btil2_m_n_k2_i2_j2 D_e_m_n_k1_k2",
|
||||
S, *Btil[0], *Btil[1], D);
|
||||
}
|
||||
else if (nDim == 3)
|
||||
{
|
||||
TE("S_e_i1_i2_i3_j1_j2_j3 += Btil1_m_n_k1_i1_j1 Btil2_m_n_k2_i2_j2 Btil3_m_n_k3_i3_j3 D_e_m_n_k1_k2_k3",
|
||||
S, *Btil[0], *Btil[1], *Btil[2], D);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
TE("S_e_i_j += G_k_i_m G_k_i_n D_e_m_n_k",
|
||||
S, G, G, D);
|
||||
}
|
||||
|
||||
S.MoveFromGPU();
|
||||
|
||||
double *edata = elmats.GetData<double>();
|
||||
for (int e = 0; e < nElem; ++e)
|
||||
{
|
||||
const int e_offset = e * nDof * nDof;
|
||||
for (int ei = 0; ei < nDof; ++ei)
|
||||
{
|
||||
const int offset = e_offset + ei * tDofMap[ei] * nDof;
|
||||
for (int ej = 0; ej < nDof; ++ej)
|
||||
{
|
||||
const int index = offset + tDofMap[ej];
|
||||
edata[index] = S[e*nDof*nDof + ei*nDof + ej];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void AcroDiffusionIntegrator::ReassembleOperator()
|
||||
{
|
||||
BatchedPartialAssemble();
|
||||
}
|
||||
|
||||
void AcroDiffusionIntegrator::PAMult(const Vector &x, Vector &y)
|
||||
{
|
||||
MFEM_ASSERT(hasTensorBasis,"AcroDiffusionIntegrator PAMult on simplices not supported");
|
||||
|
||||
if (!U.IsInitialized())
|
||||
{
|
||||
// NOTE: x and y are already sized for the fespace in the constructor
|
||||
double *Xptr = const_cast<double*>(x.GetData<double>());
|
||||
double *Yptr = y.GetData<double>();
|
||||
if (nDim == 1) {
|
||||
X.Init(nElem,nDof1D,Xptr,Xptr,onGPU);
|
||||
Y.Init(nElem,nDof1D,Yptr,Yptr,onGPU);
|
||||
U.Init(nDim, nElem, nQuad1D);
|
||||
Z.Init(nDim, nElem, nQuad1D);
|
||||
if (onGPU)
|
||||
{
|
||||
U.SwitchToGPU();
|
||||
Z.SwitchToGPU();
|
||||
}
|
||||
}
|
||||
else if (nDim == 2)
|
||||
{
|
||||
X.Init(nElem,nDof1D,nDof1D,Xptr,Xptr,onGPU);
|
||||
Y.Init(nElem,nDof1D,nDof1D,Yptr,Yptr,onGPU);
|
||||
U.Init(nDim, nElem, nQuad1D, nQuad1D);
|
||||
Z.Init(nDim, nElem, nQuad1D, nQuad1D);
|
||||
T1.Init(nElem,nDof1D,nQuad1D);
|
||||
if (onGPU)
|
||||
{
|
||||
U.SwitchToGPU();
|
||||
Z.SwitchToGPU();
|
||||
T1.SwitchToGPU();
|
||||
}
|
||||
}
|
||||
else if (nDim == 3)
|
||||
{
|
||||
X.Init(nElem,nDof1D,nDof1D,nDof1D,Xptr,Xptr,onGPU);
|
||||
Y.Init(nElem,nDof1D,nDof1D,nDof1D,Yptr,Yptr,onGPU);
|
||||
U.Init(nDim, nElem, nQuad1D, nQuad1D, nQuad1D);
|
||||
Z.Init(nDim, nElem, nQuad1D, nQuad1D, nQuad1D);
|
||||
T1.Init(nElem, nDof1D, nQuad1D, nQuad1D);
|
||||
T2.Init(nElem, nDof1D, nDof1D, nQuad1D);
|
||||
if (onGPU)
|
||||
{
|
||||
U.SwitchToGPU();
|
||||
Z.SwitchToGPU();
|
||||
T1.SwitchToGPU();
|
||||
T2.SwitchToGPU();
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// NOTE: x and y are already sized for the fespace in the constructor
|
||||
double *Xptr = const_cast<double*>(x.GetData<double>());
|
||||
double *Yptr = y.GetData<double>();
|
||||
X.Retarget(Xptr,Xptr);
|
||||
Y.Retarget(Yptr,Yptr);
|
||||
}
|
||||
|
||||
acro::SliceTensor U1,U2,U3,Z1,Z2,Z3;
|
||||
if (nDim == 1)
|
||||
{
|
||||
TE("U_n_e_k1 = G_k1_i1 X_e_i1", U, G, X);
|
||||
TE("Z_m_e_k1 = D_e_m_n_k1 U_n_e_k1", Z, D, U);
|
||||
TE("Y_e_i1 = G_k1_i1 Z_m_e_k1", Y, G, Z);
|
||||
}
|
||||
else if (nDim == 2)
|
||||
{
|
||||
U1.SliceInit(U, 0); U2.SliceInit(U, 1);
|
||||
Z1.SliceInit(Z, 0); Z2.SliceInit(Z, 1);
|
||||
|
||||
//U1_e_k1_k2 = G_k1_i1 B_k2_i2 X_e_i1_i2
|
||||
TE("BX_e_i1_k2 = B_k2_i2 X_e_i2_i1", T1, B, X);
|
||||
TE("U1_e_k1_k2 = G_k1_i1 BX_e_i1_k2", U1, G, T1);
|
||||
|
||||
//U2_e_k1_k2 = B_k1_i1 G_k2_i2 X_e_i1_i2
|
||||
TE("GX_e_i1_k2 = G_k2_i2 X_e_i2_i1", T1, G, X);
|
||||
TE("U2_e_k1_k2 = B_k1_i1 GX_e_i1_k2", U2, B, T1);
|
||||
|
||||
TE("Z_m_e_k1_k2 = D_e_m_n_k1_k2 U_n_e_k1_k2", Z, D, U);
|
||||
|
||||
//Y_e_i1_i2 = G_k1_i1 B_k2_i2 Z1_e_k1_k2
|
||||
TE("BZ1_e_i2_k1 = B_k2_i2 Z1_e_k1_k2", T1, B, Z1);
|
||||
TE("Y_e_i2_i1 = G_k1_i1 BZ1_e_i2_k1", Y, G, T1);
|
||||
|
||||
//Y_e_i1_i2 += B_k1_i1 G_k2_i2 Z2_e_k1_k2
|
||||
TE("GZ2_e_i2_k1 = G_k2_i2 Z2_e_k1_k2", T1, G, Z2);
|
||||
TE("Y_e_i2_i1 += B_k1_i1 GZ2_e_i2_k1", Y, B, T1);
|
||||
}
|
||||
else if (nDim == 3)
|
||||
{
|
||||
U1.SliceInit(U, 0); U2.SliceInit(U, 1); U3.SliceInit(U, 2);
|
||||
Z1.SliceInit(Z, 0); Z2.SliceInit(Z, 1); Z3.SliceInit(Z, 2);
|
||||
|
||||
TE.BeginMultiKernelLaunch();
|
||||
//U1_e_k1_k2_k3 = G_k1_i1 B_k2_i2 B_k3_i3 X_e_i1_i2_i3
|
||||
TE("T2_e_i1_i2_k3 = B_k3_i3 X_e_i1_i2_i3", T2, B, X);
|
||||
TE("T1_e_i1_k2_k3 = B_k2_i2 T2_e_i1_i2_k3", T1, B, T2);
|
||||
TE("U1_e_k1_k2_k3 = G_k1_i1 T1_e_i1_k2_k3", U1, G, T1);
|
||||
|
||||
//U2_e_k1_k2_k3 = B_k1_i1 G_k2_i2 B_k3_i3 X_e_i1_i2_i3
|
||||
TE("T1_e_i1_k2_k3 = G_k2_i2 T2_e_i1_i2_k3", T1, G, T2);
|
||||
TE("U2_e_k1_k2_k3 = B_k1_i1 T1_e_i1_k2_k3", U2, B, T1);
|
||||
|
||||
//U3_e_k1_k2_k3 = B_k1_i1 B_k2_i2 G_k3_i3 X_e_i1_i2_i3
|
||||
TE("T2_e_i1_i2_k3 = G_k3_i3 X_e_i1_i2_i3", T2, G, X);
|
||||
TE("T1_e_i1_k2_k3 = B_k2_i2 T2_e_i1_i2_k3", T1, B, T2);
|
||||
TE("U3_e_k1_k2_k3 = B_k1_i1 T1_e_i1_k2_k3", U3, B, T1);
|
||||
|
||||
TE("Z_m_e_k1_k2_k3 = D_e_m_n_k1_k2_k3 U_n_e_k1_k2_k3", Z, D, U);
|
||||
|
||||
//Y_e_i1_i2_i3 = G_k1_i1 B_k2_i2 B_k3_i3 Z1_e_k1_k2_k3
|
||||
TE("T1_e_i3_k1_k2 = B_k3_i3 Z1_e_k1_k2_k3", T1, B, Z1);
|
||||
TE("T2_e_i2_i3_k1 = B_k2_i2 T1_e_i3_k1_k2", T2, B, T1);
|
||||
TE("Y_e_i1_i2_i3 = G_k1_i1 T2_e_i2_i3_k1", Y, G, T2);
|
||||
|
||||
//Y_e_i1_i2_i3 += B_k1_i1 G_k2_i2 B_k3_i3 Z2_e_k1_k2_k3
|
||||
TE("T1_e_i3_k1_k2 = B_k3_i3 Z2_e_k1_k2_k3", T1, B, Z2);
|
||||
TE("T2_e_i2_i3_k1 = G_k2_i2 T1_e_i3_k1_k2", T2, G, T1);
|
||||
TE("Y_e_i1_i2_i3 += B_k1_i1 T2_e_i2_i3_k1", Y, B, T2);
|
||||
|
||||
//Y_e_i1_i2_i3 += B_k1_i1 B_k2_i2 G_k3_i3 Z3_e_k1_k2_k3
|
||||
TE("T1_e_i3_k1_k2 = G_k3_i3 Z3_e_k1_k2_k3", T1, G, Z3);
|
||||
TE("T2_e_i2_i3_k1 = B_k2_i2 T1_e_i3_k1_k2", T2, B, T1);
|
||||
TE("Y_e_i1_i2_i3 += B_k1_i1 T2_e_i2_i3_k1", Y, B, T2);
|
||||
TE.EndMultiKernelLaunch();
|
||||
}
|
||||
}
|
||||
|
||||
void AcroDiffusionIntegrator::MultAdd(const Vector &x, Vector &y) const
|
||||
{
|
||||
const_cast<AcroDiffusionIntegrator*>(this)->PAMult(x, y);
|
||||
}
|
||||
|
||||
void AcroDiffusionIntegrator::MultTransposeAdd(const Vector &x, Vector &y) const
|
||||
{
|
||||
mfem_error("Not supported");
|
||||
}
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
@@ -1,95 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OMP_ADIFFUSIONINTEG_HPP
|
||||
#define MFEM_BACKENDS_OMP_ADIFFUSIONINTEG_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && \
|
||||
defined(MFEM_USE_OMP) && \
|
||||
defined(MFEM_USE_ACROTENSOR)
|
||||
|
||||
#include "../../fem/bilininteg.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
#include "vector.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "AcroTensor.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
class PAIntegrator : public TensorBilinearFormIntegrator
|
||||
{
|
||||
protected:
|
||||
Coefficient *Q;
|
||||
FiniteElementSpace *ofes;
|
||||
mfem::FiniteElementSpace *fes;
|
||||
const FiniteElement *fe;
|
||||
const TensorBasisElement *tfe;
|
||||
const IntegrationRule *ir;
|
||||
mfem::Array<int> tDofMap;
|
||||
int GeomType;
|
||||
int FEOrder;
|
||||
bool onGPU;
|
||||
bool hasTensorBasis;
|
||||
int nDim;
|
||||
int nElem;
|
||||
int nDof;
|
||||
int nQuad;
|
||||
|
||||
public:
|
||||
PAIntegrator(Coefficient &q, FiniteElementSpace &f);
|
||||
virtual ~PAIntegrator();
|
||||
};
|
||||
|
||||
class AcroDiffusionIntegrator : public PAIntegrator
|
||||
{
|
||||
private:
|
||||
acro::TensorEngine TE;
|
||||
int nDof1D;
|
||||
int nQuad1D;
|
||||
|
||||
acro::Tensor B, G; //Basis and dbasis evaluated on the quad points
|
||||
acro::Tensor W; //Integration weights
|
||||
mfem::Array<acro::Tensor*> Btil; //Btilde used to compute stiffness matrix
|
||||
acro::Tensor D; //Product of integration weight, physical consts, and element shape info
|
||||
acro::Tensor S; //The assembled local stiffness matrices
|
||||
acro::Tensor U, Z, T1, T2; //Intermediate computations for tensor product partial assembly
|
||||
acro::Tensor X, Y;
|
||||
|
||||
void ComputeBTilde();
|
||||
|
||||
public:
|
||||
AcroDiffusionIntegrator(BilinearFormIntegrator *integ);
|
||||
AcroDiffusionIntegrator(Coefficient &q, FiniteElementSpace &f);
|
||||
virtual ~AcroDiffusionIntegrator();
|
||||
|
||||
void BatchedPartialAssemble();
|
||||
void BatchedAssembleElementMatrices(DenseTensor &elmats);
|
||||
void ComputeElementMatrices(Vector &elmats);
|
||||
void PAMult(const Vector &x, Vector &y);
|
||||
virtual void MultTransposeAdd(const Vector &x, Vector &y) const;
|
||||
virtual void MultAdd(const Vector &x, Vector &y) const;
|
||||
virtual void ReassembleOperator();
|
||||
};
|
||||
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -1,128 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include <cstring>
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
PArray *Array::DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
Array *new_array = new Array(OmpLayout(), item_size);
|
||||
if (copy_data)
|
||||
{
|
||||
if (!ComputeOnDevice())
|
||||
std::memcpy(new_array->GetData<void>(), data, bytes);
|
||||
else
|
||||
{
|
||||
char *new_data = new_array->GetData<char>();
|
||||
const bool use_target = ComputeOnDevice();
|
||||
const bool use_parallel = Size() > 1000;
|
||||
#pragma omp target teams distribute parallel for \
|
||||
if (target: use_target) if (parallel: use_parallel) \
|
||||
is_device_ptr(new_data)
|
||||
for (std::size_t i = 0; i < bytes; i++) new_data[i] = data[i];
|
||||
}
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
*buffer = new_array->GetData<void>();
|
||||
}
|
||||
return new_array;
|
||||
}
|
||||
|
||||
int Array::DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size)
|
||||
{
|
||||
MFEM_ASSERT(dynamic_cast<Layout *>(&new_layout) != NULL,
|
||||
"new_layout is not an OMP Layout");
|
||||
Layout *lt = static_cast<Layout *>(&new_layout);
|
||||
layout.Reset(lt); // Reset() checks if the pointer is the same
|
||||
int err = ResizeData(lt, item_size);
|
||||
if (!err && buffer)
|
||||
{
|
||||
*buffer = GetData<void>();
|
||||
}
|
||||
return err;
|
||||
}
|
||||
|
||||
void *Array::DoPullData(void *buffer, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
if (!IsUnifiedMemory() && ComputeOnDevice() && (buffer != NULL))
|
||||
{
|
||||
#pragma omp target update from(data)
|
||||
std::memcpy(buffer, data, bytes);
|
||||
}
|
||||
else
|
||||
{
|
||||
buffer = data;
|
||||
}
|
||||
|
||||
return buffer;
|
||||
}
|
||||
|
||||
void Array::DoFill(const void *value_ptr, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
switch (item_size)
|
||||
{
|
||||
case sizeof(int):
|
||||
OmpFill((const int *)value_ptr);
|
||||
break;
|
||||
case sizeof(double):
|
||||
OmpFill((const double *)value_ptr);
|
||||
break;
|
||||
default:
|
||||
MFEM_ABORT("item_size = " << item_size << " is not supported");
|
||||
}
|
||||
}
|
||||
|
||||
void Array::DoPushData(const void *src_buffer, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
std::memcpy(data, (char *) src_buffer, bytes);
|
||||
|
||||
if ((!IsUnifiedMemory() && ComputeOnDevice()) && (data != src_buffer))
|
||||
{
|
||||
#pragma omp target update to(data)
|
||||
}
|
||||
}
|
||||
|
||||
void Array::DoAssign(const PArray &src, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
// Note: static_cast can not be used here since PArray is a virtual base
|
||||
// class.
|
||||
const Array *source = dynamic_cast<const Array *>(&src);
|
||||
MFEM_ASSERT(source != NULL, "invalid source Array type");
|
||||
MFEM_ASSERT(Size() == source->Size(), "");
|
||||
// All arrays from this engine are of the same type, so we can simply check *this and assume the same is used in src.
|
||||
DoPushData(source->GetData<void>(), item_size);
|
||||
}
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
@@ -1,143 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OMP_ARRAY_HPP
|
||||
#define MFEM_BACKENDS_OMP_ARRAY_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "layout.hpp"
|
||||
#include "../base/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
class Array : public virtual mfem::PArray
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// DLayout layout;
|
||||
|
||||
bool own_data;
|
||||
std::size_t bytes;
|
||||
char *data;
|
||||
|
||||
//
|
||||
// Virtual interface
|
||||
//
|
||||
|
||||
virtual void *DoGetData() const { return (void *) data; }
|
||||
|
||||
virtual PArray *DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const;
|
||||
|
||||
virtual int DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size);
|
||||
|
||||
virtual void *DoPullData(void *buffer, std::size_t item_size);
|
||||
|
||||
virtual void DoFill(const void *value_ptr, std::size_t item_size);
|
||||
|
||||
virtual void DoPushData(const void *src_buffer, std::size_t item_size);
|
||||
|
||||
virtual void DoAssign(const PArray &src, std::size_t item_size);
|
||||
|
||||
//
|
||||
// Auxiliary methods
|
||||
//
|
||||
|
||||
inline int ResizeData(const Layout *lt, std::size_t item_size);
|
||||
|
||||
inline bool IsUnifiedMemory() const { return OmpLayout().OmpEngine().UnifiedMemory(); }
|
||||
|
||||
template <typename T>
|
||||
void OmpFill(const T *pval)
|
||||
{
|
||||
T *ptr = (T*) data;
|
||||
T val = *pval;
|
||||
const bool use_target = ComputeOnDevice();
|
||||
const bool use_parallel = (use_target || layout->Size() > 1000);
|
||||
const std::size_t size = layout->Size();
|
||||
#pragma omp target teams distribute parallel for \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel) map (to: ptr, val)
|
||||
for (int i = 0; i < size; i++) ptr[i] = val;
|
||||
}
|
||||
|
||||
public:
|
||||
Array(Layout <, std::size_t item_size)
|
||||
: PArray(lt),
|
||||
own_data(true),
|
||||
bytes(lt.Size() * item_size),
|
||||
data(static_cast<char *>(lt.Alloc(bytes)))
|
||||
{
|
||||
#pragma omp target enter data map(alloc:data[:bytes]) if (!IsUnifiedMemory() && ComputeOnDevice())
|
||||
}
|
||||
|
||||
Array(const Array &array)
|
||||
: PArray(array.GetLayout()),
|
||||
own_data(false),
|
||||
bytes(array.bytes),
|
||||
data(array.data) { }
|
||||
|
||||
inline bool ComputeOnDevice() const { return (OmpLayout().OmpEngine().ExecTarget() == Device); }
|
||||
|
||||
virtual ~Array()
|
||||
{
|
||||
#pragma omp target exit data map(delete:data[:bytes]) if (!IsUnifiedMemory() && ComputeOnDevice())
|
||||
if (own_data) layout->As<Layout>().Dealloc(data);
|
||||
}
|
||||
|
||||
inline void MakeRef(Array &master);
|
||||
|
||||
Layout &OmpLayout() const
|
||||
{ return *static_cast<Layout *>(layout.Get()); }
|
||||
};
|
||||
|
||||
|
||||
//
|
||||
// Inline methods
|
||||
//
|
||||
|
||||
inline int Array::ResizeData(const Layout *lt, std::size_t item_size)
|
||||
{
|
||||
const std::size_t new_bytes = lt->Size() * item_size;
|
||||
if (bytes < new_bytes)
|
||||
{
|
||||
#pragma omp target exit data map(delete:data)
|
||||
OmpLayout().Dealloc(data);
|
||||
data = static_cast<char *>(OmpLayout().Alloc(new_bytes));
|
||||
MFEM_VERIFY(data != NULL, "");
|
||||
// If memory allocation fails - an exception is thrown.
|
||||
#pragma omp target enter data map(alloc:data[:new_bytes])
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
inline void Array::MakeRef(Array &master)
|
||||
{
|
||||
layout = master.layout;
|
||||
data = master.data;
|
||||
}
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#endif // MFEM_BACKENDS_OMP_ARRAY_HPP
|
||||
@@ -1,46 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
bool Backend::Supports(const std::string &engine_spec) const
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
mfem::Engine *Create(const std::string &engine_spec)
|
||||
{
|
||||
return new Engine(engine_spec);
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec)
|
||||
{
|
||||
return new Engine(comm, engine_spec);
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
@@ -1,48 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OMP_BACKEND_HPP
|
||||
#define MFEM_BACKENDS_OMP_BACKEND_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
// Only the Backend and Engine classes should be exposed through "backend.hpp"
|
||||
#include "../base/backend.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
class Backend : public mfem::Backend
|
||||
{
|
||||
public:
|
||||
virtual ~Backend();
|
||||
|
||||
virtual bool Supports(const std::string &engine_spec) const;
|
||||
|
||||
virtual mfem::Engine *Create(const std::string &engine_spec);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
virtual mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec);
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#endif // MFEM_BACKENDS_OMP_BACKEND_HPP
|
||||
@@ -1,399 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "adiffusioninteg.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
BilinearForm::~BilinearForm()
|
||||
{
|
||||
// Make sure all integrators free their data
|
||||
for (int i = 0; i < tbfi.Size(); i++) delete tbfi[i];
|
||||
|
||||
delete element_matrices;
|
||||
}
|
||||
|
||||
void BilinearForm::TransferIntegrators()
|
||||
{
|
||||
mfem::Array<mfem::BilinearFormIntegrator*> &dbfi = *bform->GetDBFI();
|
||||
for (int i = 0; i < dbfi.Size(); i++)
|
||||
{
|
||||
std::string integ_name(dbfi[i]->Name());
|
||||
Coefficient *scal_coeff = dbfi[i]->GetScalarCoefficient();
|
||||
// ConstantCoefficient *const_coeff =
|
||||
// dynamic_cast<ConstantCoefficient*>(scal_coeff);
|
||||
// // TODO: other types of coefficients ...
|
||||
// double val = const_coeff ? const_coeff->constant : 1.0;
|
||||
|
||||
if (integ_name == "(undefined)")
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator does not define Name()");
|
||||
}
|
||||
else if (integ_name == "diffusion")
|
||||
{
|
||||
switch (OmpEngine().IntegType())
|
||||
{
|
||||
case Acrotensor:
|
||||
tbfi.Append(new AcroDiffusionIntegrator(*scal_coeff, bform->FESpace()->Get_PFESpace()->As<FiniteElementSpace>()));
|
||||
break;
|
||||
default:
|
||||
mfem_error("integrator is not supported for any MultType");
|
||||
break;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator [Name() = " << integ_name
|
||||
<< "] is not supported");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::InitRHS(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::Vector &mfem_x, mfem::Vector &mfem_b,
|
||||
mfem::OperatorHandle &A,
|
||||
mfem::Vector &mfem_X, mfem::Vector &mfem_B,
|
||||
int copy_interior) const
|
||||
{
|
||||
const mfem::Operator *P = GetProlongation();
|
||||
const mfem::Operator *R = GetRestriction();
|
||||
|
||||
if (P)
|
||||
{
|
||||
// Variational restriction with P
|
||||
mfem_B.Resize(P->InLayout());
|
||||
P->MultTranspose(mfem_b, mfem_B);
|
||||
mfem_X.Resize(R->OutLayout());
|
||||
R->Mult(mfem_x, mfem_X);
|
||||
}
|
||||
else
|
||||
{
|
||||
// rap, X and B point to the same data as this, x and b
|
||||
mfem_X.MakeRef(mfem_x);
|
||||
mfem_B.MakeRef(mfem_b);
|
||||
}
|
||||
|
||||
if (A.Type() != mfem::Operator::ANY_TYPE)
|
||||
{
|
||||
A.EliminateBC(mat_e, ess_tdof_list, mfem_X, mfem_B);
|
||||
}
|
||||
|
||||
if (!copy_interior && ess_tdof_list.Size() > 0)
|
||||
{
|
||||
Vector &X = mfem_X.Get_PVector()->As<Vector>();
|
||||
const Array &constraint_list = ess_tdof_list.Get_PArray()->As<Array>();
|
||||
|
||||
double *X_data = X.GetData<double>();
|
||||
const int* constraint_data = constraint_list.GetData<int>();
|
||||
|
||||
Vector subvec(constraint_list.OmpLayout());
|
||||
double *subvec_data = subvec.GetData<double>();
|
||||
|
||||
const std::size_t num_constraint = constraint_list.Size();
|
||||
const bool use_target = constraint_list.ComputeOnDevice();
|
||||
const bool use_parallel = (use_target || num_constraint > 1000);
|
||||
|
||||
// This operation is a general version of mfem::Vector::SetSubVectorComplement()
|
||||
// {
|
||||
#pragma omp target teams distribute parallel for \
|
||||
map(to: subvec_data, constraint_data, X_data) \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel)
|
||||
for (std::size_t i = 0; i < num_constraint; i++) subvec_data[i] = X_data[constraint_data[i]];
|
||||
|
||||
X.Fill(0.0);
|
||||
|
||||
#pragma omp target teams distribute parallel for \
|
||||
map(to: X_data, constraint_data, subvec_data) \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel)
|
||||
for (std::size_t i = 0; i < num_constraint; i++) X_data[constraint_data[i]] = subvec_data[i];
|
||||
// }
|
||||
}
|
||||
|
||||
if (A.Type() == mfem::Operator::ANY_TYPE)
|
||||
{
|
||||
ConstrainedOperator *A_constrained = static_cast<ConstrainedOperator*>(A.Ptr());
|
||||
A_constrained->EliminateRHS(mfem_X, mfem_B);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
bool BilinearForm::Assemble()
|
||||
{
|
||||
if (!has_assembled)
|
||||
{
|
||||
TransferIntegrators();
|
||||
has_assembled = true;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void BilinearForm::ComputeElementMatrices()
|
||||
{
|
||||
// Only called if performing full assembly
|
||||
const int nelements = trial_fes->GetFESpace()->GetNE();
|
||||
const int trial_ndofs = trial_fes->GetFESpace()->GetFE(0)->GetDof() * trial_fes->GetFESpace()->GetVDim();
|
||||
const int test_ndofs = test_fes->GetFESpace()->GetFE(0)->GetDof() * test_fes->GetFESpace()->GetVDim();
|
||||
const std::size_t length = nelements * trial_ndofs * test_ndofs;
|
||||
|
||||
if (!element_matrices) element_matrices = new mfem::Vector(*(new Layout(OmpEngine(), length)));
|
||||
else element_matrices->Push();
|
||||
|
||||
element_matrices->Fill(0.0);
|
||||
Vector &elmats = element_matrices->Get_PVector()->As<Vector>();
|
||||
|
||||
tbfi[0]->ComputeElementMatrices(elmats);
|
||||
|
||||
if (tbfi.Size() > 1)
|
||||
{
|
||||
for (int k = 1; k < tbfi.Size(); k++)
|
||||
{
|
||||
tbfi[k]->ComputeElementMatrices(elmats);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::OperatorHandle &A)
|
||||
{
|
||||
if (A.Type() == mfem::Operator::ANY_TYPE)
|
||||
{
|
||||
// FIXME: Support different test and trial spaces (MixedBilinearForm)
|
||||
const mfem::Operator *P = GetProlongation();
|
||||
|
||||
mfem::Operator *rap = this;
|
||||
if (P != NULL) rap = new mfem::RAPOperator(*P, *this, *P);
|
||||
|
||||
A.Reset(new ConstrainedOperator(rap, ess_tdof_list, (rap != this)));
|
||||
|
||||
return;
|
||||
}
|
||||
else
|
||||
{
|
||||
// ASSUMPTION: some sort of sparse matrix
|
||||
// Compute the local matrices (stored in bform->element_matrices
|
||||
ComputeElementMatrices();
|
||||
bform->AllocateMatrix();
|
||||
mfem::SparseMatrix &mat = bform->SpMat();
|
||||
|
||||
element_matrices->Pull();
|
||||
double *data = element_matrices->GetData();
|
||||
|
||||
const bool skip_zeros = true;
|
||||
mfem::Array<int> tr_vdofs, te_vdofs;
|
||||
for (int i = 0; i < trial_fes->GetFESpace()->GetNE(); i++)
|
||||
{
|
||||
trial_fes->GetFESpace()->GetElementVDofs(i, tr_vdofs);
|
||||
test_fes->GetFESpace()->GetElementVDofs(i, te_vdofs);
|
||||
const mfem::DenseMatrix elmat(data, te_vdofs.Size(), tr_vdofs.Size());
|
||||
mat.AddSubMatrix(te_vdofs, tr_vdofs, elmat, skip_zeros);
|
||||
data += tr_vdofs.Size() * te_vdofs.Size();
|
||||
}
|
||||
}
|
||||
|
||||
if (A.Type() == mfem::Operator::MFEM_SPARSEMAT)
|
||||
{
|
||||
// This works because the FormSystemMatrix call with an explicit
|
||||
// SparseMatrix doesnt call the backend version... This might
|
||||
// change in the future.
|
||||
bform->FormSystemMatrix(ess_tdof_list, static_cast<mfem::SparseMatrix&>(*A.Ptr()));
|
||||
}
|
||||
#ifdef MFEM_USE_MPI
|
||||
else if (A.Type() == mfem::Operator::Hypre_ParCSR)
|
||||
{
|
||||
mfem::SparseMatrix &mat = bform->SpMat();
|
||||
mfem::ParBilinearForm *pbform = dynamic_cast<mfem::ParBilinearForm*>(bform);
|
||||
|
||||
const bool skip_zeros = false;
|
||||
mat.Finalize(skip_zeros);
|
||||
|
||||
// -------- FOR SOME VERY AGGREVATING REASON THIS DOESN'T WORK ---------
|
||||
// mfem::ParFiniteElementSpace *pfes = pbform->ParFESpace();
|
||||
// OperatorHandle dA(Operator::Hypre_ParCSR);
|
||||
// // construct a parallel block-diagonal matrix 'A' based on 'a'
|
||||
// dA.MakeSquareBlockDiag(pfes->GetComm(), *engine->MakeLayout(pfes->GlobalTrueVSize()),
|
||||
// pfes->GetDofOffsets(), &mat);
|
||||
// OperatorHandle Ph(pfes->Dof_TrueDof_Matrix());
|
||||
// A.MakePtAP(dA, Ph);
|
||||
// A.SetOperatorOwner(false);
|
||||
// -------- BUT THIS DOES ---------
|
||||
pbform->ParallelAssemble(A, &mat);
|
||||
A.SetOperatorOwner(false);
|
||||
// ---------------------
|
||||
mat.Clear();
|
||||
mat_e.Clear();
|
||||
std::cout << "operator size (FormSystemMatrix): " << A.Ptr()->InLayout()->Size() << " " << A.Ptr()->OutLayout()->Size() << std::endl;
|
||||
|
||||
mat_e.EliminateRowsCols(A, ess_tdof_list);
|
||||
}
|
||||
#endif
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Operator::Type is not supported, type = " << A.Type());
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::OperatorHandle &A, mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior)
|
||||
{
|
||||
FormSystemMatrix(ess_tdof_list, A);
|
||||
std::cout << "operator size (FormLinearSystem 1): " << A.Ptr()->InLayout()->Size() << " " << A.Ptr()->OutLayout()->Size() << std::endl;
|
||||
InitRHS(ess_tdof_list, x, b, A, X, B, copy_interior);
|
||||
}
|
||||
|
||||
void BilinearForm::RecoverFEMSolution(const mfem::Vector &X, const mfem::Vector &b,
|
||||
mfem::Vector &x)
|
||||
{
|
||||
const mfem::Operator *P = GetProlongation();
|
||||
if (P)
|
||||
{
|
||||
// Apply conforming prolongation
|
||||
x.Resize(P->OutLayout());
|
||||
P->Mult(X, x);
|
||||
}
|
||||
// Otherwise X and x point to the same data
|
||||
}
|
||||
|
||||
void BilinearForm::Mult(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{
|
||||
trial_fes->ToEVector(x.Get_PVector()->As<Vector>(), x_local);
|
||||
|
||||
y_local.Fill<double>(0.0);
|
||||
for (int i = 0; i < tbfi.Size(); i++) tbfi[i]->MultAdd(x_local, y_local);
|
||||
|
||||
test_fes->ToLVector(y_local, y.Get_PVector()->As<Vector>());
|
||||
}
|
||||
|
||||
void BilinearForm::MultTranspose(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{ mfem_error("mfem::omp::BilinearForm::MultTranspose() is not supported!"); }
|
||||
|
||||
|
||||
ConstrainedOperator::ConstrainedOperator(mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraint_list_,
|
||||
bool own_A_)
|
||||
: Operator(A_->InLayout()->As<Layout>()),
|
||||
A(A_),
|
||||
own_A(own_A_),
|
||||
// FIXME: @dudouit1 has a general fix for this
|
||||
constraint_list(constraint_list_.Get_PArray()->As<Array>()),
|
||||
z(OutLayout()->As<Layout>()),
|
||||
w(OutLayout()->As<Layout>()),
|
||||
mfem_z((z.DontDelete(), z)),
|
||||
mfem_w((w.DontDelete(), w)) { }
|
||||
|
||||
void ConstrainedOperator::EliminateRHS(const mfem::Vector &mfem_x, mfem::Vector &mfem_b) const
|
||||
{
|
||||
w.Fill<double>(0.0);
|
||||
|
||||
const Vector &x = mfem_x.Get_PVector()->As<Vector>();
|
||||
Vector &b = mfem_b.Get_PVector()->As<Vector>();
|
||||
|
||||
const double *x_data = x.GetData<double>();
|
||||
double *b_data = b.GetData<double>();
|
||||
double *w_data = w.GetData<double>();
|
||||
const int* constraint_data = constraint_list.GetData<int>();
|
||||
|
||||
const std::size_t num_constraint = constraint_list.Size();
|
||||
const bool use_target = constraint_list.ComputeOnDevice();
|
||||
const bool use_parallel = (use_target || num_constraint > 1000);
|
||||
|
||||
if (num_constraint > 0)
|
||||
{
|
||||
#pragma omp target teams distribute parallel for \
|
||||
map(to: w_data, constraint_data, x_data) \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel)
|
||||
for (std::size_t i = 0; i < num_constraint; i++)
|
||||
w_data[constraint_data[i]] = x_data[constraint_data[i]];
|
||||
}
|
||||
|
||||
A->Mult(mfem_w, mfem_z);
|
||||
|
||||
b.Axpby<double>(1.0, b, -1.0, z);
|
||||
|
||||
if (num_constraint > 0)
|
||||
{
|
||||
#pragma omp target teams distribute parallel for \
|
||||
map(to: b_data, constraint_data, x_data) \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel)
|
||||
for (std::size_t i = 0; i < num_constraint; i++)
|
||||
b_data[constraint_data[i]] = x_data[constraint_data[i]];
|
||||
}
|
||||
}
|
||||
|
||||
void ConstrainedOperator::Mult(const mfem::Vector &mfem_x, mfem::Vector &mfem_y) const
|
||||
{
|
||||
if (constraint_list.Size() == 0)
|
||||
{
|
||||
A->Mult(mfem_x, mfem_y);
|
||||
return;
|
||||
}
|
||||
|
||||
const Vector &x = mfem_x.Get_PVector()->As<Vector>();
|
||||
Vector &y = mfem_y.Get_PVector()->As<Vector>();
|
||||
|
||||
const double *x_data = x.GetData<double>();
|
||||
double *y_data = y.GetData<double>();
|
||||
double *z_data = z.GetData<double>();
|
||||
const int* constraint_data = constraint_list.GetData<int>();
|
||||
|
||||
const std::size_t num_constraint = constraint_list.Size();
|
||||
const bool use_target = constraint_list.ComputeOnDevice();
|
||||
const bool use_parallel = (use_target || num_constraint > 1000);
|
||||
|
||||
z.Assign<double>(x); // z = x
|
||||
|
||||
// z[constraint_list] = 0.0
|
||||
#pragma omp target teams distribute parallel for \
|
||||
map(to: z_data, constraint_data) \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel)
|
||||
for (std::size_t i = 0; i < num_constraint; i++)
|
||||
z_data[constraint_data[i]] = 0.0;
|
||||
|
||||
// y = A * z
|
||||
A->Mult(mfem_z, mfem_y);
|
||||
|
||||
// y[constraint_list] = x[constraint_list]
|
||||
#pragma omp target teams distribute parallel for \
|
||||
map(to: y_data, constraint_data, x_data) \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel)
|
||||
for (std::size_t i = 0; i < num_constraint; i++)
|
||||
y_data[constraint_data[i]] = x_data[constraint_data[i]];
|
||||
}
|
||||
|
||||
// Destructor: destroys the unconstrained Operator @a A if @a own_A is true.
|
||||
ConstrainedOperator::~ConstrainedOperator()
|
||||
{
|
||||
if (own_A) delete A;
|
||||
}
|
||||
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
@@ -1,176 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OMP_BILINEARFORM_HPP
|
||||
#define MFEM_BACKENDS_OMP_BILINEARFORM_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "fespace.hpp"
|
||||
#include "array.hpp"
|
||||
#include "vector.hpp"
|
||||
#include "../../fem/bilininteg.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
class TensorBilinearFormIntegrator
|
||||
{
|
||||
public:
|
||||
virtual ~TensorBilinearFormIntegrator() { }
|
||||
|
||||
virtual void ReassembleOperator() = 0;
|
||||
|
||||
virtual void ComputeElementMatrices(Vector &element_matrices)
|
||||
{ mfem_error("TensorBilinaerFormIntegrator::ComputeElementMatrices is not overloaded"); }
|
||||
|
||||
virtual void MultAdd(const Vector &x, Vector &y) const = 0;
|
||||
|
||||
virtual void Mult(const Vector &x, Vector &y) const
|
||||
{ y.Fill<double>(0.0); MultAdd(x, y); }
|
||||
};
|
||||
|
||||
/// TODO: doxygen
|
||||
class BilinearForm : public mfem::PBilinearForm, public mfem::Operator
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// mfem::BilinearForm *bform;
|
||||
|
||||
mfem::Array<TensorBilinearFormIntegrator*> tbfi;
|
||||
bool has_assembled;
|
||||
|
||||
mutable FiniteElementSpace *trial_fes, *test_fes;
|
||||
|
||||
mutable Vector x_local, y_local;
|
||||
|
||||
mfem::Vector *element_matrices;
|
||||
OperatorHandle mat_e;
|
||||
|
||||
void TransferIntegrators();
|
||||
|
||||
void ComputeElementMatrices();
|
||||
|
||||
void InitRHS(const mfem::Array<int> &constraint_list,
|
||||
mfem::Vector &mfem_x, mfem::Vector &mfem_b,
|
||||
mfem::OperatorHandle &A,
|
||||
mfem::Vector &mfem_X, mfem::Vector &mfem_B,
|
||||
int copy_interior = 0) const;
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
BilinearForm(const Engine &e, mfem::BilinearForm &bf)
|
||||
: mfem::PBilinearForm(e, bf),
|
||||
// FIXME: for mixed bilinear forms
|
||||
mfem::Operator(*bf.FESpace()->GetVLayout().As<Layout>()),
|
||||
tbfi(),
|
||||
has_assembled(false),
|
||||
trial_fes(&bf.FESpace()->Get_PFESpace()->As<FiniteElementSpace>()),
|
||||
test_fes(&bf.FESpace()->Get_PFESpace()->As<FiniteElementSpace>()),
|
||||
x_local(trial_fes->GetELayout()),
|
||||
y_local(test_fes->GetELayout()),
|
||||
element_matrices(NULL),
|
||||
mat_e() { }
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~BilinearForm();
|
||||
|
||||
/// Return the engine as an OpenMP engine
|
||||
const Engine &OmpEngine() { return static_cast<const Engine&>(*engine); }
|
||||
|
||||
/** @brief Prolongation operator from linear algebra (linear system) vectors,
|
||||
to input vectors for the operator. `NULL` means identity. */
|
||||
virtual const Operator *GetProlongation() const { return trial_fes->GetProlongation(); }
|
||||
|
||||
/** @brief Restriction operator from input vectors for the operator to linear
|
||||
algebra (linear system) vectors. `NULL` means identity. */
|
||||
virtual const Operator *GetRestriction() const { return test_fes->GetRestriction(); }
|
||||
|
||||
/// Assemble the PBilinearForm.
|
||||
/** This method is called from the method BilinearForm::Assemble() of the
|
||||
associated BilinearForm #bform.
|
||||
@returns True, if the host assembly should be skipped. */
|
||||
virtual bool Assemble();
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::OperatorHandle &A);
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::OperatorHandle &A, mfem::Vector &mfem_X, mfem::Vector &mfem_B,
|
||||
int copy_interior);
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void RecoverFEMSolution(const mfem::Vector &mfem_X, const mfem::Vector &mfem_b,
|
||||
mfem::Vector &mfem_x);
|
||||
|
||||
/// Operator application: `y=A(x)`.
|
||||
virtual void Mult(const mfem::Vector &mfem_x, mfem::Vector &mfem_y) const;
|
||||
|
||||
/** @brief Action of the transpose operator: `y=A^t(x)`. The default behavior
|
||||
in class Operator is to generate an error. */
|
||||
virtual void MultTranspose(const mfem::Vector &mfem_x, mfem::Vector &mfem_y) const;
|
||||
};
|
||||
|
||||
class ConstrainedOperator : public mfem::Operator
|
||||
{
|
||||
const mfem::Operator *A;
|
||||
const bool own_A;
|
||||
const Array constraint_list;
|
||||
mutable Vector z, w;
|
||||
mutable mfem::Vector mfem_z, mfem_w;
|
||||
|
||||
public:
|
||||
ConstrainedOperator(mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraint_list_,
|
||||
bool own_A_ = false);
|
||||
|
||||
// Destructor: destroys the unconstrained Operator @a A if @a own_A is true.
|
||||
virtual ~ConstrainedOperator();
|
||||
|
||||
/** @brief Eliminate "essential boundary condition" values specified in @a x
|
||||
from the given right-hand side @a b.
|
||||
|
||||
Performs the following steps:
|
||||
|
||||
z = A((0,x_b)); b_i -= z_i; b_b = x_b;
|
||||
|
||||
where the "_b" subscripts denote the essential (boundary) indices/dofs of
|
||||
the vectors, and "_i" -- the rest of the entries. */
|
||||
void EliminateRHS(const mfem::Vector &mfem_x, mfem::Vector &mfem_b) const;
|
||||
|
||||
/** @brief Constrained operator action.
|
||||
|
||||
Performs the following steps:
|
||||
|
||||
z = A((x_i,0)); y_i = z_i; y_b = x_b;
|
||||
|
||||
where the "_b" subscripts denote the essential (boundary) indices/dofs of
|
||||
the vectors, and "_i" -- the rest of the entries. */
|
||||
virtual void Mult(const mfem::Vector &mfem_x, mfem::Vector &mfem_y) const;
|
||||
};
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#endif // MFEM_BACKENDS_OMP_BILINEAR_FORM_HPP
|
||||
@@ -1,253 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "array.hpp"
|
||||
#include "layout.hpp"
|
||||
#include "vector.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "memory_resource.hpp"
|
||||
|
||||
#include <map>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
typedef std::map<std::string, std::string> keyval_pair_t;
|
||||
|
||||
template<typename T, typename P>
|
||||
static T remove_if(T beg, T end, P pred)
|
||||
{
|
||||
T dest = beg;
|
||||
for (T itr = beg;itr != end; ++itr)
|
||||
if (!pred(*itr))
|
||||
*(dest++) = *itr;
|
||||
return dest;
|
||||
}
|
||||
|
||||
void parse_token(const std::string &token, std::string &key, std::string &val)
|
||||
{
|
||||
std::size_t sep = token.find_first_of(':');
|
||||
if (sep > token.size()) mfem_error("Parse error");
|
||||
|
||||
key = token.substr(0, sep);
|
||||
key.erase(mfem::omp::remove_if(key.begin(), key.end(), isspace), key.end());
|
||||
key.erase(std::remove(key.begin(), key.end(), '\''), key.end());
|
||||
|
||||
val = token.substr(sep+1);
|
||||
val.erase(mfem::omp::remove_if(val.begin(), val.end(), isspace), val.end());
|
||||
val.erase(std::remove(val.begin(), val.end(), '\''), val.end());
|
||||
}
|
||||
|
||||
keyval_pair_t parse_engine_spec(const std::string &engine_spec)
|
||||
{
|
||||
keyval_pair_t map;
|
||||
std::size_t token_extent = 0;
|
||||
std::string key, val;
|
||||
while (token_extent < engine_spec.size())
|
||||
{
|
||||
const std::string remaining(engine_spec, token_extent);
|
||||
|
||||
std::size_t next_comma = remaining.find_first_of(',');
|
||||
if (next_comma == std::string::npos) next_comma = engine_spec.size() - 1;
|
||||
|
||||
const std::string token(remaining, 0, next_comma);
|
||||
parse_token(token, key, val);
|
||||
|
||||
map[key] = val;
|
||||
token_extent += next_comma+1;
|
||||
}
|
||||
return map;
|
||||
}
|
||||
|
||||
void Engine::Init(const std::string &engine_spec)
|
||||
{
|
||||
keyval_pair_t tokens(parse_engine_spec(engine_spec));
|
||||
keyval_pair_t::iterator it;
|
||||
|
||||
it = tokens.find("exec_target");
|
||||
if (it != tokens.end())
|
||||
{
|
||||
if (!std::strncmp(it->second.data(), "device", 6))
|
||||
{
|
||||
exec_target = Device;
|
||||
device_number = 0;
|
||||
}
|
||||
else if (!std::strncmp(it->second.data(), "host", 4))
|
||||
{
|
||||
exec_target = Host;
|
||||
device_number = -1;
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem_error("Parse error. Possible values for exec_target are: ['host', 'device']");
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// Default to host if not specified
|
||||
mfem::out << "Did not specify exec_target. Defaulting to host..." << std::endl;
|
||||
exec_target = Host;
|
||||
device_number = -1;
|
||||
}
|
||||
|
||||
it = tokens.find("mem_type");
|
||||
if (it != tokens.end())
|
||||
{
|
||||
if (!std::strncmp(it->second.data(), "unified", 7))
|
||||
{
|
||||
#if defined(MFEM_USE_CUDAUM)
|
||||
memory_resources[0] = new UnifiedMemoryResource();
|
||||
unified_memory = true;
|
||||
#else
|
||||
mfem_error("Have not compiled support for CUDA unified memory.");
|
||||
#endif
|
||||
}
|
||||
else if (!std::strncmp(it->second.data(), "separate", 4))
|
||||
{
|
||||
memory_resources[0] = new NewDeleteMemoryResource();
|
||||
unified_memory = false;
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem_error("Parse error. Possible values for mem_type are: ['separate', 'unified']");
|
||||
}
|
||||
}
|
||||
else {
|
||||
if (exec_target == Device)
|
||||
{
|
||||
#if defined(MFEM_USE_CUDAUM)
|
||||
mfem::out << "Did not specify mem_type in engine spec. Defaulting to unified memory..." << std::endl;
|
||||
// Default to unified memory
|
||||
memory_resources[0] = new UnifiedMemoryResource();
|
||||
unified_memory = true;
|
||||
#else
|
||||
mfem::out << "Did not specify mem_type in engine spec. Defaulting to standard host memory..." << std::endl;
|
||||
memory_resources[0] = new NewDeleteMemoryResource();
|
||||
unified_memory = false;
|
||||
#endif
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem::out << "Did not specify mem_type in engine spec. Defaulting to standard host memory..." << std::endl;
|
||||
memory_resources[0] = new NewDeleteMemoryResource();
|
||||
unified_memory = false;
|
||||
}
|
||||
}
|
||||
|
||||
it = tokens.find("mult_engine");
|
||||
if (it != tokens.end())
|
||||
{
|
||||
if (!std::strncmp(it->second.data(), "acrotensor", 10))
|
||||
{
|
||||
mult_type = Acrotensor;
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem_error("Parse error. Possible values for mem_type are: ['acrotensor'].");
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem::out << "Did not specify mult_engine in engine spec. Defaulting to Acrotensor..." << std::endl;
|
||||
#ifndef MFEM_USE_ACROTENSOR
|
||||
mfem_error("Must compile with Acrotensor support");
|
||||
#endif
|
||||
mult_type = Acrotensor;
|
||||
}
|
||||
}
|
||||
|
||||
Engine::Engine(const std::string &engine_spec)
|
||||
: mfem::Engine(NULL, 1, 1)
|
||||
{
|
||||
Init(engine_spec);
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
Engine::Engine(MPI_Comm _comm, const std::string &engine_spec)
|
||||
: mfem::Engine(NULL, 1, 1)
|
||||
{
|
||||
comm = _comm;
|
||||
Init(engine_spec);
|
||||
}
|
||||
#endif
|
||||
|
||||
DLayout Engine::MakeLayout(std::size_t size) const
|
||||
{
|
||||
return DLayout(new Layout(*this, size));
|
||||
}
|
||||
|
||||
DLayout Engine::MakeLayout(const mfem::Array<std::size_t> &offsets) const
|
||||
{
|
||||
MFEM_ASSERT(offsets.Size() == 2,
|
||||
"multiple workers are not supported yet");
|
||||
return DLayout(new Layout(*this, offsets.Last()));
|
||||
}
|
||||
|
||||
DArray Engine::MakeArray(PLayout &layout, std::size_t item_size) const
|
||||
{
|
||||
MFEM_ASSERT(dynamic_cast<Layout *>(&layout) != NULL,
|
||||
"invalid input layout");
|
||||
Layout *lt = static_cast<Layout *>(&layout);
|
||||
return DArray(new Array(*lt, item_size));
|
||||
}
|
||||
|
||||
DVector Engine::MakeVector(PLayout &layout, int type_id) const
|
||||
{
|
||||
MFEM_ASSERT(type_id == ScalarId<double>::value, "invalid type_id");
|
||||
MFEM_ASSERT(dynamic_cast<Layout *>(&layout) != NULL,
|
||||
"invalid input layout");
|
||||
Layout *lt = static_cast<Layout *>(&layout);
|
||||
return DVector(new Vector(*lt));
|
||||
}
|
||||
|
||||
DFiniteElementSpace Engine::MakeFESpace(mfem::FiniteElementSpace &fespace) const
|
||||
{
|
||||
return DFiniteElementSpace(new FiniteElementSpace(*this, fespace));
|
||||
}
|
||||
|
||||
DBilinearForm Engine::MakeBilinearForm(mfem::BilinearForm &bf) const
|
||||
{
|
||||
return DBilinearForm(new BilinearForm(*this, bf));
|
||||
}
|
||||
|
||||
void Engine::AssembleLinearForm(LinearForm &l_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
}
|
||||
|
||||
mfem::Operator *Engine::MakeOperator(const MixedBilinearForm &mbl_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
mfem::Operator *Engine::MakeOperator(const NonlinearForm &nl_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
@@ -1,123 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OMP_ENGINE_HPP
|
||||
#define MFEM_BACKENDS_OMP_ENGINE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "../base/engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
enum ExecutionTarget { Host, Device };
|
||||
|
||||
enum IntegratorType { Acrotensor };
|
||||
|
||||
class Engine : public mfem::Engine
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// mfem::Backend *backend;
|
||||
#ifdef MFEM_USE_MPI
|
||||
// MPI_Comm comm;
|
||||
#endif
|
||||
// int num_mem_res;
|
||||
// int num_workers;
|
||||
// MemoryResource **memory_resources;
|
||||
// double *workers_weights;
|
||||
// int *workers_mem_res;
|
||||
|
||||
enum ExecutionTarget exec_target;
|
||||
bool unified_memory;
|
||||
int device_number;
|
||||
IntegratorType mult_type;
|
||||
|
||||
void Init(const std::string &engine_spec);
|
||||
|
||||
public:
|
||||
Engine(const std::string &engine_spec);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
Engine(MPI_Comm comm, const std::string &engine_spec);
|
||||
#endif
|
||||
|
||||
virtual ~Engine() { }
|
||||
|
||||
/**
|
||||
@name OMP specific interface, used by other objects in the OMP backend
|
||||
*/
|
||||
///@{
|
||||
|
||||
IntegratorType IntegType() const { return mult_type; }
|
||||
|
||||
ExecutionTarget ExecTarget() const { return exec_target; }
|
||||
|
||||
inline bool UnifiedMemory() const { return unified_memory; }
|
||||
|
||||
void* Malloc(std::size_t bytes) const
|
||||
{
|
||||
return memory_resources[0]->Allocate(bytes, 16);
|
||||
}
|
||||
|
||||
void Dealloc(void *ptr, std::size_t bytes = 0) const
|
||||
{
|
||||
memory_resources[0]->Deallocate(ptr, bytes);
|
||||
}
|
||||
|
||||
///@}
|
||||
// End: OMP specific interface
|
||||
|
||||
/**
|
||||
@name Virtual interface: finite element data structures and algorithms
|
||||
*/
|
||||
///@{
|
||||
|
||||
virtual DLayout MakeLayout(std::size_t size) const;
|
||||
virtual DLayout MakeLayout(const mfem::Array<std::size_t> &offsets) const;
|
||||
|
||||
virtual DArray MakeArray(PLayout &layout, std::size_t item_size) const;
|
||||
|
||||
virtual DVector MakeVector(PLayout &layout,
|
||||
int type_id = ScalarId<double>::value) const;
|
||||
|
||||
virtual DFiniteElementSpace MakeFESpace(mfem::FiniteElementSpace &
|
||||
fespace) const;
|
||||
|
||||
virtual DBilinearForm MakeBilinearForm(mfem::BilinearForm &bf) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual void AssembleLinearForm(LinearForm &l_form) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual mfem::Operator *MakeOperator(const MixedBilinearForm &mbl_form) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual mfem::Operator *MakeOperator(const NonlinearForm &nl_form) const;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#endif // MFEM_BACKENDS_OMP_ENGINE_HPP
|
||||
@@ -1,237 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "fespace.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
FiniteElementSpace::FiniteElementSpace(const Engine &e,
|
||||
mfem::FiniteElementSpace &fespace)
|
||||
: PFiniteElementSpace(e, fespace),
|
||||
e_layout(e, 0),
|
||||
tensor_offsets(NULL),
|
||||
tensor_indices(NULL),
|
||||
prolongation(NULL),
|
||||
restriction(NULL)
|
||||
{
|
||||
std::size_t lsize = 0;
|
||||
for (int e = 0; e < fespace.GetNE(); e++) { lsize += fespace.GetFE(e)->GetDof(); }
|
||||
e_layout.Resize(lsize);
|
||||
// The e_layout will be stored inside multiple shared DLayout objects
|
||||
e_layout.DontDelete();
|
||||
}
|
||||
|
||||
void FiniteElementSpace::BuildDofMaps()
|
||||
{
|
||||
mfem::FiniteElementSpace *mfem_fes = GetFESpace();
|
||||
|
||||
const int local_size = GetELayout().Size();
|
||||
const int global_size = mfem_fes->GetVLayout()->Size();
|
||||
const int vdim = mfem_fes->GetVDim();
|
||||
|
||||
// Now we can allocate and fill the global map
|
||||
tensor_offsets = new mfem::Array<int>(*(new Layout(OmpEngine(), global_size + 1)));
|
||||
tensor_indices = new mfem::Array<int>(*(new Layout(OmpEngine(), local_size)));
|
||||
|
||||
mfem::Array<int> &offsets = *tensor_offsets;
|
||||
mfem::Array<int> &indices = *tensor_indices;
|
||||
|
||||
mfem::Array<int> global_map(local_size);
|
||||
mfem::Array<int> elem_vdof;
|
||||
|
||||
int offset = 0;
|
||||
for (int e = 0; e < mfem_fes->GetNE(); e++)
|
||||
{
|
||||
const FiniteElement *fe = mfem_fes->GetFE(e);
|
||||
const int dofs = fe->GetDof();
|
||||
const TensorBasisElement *tfe = dynamic_cast<const TensorBasisElement *>(fe);
|
||||
const mfem::Array<int> &dof_map = tfe->GetDofMap();
|
||||
|
||||
mfem_fes->GetElementVDofs(e, elem_vdof);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
for (int i = 0; i < dofs; i++)
|
||||
{
|
||||
global_map[offset + dofs*vd + i] = elem_vdof[dofs*vd + dof_map[i]];
|
||||
}
|
||||
offset += dofs * vdim;
|
||||
}
|
||||
|
||||
// global_map[i] = index in global vector for local dof i
|
||||
// NOTE: multiple i values will yield same global_map[i] for shared DOF.
|
||||
|
||||
// We want to now invert this map so we have indices[j] = (local dof for global dof j).
|
||||
|
||||
// Zero the offset vector
|
||||
offsets = 0;
|
||||
|
||||
// Keep track of how many local dof point to its global dof
|
||||
// Count how many times each dof gets hit
|
||||
for (int i = 0; i < local_size; i++)
|
||||
{
|
||||
const int g = global_map[i];
|
||||
++offsets[g + 1];
|
||||
}
|
||||
// Aggregate the offsets
|
||||
for (int i = 1; i <= global_size; i++)
|
||||
{
|
||||
offsets[i] += offsets[i - 1];
|
||||
}
|
||||
|
||||
for (int i = 0; i < local_size; i++)
|
||||
{
|
||||
const int g = global_map[i];
|
||||
indices[offsets[g]++] = i;
|
||||
}
|
||||
|
||||
// Shift the offset vector back by one, since it was used as a
|
||||
// counter above.
|
||||
for (int i = global_size; i > 0; i--)
|
||||
{
|
||||
offsets[i] = offsets[i - 1];
|
||||
}
|
||||
offsets[0] = 0;
|
||||
|
||||
offsets.Push();
|
||||
indices.Push();
|
||||
}
|
||||
|
||||
/// Convert an E vector to L vector
|
||||
void FiniteElementSpace::ToLVector(const Vector &e_vector, Vector &l_vector)
|
||||
{
|
||||
if (tensor_indices == NULL) BuildDofMaps();
|
||||
|
||||
if (l_vector.Size() != (std::size_t) GetFESpace()->GetVSize())
|
||||
{
|
||||
l_vector.Resize<double>(GetFESpace()->GetVLayout(), NULL);
|
||||
}
|
||||
|
||||
const int lsize = l_vector.Size();
|
||||
const int *offsets = tensor_offsets->Get_PArray()->As<Array>().GetData<int>();
|
||||
const int *indices = tensor_indices->Get_PArray()->As<Array>().GetData<int>();
|
||||
|
||||
const double *e_data = e_vector.GetData<double>();
|
||||
double *l_data = l_vector.GetData<double>();
|
||||
|
||||
const bool use_target = l_vector.ComputeOnDevice();
|
||||
const bool use_parallel = (use_target || lsize > 1000);
|
||||
|
||||
#pragma omp target teams distribute parallel for \
|
||||
map (to: offsets, indices, l_data, e_data) \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel)
|
||||
for (int i = 0; i < lsize; i++)
|
||||
{
|
||||
const int offset = offsets[i];
|
||||
const int next_offset = offsets[i + 1];
|
||||
double dof_value = 0;
|
||||
for (int j = offset; j < next_offset; j++)
|
||||
{
|
||||
dof_value += e_data[indices[j]];
|
||||
}
|
||||
l_data[i] = dof_value;
|
||||
}
|
||||
}
|
||||
|
||||
/// Covert an L vector to E vector
|
||||
void FiniteElementSpace::ToEVector(const Vector &l_vector, Vector &e_vector)
|
||||
{
|
||||
if (tensor_indices == NULL) BuildDofMaps();
|
||||
|
||||
if (e_vector.Size() != (std::size_t) e_layout.Size())
|
||||
{
|
||||
e_vector.Resize<double>(GetELayout(), NULL);
|
||||
}
|
||||
|
||||
const int lsize = l_vector.Size();
|
||||
const int *offsets = tensor_offsets->Get_PArray()->As<Array>().GetData<int>();
|
||||
const int *indices = tensor_indices->Get_PArray()->As<Array>().GetData<int>();
|
||||
|
||||
const double *l_data = l_vector.GetData<double>();
|
||||
double *e_data = e_vector.GetData<double>();
|
||||
|
||||
const bool use_target = l_vector.ComputeOnDevice();
|
||||
const bool use_parallel = (use_target || lsize > 1000);
|
||||
|
||||
#pragma omp target teams distribute parallel for \
|
||||
map (to: offsets, indices, l_data, e_data) \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel)
|
||||
for (int i = 0; i < lsize; i++)
|
||||
{
|
||||
const int offset = offsets[i];
|
||||
const int next_offset = offsets[i + 1];
|
||||
const double dof_value = l_data[i];
|
||||
for (int j = offset; j < next_offset; j++)
|
||||
{
|
||||
e_data[indices[j]] = dof_value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Get the finite element space prolongation matrix
|
||||
const Operator *FiniteElementSpace::GetProlongation() const
|
||||
{
|
||||
// FIXME: This relies on unified memory if using a device other than the CPU
|
||||
if (!prolongation)
|
||||
{
|
||||
Layout &v_layout = GetVLayout();
|
||||
Layout &t_layout = GetTrueVLayout();
|
||||
|
||||
const mfem::Operator *op = GetFESpace()->GetProlongationMatrix();
|
||||
if (!op)
|
||||
{
|
||||
prolongation = new mfem::IdentityOperator(t_layout);
|
||||
}
|
||||
else
|
||||
{
|
||||
prolongation = new BackendOperator(t_layout, v_layout, op);
|
||||
}
|
||||
}
|
||||
return prolongation;
|
||||
}
|
||||
|
||||
/// Get the finite element space restriction matrix
|
||||
const Operator *FiniteElementSpace::GetRestriction() const
|
||||
{
|
||||
// FIXME: This relies on unified memory if using a device other than the CPU
|
||||
if (!restriction)
|
||||
{
|
||||
Layout &v_layout = GetVLayout();
|
||||
Layout &t_layout = GetTrueVLayout();
|
||||
|
||||
const mfem::Operator *op = GetFESpace()->GetRestrictionMatrix();
|
||||
if (!op)
|
||||
{
|
||||
restriction = new mfem::IdentityOperator(t_layout);
|
||||
}
|
||||
else
|
||||
{
|
||||
restriction = new BackendOperator(v_layout, t_layout, op);
|
||||
}
|
||||
}
|
||||
return restriction;
|
||||
}
|
||||
|
||||
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
@@ -1,106 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OMP_FESPACE_HPP
|
||||
#define MFEM_BACKENDS_OMP_FESPACE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "array.hpp"
|
||||
#include "vector.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
/*
|
||||
Wraps an mfem::Operator that does not contain layout information.
|
||||
*/
|
||||
class BackendOperator : public mfem::Operator
|
||||
{
|
||||
const mfem::Operator *op;
|
||||
|
||||
public:
|
||||
BackendOperator(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::Operator *op_) : Operator(in_layout, out_layout), op(op_) { }
|
||||
|
||||
virtual void Mult(const mfem::Vector &x, mfem::Vector &y) const { op->Mult(x, y); }
|
||||
virtual void MultTranspose(const mfem::Vector &x, mfem::Vector &y) const { op->MultTranspose(x, y); }
|
||||
};
|
||||
|
||||
/// TODO: doxygen
|
||||
class FiniteElementSpace : public mfem::PFiniteElementSpace
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// mfem::FiniteElementSpace *fes;
|
||||
|
||||
Layout e_layout;
|
||||
|
||||
mfem::Array<int> *tensor_offsets, *tensor_indices;
|
||||
|
||||
mutable mfem::Operator *prolongation, *restriction;
|
||||
|
||||
void BuildDofMaps();
|
||||
|
||||
public:
|
||||
/// Nearly-empty class that stores a pointer to a mfem::FiniteElementSpace instance and the engine
|
||||
FiniteElementSpace(const Engine &e, mfem::FiniteElementSpace &fespace);
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~FiniteElementSpace()
|
||||
{
|
||||
delete tensor_offsets;
|
||||
delete tensor_indices;
|
||||
delete prolongation;
|
||||
delete restriction;
|
||||
}
|
||||
|
||||
Layout &GetELayout() { return e_layout; }
|
||||
|
||||
Layout &GetVLayout() const
|
||||
{ return *fes->GetVLayout().As<Layout>(); }
|
||||
|
||||
Layout &GetTrueVLayout() const
|
||||
{ return *fes->GetTrueVLayout().As<Layout>(); }
|
||||
|
||||
/// Return the engine as an OpenMP engine
|
||||
const Engine &OmpEngine() { return static_cast<const Engine&>(*engine); }
|
||||
|
||||
/// Convert an E vector to L vector
|
||||
void ToLVector(const Vector &e_vector, Vector &l_vector);
|
||||
|
||||
/// Covert an L vector to E vector
|
||||
void ToEVector(const Vector &l_vector, Vector &e_vector);
|
||||
|
||||
/// Get the finite element space prolongation matrix
|
||||
const Operator *GetProlongation() const;
|
||||
|
||||
/// Get the finite element space restriction matrix
|
||||
const Operator *GetRestriction() const;
|
||||
|
||||
};
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#endif // MFEM_BACKENDS_OMP_FESPACE_HPP
|
||||
@@ -1,40 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "layout.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
void Layout::Resize(std::size_t new_size)
|
||||
{
|
||||
size = new_size;
|
||||
}
|
||||
|
||||
void Layout::Resize(const Array<std::size_t> &offsets)
|
||||
{
|
||||
MFEM_ASSERT(offsets.Size() == 2,
|
||||
"multiple workers are not supported yet");
|
||||
size = offsets.Last();
|
||||
}
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
@@ -1,71 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OMP_LAYOUT_HPP
|
||||
#define MFEM_BACKENDS_OMP_LAYOUT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "../base/layout.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
class Layout : public mfem::PLayout
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// std::size_t size;
|
||||
|
||||
public:
|
||||
Layout(const Engine &e, std::size_t s = 0) : PLayout(e, s) { }
|
||||
|
||||
const Engine &OmpEngine() const
|
||||
{ return *static_cast<const Engine *>(engine.Get()); }
|
||||
|
||||
void *Alloc(std::size_t bytes) const
|
||||
{ return OmpEngine().Malloc(bytes); }
|
||||
|
||||
void Dealloc(void *ptr) const
|
||||
{ return OmpEngine().Dealloc(ptr); }
|
||||
|
||||
virtual ~Layout() { }
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// Resize the layout
|
||||
virtual void Resize(std::size_t new_size);
|
||||
|
||||
/// Resize the layout based on the given worker offsets
|
||||
virtual void Resize(const Array<std::size_t> &offsets);
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#endif // MFEM_BACKENDS_OMP_LAYOUT_HPP
|
||||
@@ -1,57 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "memory_resource.hpp"
|
||||
#include "../../general/error.hpp"
|
||||
|
||||
#ifdef MFEM_USE_CUDAUM
|
||||
#include "cuda_runtime.h"
|
||||
#include "cuda.h"
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
#ifdef MFEM_USE_CUDAUM
|
||||
void *UnifiedMemoryResource::DoAllocate(std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
void *p = NULL;
|
||||
if (bytes > 0)
|
||||
{
|
||||
cudaError_t ret = cudaMallocManaged(&p, bytes);
|
||||
MFEM_VERIFY(ret == cudaSuccess, "");
|
||||
}
|
||||
return p;
|
||||
}
|
||||
|
||||
void UnifiedMemoryResource::DoDeallocate(void *p, std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
if (p != NULL)
|
||||
{
|
||||
cudaFree(p);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
@@ -1,44 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OMP_MEMORY_RESOURCE_HPP
|
||||
#define MFEM_BACKENDS_OMP_MEMORY_RESOURCE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "../../backends/base/memory_resource.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
/// Polymorphic memory resource. Similar to C++17's std::pmr::memory_resource.
|
||||
|
||||
#ifdef MFEM_USE_CUDAUM
|
||||
/** @brief Memory resource using unified memory. */
|
||||
class UnifiedMemoryResource : public MemoryResource
|
||||
{
|
||||
protected:
|
||||
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment);
|
||||
virtual void DoDeallocate(void *p, std::size_t bytes, std::size_t alignment);
|
||||
};
|
||||
#endif
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#endif // MFEM_BACKENDS_OMP_MEMORY_RESOURCE_HPP
|
||||
@@ -1,205 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "../../linalg/vector.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
PVector *Vector::DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const
|
||||
{
|
||||
MFEM_ASSERT(buffer_type_id == ScalarId<double>::value, "");
|
||||
Vector *new_vector = new Vector(OmpLayout());
|
||||
if (copy_data)
|
||||
{
|
||||
const std::size_t total_size = sizeof(double) * OmpLayout().Size();
|
||||
if (!ComputeOnDevice())
|
||||
std::memcpy(new_vector->GetData<void>(), data, total_size);
|
||||
else
|
||||
{
|
||||
char *new_data = new_vector->GetData<char>();
|
||||
#pragma omp target teams distribute parallel for is_device_ptr(new_data)
|
||||
for (std::size_t i = 0; i < total_size; i++) new_data[i] = data[i];
|
||||
}
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
*buffer = new_vector->GetData<void>();
|
||||
}
|
||||
return new_vector;
|
||||
}
|
||||
|
||||
void Vector::DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const
|
||||
{
|
||||
// Can be called when Size() == 0, e.g. when an MPI-parallel vector has a
|
||||
// local size of 0.
|
||||
|
||||
MFEM_ASSERT(result_type_id == ScalarId<double>::value, "");
|
||||
double *res = (double *)result;
|
||||
double local_dot = 0.;
|
||||
MFEM_ASSERT(dynamic_cast<const Vector *>(&x) != NULL, "invalid Vector type");
|
||||
const Vector *xp = static_cast<const Vector *>(&x);
|
||||
MFEM_ASSERT(this->Size() == xp->Size(), "");
|
||||
|
||||
const double *ptr = GetData<double>();
|
||||
const double *xptr = xp->GetData<double>();
|
||||
const std::size_t size = Size();
|
||||
|
||||
if (!ComputeOnDevice())
|
||||
{
|
||||
for (std::size_t i = 0; i < size; i++) local_dot += ptr[i] * xptr[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
#pragma omp target teams distribute parallel for map(to: ptr, xptr) reduction(+:local_dot)
|
||||
for (std::size_t i = 0; i < size; i++) local_dot += ptr[i] * xptr[i];
|
||||
}
|
||||
|
||||
*res = local_dot;
|
||||
#ifdef MFEM_USE_MPI
|
||||
MPI_Comm comm = OmpLayout().OmpEngine().GetComm();
|
||||
if (comm != MPI_COMM_NULL)
|
||||
{
|
||||
MPI_Allreduce(&local_dot, res, 1, MPI_DOUBLE, MPI_SUM, comm);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void Vector::DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
MFEM_ASSERT(ab_type_id == ScalarId<double>::value, "");
|
||||
|
||||
const double da = *static_cast<const double *>(a);
|
||||
const double db = *static_cast<const double *>(b);
|
||||
MFEM_ASSERT(da == 0.0 || dynamic_cast<const Vector *>(&x) != NULL,
|
||||
"invalid Vector x");
|
||||
MFEM_ASSERT(db == 0.0 || dynamic_cast<const Vector *>(&y) != NULL,
|
||||
"invalid Vector y");
|
||||
const Vector *xp = static_cast<const Vector *>(&x);
|
||||
const Vector *yp = static_cast<const Vector *>(&y);
|
||||
|
||||
MFEM_ASSERT(da == 0.0 || this->Size() == xp->Size(), "");
|
||||
MFEM_ASSERT(db == 0.0 || this->Size() == yp->Size(), "");
|
||||
|
||||
const std::size_t size = Size();
|
||||
const std::size_t critical_size = 1000;
|
||||
const double *xd = xp->GetData<double>();
|
||||
const double *yd = yp->GetData<double>();
|
||||
double *td = GetData<double>();
|
||||
|
||||
const bool use_target = ComputeOnDevice();
|
||||
const bool use_parallel = (use_target || size > critical_size);
|
||||
|
||||
if (da == 0.0)
|
||||
{
|
||||
if (db == 0.0)
|
||||
{
|
||||
OmpFill(&da);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (td == yd)
|
||||
{
|
||||
// *this *= db
|
||||
#pragma omp target teams distribute parallel for \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel) map (to: db)
|
||||
for (std::size_t i = 0; i < size; i++) td[i] *= db;
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = db * y
|
||||
#pragma omp target teams distribute parallel for \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel) map (to: yd, db)
|
||||
for (std::size_t i = 0; i < size; i++) td[i] = yd[i] * db;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (db == 0.0)
|
||||
{
|
||||
if (td == xd)
|
||||
{
|
||||
// *this *= da
|
||||
#pragma omp target teams distribute parallel for \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel) map (to: da)
|
||||
for (std::size_t i = 0; i < size; i++) td[i] *= da;
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = da * x
|
||||
#pragma omp target teams distribute parallel for \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel) map (to: xd, da)
|
||||
for (std::size_t i = 0; i < size; i++) td[i] = xd[i] * da;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ASSERT(xd != yd, "invalid input");
|
||||
if (td == xd)
|
||||
{
|
||||
// *this = da * (*this) + db * y
|
||||
#pragma omp target teams distribute parallel for \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel) map (to: da, td, db, yd)
|
||||
for (std::size_t i = 0; i < size; i++) td[i] = da * td[i] + db * yd[i];
|
||||
}
|
||||
else if (td == yd)
|
||||
{
|
||||
// *this = da * x + db * (*this)
|
||||
#pragma omp target teams distribute parallel for \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel) map (to: da, xd, db, td)
|
||||
for (std::size_t i = 0; i < size; i++) td[i] = da * xd[i] + db * td[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = da * x + db * y
|
||||
#pragma omp target teams distribute parallel for \
|
||||
if (target: use_target) \
|
||||
if (parallel: use_parallel) map (to: da, xd, db, yd)
|
||||
for (std::size_t i = 0; i < size; i++) td[i] = da * xd[i] + db * yd[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mfem::Vector Vector::Wrap()
|
||||
{
|
||||
return mfem::Vector(*this);
|
||||
}
|
||||
|
||||
const mfem::Vector Vector::Wrap() const
|
||||
{
|
||||
return mfem::Vector(*const_cast<Vector*>(this));
|
||||
}
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
@@ -1,71 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OMP_VECTOR_HPP
|
||||
#define MFEM_BACKENDS_OMP_VECTOR_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#include "../base/vector.hpp"
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace omp
|
||||
{
|
||||
|
||||
class Vector : virtual public Array, public mfem::PVector
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// DLayout layout;
|
||||
// char *data;
|
||||
// std::size_t size;
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
virtual PVector *DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const;
|
||||
|
||||
virtual void DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const;
|
||||
|
||||
virtual void DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id);
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
public:
|
||||
Vector(Layout <)
|
||||
: PArray(lt), Array(lt, sizeof(double)), PVector(lt)
|
||||
{ }
|
||||
|
||||
mfem::Vector Wrap();
|
||||
|
||||
const mfem::Vector Wrap() const;
|
||||
};
|
||||
|
||||
} // namespace mfem::omp
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
|
||||
|
||||
#endif // MFEM_BACKENDS_OMP_VECTOR_HPP
|
||||
@@ -1,182 +0,0 @@
|
||||
##################################################################################
|
||||
#
|
||||
# Set defaults for XSDK CMake projects
|
||||
#
|
||||
##################################################################################
|
||||
|
||||
#
|
||||
# This module implements standard behavior for XSDK CMake projects. The main
|
||||
# thing it does in XSDK mode (i.e. USE_XSDK_DEFAULTS=TRUE) is to print out
|
||||
# when the env vars CC, CXX, FC and compiler flags CFLAGS, CXXFLAGS, and
|
||||
# FFLAGS/FCFLAGS are used to select the compilers and compiler flags (raw
|
||||
# CMake does this silently) and to set BUILD_SHARED_LIBS=TRUE and
|
||||
# CMAKE_BUILD_TYPE=DEBUG by default. It does not implement *all* of the
|
||||
# standard XSDK configuration parameters. The parent CMake project must do
|
||||
# that.
|
||||
#
|
||||
# Note that when USE_XSDK_DEFAULTS=TRUE, then the Fortran flags will be read
|
||||
# from either of the env vars FFLAGS or FCFLAGS. If both are set, but are the
|
||||
# same, then FFLAGS it used (which is the same as FCFLAGS). However, if both
|
||||
# are set but are not equal, then a FATAL_ERROR is raised and CMake configure
|
||||
# processing is stopped.
|
||||
#
|
||||
# To be used in a parent project, this module must be included after
|
||||
#
|
||||
# PROJECT(${PROJECT_NAME} NONE)
|
||||
#
|
||||
# is called but before the compilers are defined and processed using:
|
||||
#
|
||||
# ENABLE_LANGUAGE(<LANG>)
|
||||
#
|
||||
# For example, one would do:
|
||||
#
|
||||
# PROJECT(${PROJECT_NAME} NONE)
|
||||
# ...
|
||||
# SET(USE_XSDK_DEFAULTS_DEFAULT TRUE) # Set to false if desired
|
||||
# INCLUDE("${CMAKE_CURRENT_SOURCE_DIR}/stdk/XSDKDefaults.cmake")
|
||||
# ...
|
||||
# ENABLE_LANGUAGE(C)
|
||||
# ENABLE_LANGUAGE(C++)
|
||||
# ENABLE_LANGUAGE(Fortran)
|
||||
#
|
||||
# The variable `USE_XSDK_DEFAULTS_DEFAULT` is used as the default for the
|
||||
# cache var `USE_XSDK_DEFAULTS`. That way, a project can decide if it wants
|
||||
# XSDK defaults turned on or off by default and users can independently decide
|
||||
# if they want the CMake project to use standard XSDK behavior or raw CMake
|
||||
# behavior.
|
||||
#
|
||||
# By default, the XSDKDefaults.cmake module assumes that the project will need
|
||||
# C, C++, and Fortran. If any language is not needed then, set
|
||||
# XSDK_ENABLE_C=OFF, XSDK_ENABLE_CXX=OFF, or XSDK_ENABLE_Fortran=OFF *before*
|
||||
# including this module. Note, these variables are *not* cache vars because a
|
||||
# project either does or does not have C, C++ or Fortran source files, the
|
||||
# user has nothing to do with this so there is no need for cache vars. The
|
||||
# parent CMake project just needs to tell XSDKDefault.cmake what languages is
|
||||
# needs or does not need.
|
||||
#
|
||||
# For example, if the parent CMake project only needs C, then it would do:
|
||||
#
|
||||
# PROJECT(${PROJECT_NAME} NONE)'
|
||||
# ...
|
||||
# SET(USE_XSDK_DEFAULTS_DEFAULT TRUE)
|
||||
# SET(XSDK_ENABLE_CXX OFF)
|
||||
# SET(XSDK_ENABLE_Fortran OFF)
|
||||
# INCLUDE("${CMAKE_CURRENT_SOURCE_DIR}/stdk/XSDKDefaults.cmake")
|
||||
# ...
|
||||
# ENABLE_LANGAUGE(C)
|
||||
#
|
||||
# This module code will announce when it sets any variables.
|
||||
#
|
||||
|
||||
#
|
||||
# Helper functions
|
||||
#
|
||||
|
||||
IF (NOT COMMAND PRINT_VAR)
|
||||
FUNCTION(PRINT_VAR VAR_NAME)
|
||||
MESSAGE("-- " "${VAR_NAME} = '${${VAR_NAME}}'")
|
||||
ENDFUNCTION()
|
||||
ENDIF()
|
||||
|
||||
IF (NOT COMMAND SET_DEFAULT)
|
||||
MACRO(SET_DEFAULT VAR)
|
||||
IF ("${${VAR}}" STREQUAL "")
|
||||
SET(${VAR} ${ARGN})
|
||||
ENDIF()
|
||||
ENDMACRO()
|
||||
ENDIF()
|
||||
|
||||
#
|
||||
# XSDKDefaults.cmake control variables
|
||||
#
|
||||
|
||||
# USE_XSDK_DEFAULTS
|
||||
IF ("${USE_XSDK_DEFAULTS_DEFAULT}" STREQUAL "")
|
||||
SET(USE_XSDK_DEFAULTS_DEFAULT FALSE)
|
||||
ENDIF()
|
||||
SET(USE_XSDK_DEFAULTS ${USE_XSDK_DEFAULTS_DEFAULT} CACHE BOOL
|
||||
"Use XSDK defaults and behavior.")
|
||||
PRINT_VAR(USE_XSDK_DEFAULTS)
|
||||
|
||||
SET_DEFAULT(XSDK_ENABLE_C TRUE)
|
||||
SET_DEFAULT(XSDK_ENABLE_CXX TRUE)
|
||||
SET_DEFAULT(XSDK_ENABLE_Fortran TRUE)
|
||||
|
||||
# Handle the compiler and flags for a language
|
||||
MACRO(XSDK_HANDLE_LANG_DEFAULTS CMAKE_LANG_NAME ENV_LANG_NAME
|
||||
ENV_LANG_FLAGS_NAMES
|
||||
)
|
||||
|
||||
# Announce using env var ${ENV_LANG_NAME}
|
||||
IF (NOT "$ENV{${ENV_LANG_NAME}}" STREQUAL "" AND
|
||||
"${CMAKE_${CMAKE_LANG_NAME}_COMPILER}" STREQUAL ""
|
||||
)
|
||||
MESSAGE("-- " "XSDK: Setting CMAKE_${CMAKE_LANG_NAME}_COMPILER from env var"
|
||||
" ${ENV_LANG_NAME}='$ENV{${ENV_LANG_NAME}}'!")
|
||||
SET(CMAKE_${CMAKE_LANG_NAME}_COMPILER "$ENV{${ENV_LANG_NAME}}" CACHE FILEPATH
|
||||
"XSDK: Set by default from env var ${ENV_LANG_NAME}")
|
||||
ENDIF()
|
||||
|
||||
# Announce using env var ${ENV_LANG_FLAGS_NAME}
|
||||
FOREACH(ENV_LANG_FLAGS_NAME ${ENV_LANG_FLAGS_NAMES})
|
||||
IF (NOT "$ENV{${ENV_LANG_FLAGS_NAME}}" STREQUAL "" AND
|
||||
"${CMAKE_${CMAKE_LANG_NAME}_FLAGS}" STREQUAL ""
|
||||
)
|
||||
MESSAGE("-- " "XSDK: Setting CMAKE_${CMAKE_LANG_NAME}_FLAGS from env var"
|
||||
" ${ENV_LANG_FLAGS_NAME}='$ENV{${ENV_LANG_FLAGS_NAME}}'!")
|
||||
SET(CMAKE_${CMAKE_LANG_NAME}_FLAGS "$ENV{${ENV_LANG_FLAGS_NAME}} " CACHE STRING
|
||||
"XSDK: Set by default from env var ${ENV_LANG_FLAGS_NAME}")
|
||||
# NOTE: CMake adds the space after $ENV{${ENV_LANG_FLAGS_NAME}} so we
|
||||
# duplicate that here!
|
||||
ENDIF()
|
||||
ENDFOREACH()
|
||||
|
||||
ENDMACRO()
|
||||
|
||||
|
||||
#
|
||||
# Set XSDK Defaults
|
||||
#
|
||||
|
||||
# Set default compilers and flags
|
||||
IF (USE_XSDK_DEFAULTS)
|
||||
|
||||
# Handle env vars for languages C, C++, and Fortran
|
||||
|
||||
IF (XSDK_ENABLE_C)
|
||||
XSDK_HANDLE_LANG_DEFAULTS(C CC CFLAGS)
|
||||
ENDIF()
|
||||
|
||||
IF (XSDK_ENABLE_CXX)
|
||||
XSDK_HANDLE_LANG_DEFAULTS(CXX CXX CXXFLAGS)
|
||||
ENDIF()
|
||||
|
||||
IF (XSDK_ENABLE_Fortran)
|
||||
SET(ENV_FFLAGS "$ENV{FFLAGS}")
|
||||
SET(ENV_FCFLAGS "$ENV{FCFLAGS}")
|
||||
IF (
|
||||
(NOT "${ENV_FFLAGS}" STREQUAL "") AND (NOT "${ENV_FCFLAGS}" STREQUAL "")
|
||||
AND
|
||||
("${CMAKE_Fortran_FLAGS}" STREQUAL "")
|
||||
)
|
||||
IF (NOT "${ENV_FFLAGS}" STREQUAL "${ENV_FCFLAGS}")
|
||||
MESSAGE(FATAL_ERROR "Error, env vars FFLAGS='${ENV_FFLAGS}' and"
|
||||
" FCFLAGS='${ENV_FCFLAGS}' are both set in the env but are not equal!")
|
||||
ENDIF()
|
||||
ENDIF()
|
||||
XSDK_HANDLE_LANG_DEFAULTS(Fortran FC "FFLAGS;FCFLAGS")
|
||||
ENDIF()
|
||||
|
||||
# Set XSDK defaults for other CMake variables
|
||||
|
||||
IF ("${BUILD_SHARED_LIBS}" STREQUAL "")
|
||||
MESSAGE("-- " "XSDK: Setting default BUILD_SHARED_LIBS=TRUE")
|
||||
SET(BUILD_SHARED_LIBS TRUE CACHE BOOL "Set by default in XSDK mode")
|
||||
ENDIF()
|
||||
|
||||
IF ("${CMAKE_BUILD_TYPE}" STREQUAL "")
|
||||
MESSAGE("-- " "XSDK: Setting default CMAKE_BUILD_TYPE=DEBUG")
|
||||
SET(CMAKE_BUILD_TYPE DEBUG CACHE STRING "Set by default in XSDK mode")
|
||||
ENDIF()
|
||||
|
||||
ENDIF()
|
||||
@@ -12,41 +12,31 @@
|
||||
include(${CMAKE_CURRENT_LIST_DIR}/MFEMConfigVersion.cmake)
|
||||
|
||||
set(MFEM_VERSION ${PACKAGE_VERSION})
|
||||
set(MFEM_VERSION_INT @MFEM_VERSION@)
|
||||
set(MFEM_GIT_STRING "@MFEM_GIT_STRING@")
|
||||
|
||||
set(MFEM_USE_MPI @MFEM_USE_MPI@)
|
||||
set(MFEM_USE_METIS @MFEM_USE_METIS@)
|
||||
set(MFEM_USE_METIS_5 @MFEM_USE_METIS_5@)
|
||||
set(MFEM_DEBUG @MFEM_DEBUG@)
|
||||
set(MFEM_USE_EXCEPTIONS @MFEM_USE_EXCEPTIONS@)
|
||||
set(MFEM_USE_GZSTREAM @MFEM_USE_GZSTREAM@)
|
||||
set(MFEM_USE_LIBUNWIND @MFEM_USE_LIBUNWIND@)
|
||||
set(MFEM_USE_LAPACK @MFEM_USE_LAPACK@)
|
||||
set(MFEM_THREAD_SAFE @MFEM_THREAD_SAFE@)
|
||||
set(MFEM_USE_OPENMP @MFEM_USE_OPENMP@)
|
||||
set(MFEM_USE_MEMALLOC @MFEM_USE_MEMALLOC@)
|
||||
set(MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@)
|
||||
set(MFEM_USE_SUNDIALS @MFEM_USE_SUNDIALS@)
|
||||
set(MFEM_USE_MESQUITE @MFEM_USE_MESQUITE@)
|
||||
set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
|
||||
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
|
||||
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
|
||||
set(MFEM_USE_GECKO @MFEM_USE_GECKO@)
|
||||
set(MFEM_USE_GNUTLS @MFEM_USE_GNUTLS@)
|
||||
set(MFEM_USE_NETCDF @MFEM_USE_NETCDF@)
|
||||
set(MFEM_USE_PETSC @MFEM_USE_PETSC@)
|
||||
set(MFEM_USE_MPFR @MFEM_USE_MPFR@)
|
||||
set(MFEM_USE_SIDRE @MFEM_USE_SIDRE@)
|
||||
set(MFEM_USE_CONDUIT @MFEM_USE_CONDUIT@)
|
||||
|
||||
set(MFEM_CXX_COMPILER "@CMAKE_CXX_COMPILER@")
|
||||
set(MFEM_CXX_FLAGS "@CMAKE_CXX_FLAGS@")
|
||||
|
||||
@PACKAGE_INIT@
|
||||
|
||||
set(MFEM_INCLUDE_DIRS "@PACKAGE_INCLUDE_INSTALL_DIRS@")
|
||||
foreach (dir ${MFEM_INCLUDE_DIRS})
|
||||
message("DIR = ${dir}")
|
||||
|
||||
set_and_check(MFEM_INCLUDE_DIR "${dir}")
|
||||
endforeach (dir "${MFEM_INCLUDE_DIRS}")
|
||||
|
||||
|
||||
@@ -12,27 +12,6 @@
|
||||
#ifndef MFEM_CONFIG_HEADER
|
||||
#define MFEM_CONFIG_HEADER
|
||||
|
||||
// MFEM version: integer of the form: (major*100 + minor)*100 + patch.
|
||||
#cmakedefine MFEM_VERSION @MFEM_VERSION@
|
||||
|
||||
// MFEM version string of the form "3.3" or "3.3.1".
|
||||
#cmakedefine MFEM_VERSION_STRING "@MFEM_VERSION_STRING@"
|
||||
|
||||
// MFEM version type, see the MFEM_VERSION_TYPE_* constants below.
|
||||
#define MFEM_VERSION_TYPE ((MFEM_VERSION)%2)
|
||||
|
||||
// MFEM version type constants.
|
||||
#define MFEM_VERSION_TYPE_RELEASE 0
|
||||
#define MFEM_VERSION_TYPE_DEVELOPMENT 1
|
||||
|
||||
// Separate MFEM version numbers for major, minor, and patch.
|
||||
#define MFEM_VERSION_MAJOR ((MFEM_VERSION)/10000)
|
||||
#define MFEM_VERSION_MINOR (((MFEM_VERSION)/100)%100)
|
||||
#define MFEM_VERSION_PATCH ((MFEM_VERSION)%100)
|
||||
|
||||
// Description of the git commit used to build MFEM.
|
||||
#cmakedefine MFEM_GIT_STRING "@MFEM_GIT_STRING@"
|
||||
|
||||
// Build the parallel MFEM library.
|
||||
// Requires an MPI compiler, and the libraries HYPRE and METIS.
|
||||
#cmakedefine MFEM_USE_MPI
|
||||
@@ -40,18 +19,12 @@
|
||||
// Enable debug checks in MFEM.
|
||||
#cmakedefine MFEM_DEBUG
|
||||
|
||||
// Throw an exception on errors.
|
||||
#cmakedefine MFEM_USE_EXCEPTIONS
|
||||
|
||||
// Enable gzstream in MFEM.
|
||||
#cmakedefine MFEM_USE_GZSTREAM
|
||||
|
||||
// Enable backtraces for mfem_error through libunwind.
|
||||
#cmakedefine MFEM_USE_LIBUNWIND
|
||||
|
||||
// Enable MFEM features that use the METIS library (parallel MFEM).
|
||||
#cmakedefine MFEM_USE_METIS
|
||||
|
||||
// Enable this option if linking with METIS version 5 (parallel MFEM).
|
||||
#cmakedefine MFEM_USE_METIS_5
|
||||
|
||||
@@ -74,9 +47,6 @@
|
||||
// Enable MFEM functionality based on the SuperLU_DIST library.
|
||||
#cmakedefine MFEM_USE_SUPERLU
|
||||
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
#cmakedefine MFEM_USE_STRUMPACK
|
||||
|
||||
// Internal MFEM option: enable group/batch allocation for some small objects.
|
||||
#cmakedefine MFEM_USE_MEMALLOC
|
||||
|
||||
@@ -95,11 +65,12 @@
|
||||
// Enable MFEM functionality based on the Sidre library
|
||||
#cmakedefine MFEM_USE_SIDRE
|
||||
|
||||
// Enable MFEM functionality based on Conduit
|
||||
#cmakedefine MFEM_USE_CONDUIT
|
||||
|
||||
// Which library functions to use in class StopWatch for measuring time.
|
||||
// For a list of the available options, see INSTALL.
|
||||
// The available options are:
|
||||
// 0 - use std::clock from <ctime>
|
||||
// 1 - use times from <sys/times.h>
|
||||
// 2 - use high-resolution POSIX clocks
|
||||
// 3 - use QueryPerformanceCounter from <windows.h>
|
||||
// If not defined, an option is selected automatically.
|
||||
#define MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@
|
||||
|
||||
@@ -110,7 +81,4 @@
|
||||
// Macro needed to get defines like M_PI from <cmath>. (Visual Studio C++ only?)
|
||||
#cmakedefine _USE_MATH_DEFINES
|
||||
|
||||
// Version of HYPRE used for building MFEM.
|
||||
#cmakedefine MFEM_HYPRE_VERSION @MFEM_HYPRE_VERSION@
|
||||
|
||||
#endif // MFEM_CONFIG_HEADER
|
||||
|
||||
@@ -10,14 +10,15 @@
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Defines the following variables:
|
||||
# - AXOM_FOUND
|
||||
# - AXOM_LIBRARIES
|
||||
# - AXOM_INCLUDE_DIRS
|
||||
# - ATK_FOUND
|
||||
# - ATK_LIBRARIES
|
||||
# - ATK_INCLUDE_DIRS
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
# Note: components are enabled based on the find_package() parameters.
|
||||
mfem_find_package(Axom AXOM AXOM_DIR "include" "" "lib" ""
|
||||
"Paths to headers required by Axom." "Libraries required by Axom."
|
||||
mfem_find_package(ATK ATK ATK_DIR "include" "" "lib" ""
|
||||
"Paths to headers required by ATK." "Libraries required by ATK."
|
||||
ADD_COMPONENT Sidre "include" sidre/sidre.hpp "lib" sidre
|
||||
ADD_COMPONENT SPIO "include" spio/IOManager.hpp "lib" spio
|
||||
ADD_COMPONENT SLIC "include" slic/slic.hpp "lib" slic
|
||||
ADD_COMPONENT axom_utils "include" axom_utils/Utilities.hpp "lib" axom_utils)
|
||||
ADD_COMPONENT common "include" common/ATKMacros.hpp "lib" common)
|
||||
@@ -14,22 +14,9 @@
|
||||
# - CONDUIT_LIBRARIES
|
||||
# - CONDUIT_INCLUDE_DIRS
|
||||
|
||||
# check to see if relay requires hdf5, if so make sure to set HDF5
|
||||
# as a required dep
|
||||
if(EXISTS ${CONDUIT_DIR}/include/conduit/conduit_relay_hdf5.hpp)
|
||||
message(STATUS "Conduit Relay HDF5 Support is ENABLED")
|
||||
# we only need HDF5 if Conduit was built with HDF5 support
|
||||
set(Conduit_REQUIRED_PACKAGES "HDF5" CACHE STRING
|
||||
"Additional packages required by Conduit.")
|
||||
else()
|
||||
message(STATUS "Conduit Relay HDF5 Support is DISABLED")
|
||||
endif()
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(Conduit CONDUIT CONDUIT_DIR
|
||||
"include;include/conduit" conduit.hpp "lib" conduit
|
||||
"Paths to headers required by Conduit." "Libraries required by Conduit."
|
||||
ADD_COMPONENT relay
|
||||
"include;include/conduit" conduit_relay.hpp "lib" conduit_relay
|
||||
ADD_COMPONENT blueprint
|
||||
"include;include/conduit" conduit_blueprint.hpp "lib" conduit_blueprint)
|
||||
"include;include/conduit" conduit_relay.hpp "lib" conduit_relay)
|
||||
|
||||
@@ -13,23 +13,7 @@
|
||||
# - HYPRE_FOUND
|
||||
# - HYPRE_LIBRARIES
|
||||
# - HYPRE_INCLUDE_DIRS
|
||||
# - HYPRE_VERSION
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(HYPRE HYPRE HYPRE_DIR "include" "HYPRE.h" "lib" "HYPRE"
|
||||
"Paths to headers required by HYPRE." "Libraries required by HYPRE.")
|
||||
|
||||
if (HYPRE_FOUND AND (NOT HYPRE_VERSION))
|
||||
try_run(HYPRE_VERSION_RUN_RESULT HYPRE_VERSION_COMPILE_RESULT
|
||||
${CMAKE_CURRENT_BINARY_DIR}/config
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/config/get_hypre_version.cpp
|
||||
CMAKE_FLAGS -DINCLUDE_DIRECTORIES:STRING=${HYPRE_INCLUDE_DIRS}
|
||||
RUN_OUTPUT_VARIABLE HYPRE_VERSION_OUTPUT)
|
||||
if ((HYPRE_VERSION_RUN_RESULT EQUAL 0) AND HYPRE_VERSION_OUTPUT)
|
||||
string(STRIP "${HYPRE_VERSION_OUTPUT}" HYPRE_VERSION)
|
||||
set(HYPRE_VERSION ${HYPRE_VERSION} CACHE STRING "HYPRE version." FORCE)
|
||||
message(STATUS "Found HYPRE version ${HYPRE_VERSION}")
|
||||
else()
|
||||
message(FATAL_ERROR "Unable to determine HYPRE version.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -1,36 +0,0 @@
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Sets the following variables:
|
||||
# - STRUMPACK_FOUND
|
||||
# - STRUMPACK_INCLUDE_DIRS
|
||||
# - STRUMPACK_LIBRARIES
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(STRUMPACK STRUMPACK STRUMPACK_DIR
|
||||
"include" "StrumpackSparseSolverMPIDist.hpp"
|
||||
"lib" "strumpack;strumpack_sparse" # add NAMES_PER_DIR?
|
||||
"Paths to headers required by STRUMPACK."
|
||||
"Libraries required by STRUMPACK."
|
||||
CHECK_BUILD STRUMPACK_VERSION_OK TRUE
|
||||
"
|
||||
#include <StrumpackSparseSolverMPIDist.hpp>
|
||||
using namespace strumpack;
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
MPI_Init(&argc, &argv);
|
||||
MPI_Comm comm = MPI_COMM_WORLD;
|
||||
StrumpackSparseSolverMPIDist<double,int> solver(comm, argc, argv, false);
|
||||
solver.options().set_from_command_line();
|
||||
return 0;
|
||||
}
|
||||
"
|
||||
)
|
||||
@@ -1,29 +0,0 @@
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Sets the following variables:
|
||||
# - Scotch_FOUND
|
||||
# - Scotch_INCLUDE_DIRS
|
||||
# - Scotch_LIBRARIES
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(Scotch Scotch Scotch_DIR "" "" "" ""
|
||||
"Paths to headers required by Scotch."
|
||||
"Libraries required by Scotch."
|
||||
ADD_COMPONENT "scotch" "include" scotch.h "lib" scotch
|
||||
ADD_COMPONENT "scotcherr" "" "" "lib" scotcherr
|
||||
ADD_COMPONENT "scotcherrexit" "" "" "lib" scotcherrexit
|
||||
ADD_COMPONENT "scotchmetis" "include" "metis.h" "lib" scotchmetis
|
||||
ADD_COMPONENT "ptscotch" "include" ptscotch.h "lib" ptscotch
|
||||
ADD_COMPONENT "ptscotcherr" "" "" "lib" ptscotcherr
|
||||
ADD_COMPONENT "ptscotcherrexit" "" "" "lib" ptscotcherrexit
|
||||
ADD_COMPONENT "ptscotchparmetis" "include" "parmetis.h" "lib" ptscotchparmetis
|
||||
)
|
||||
@@ -9,30 +9,6 @@
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Function that converts a version string of the form 'major[.minor[.patch]]' to
|
||||
# the integer ((major * 100) + minor) * 100 + patch.
|
||||
function(mfem_version_to_int VersionString VersionIntVar)
|
||||
if ("${VersionString}" MATCHES "^([0-9]+)(.*)$")
|
||||
set(Major "${CMAKE_MATCH_1}")
|
||||
set(MinorPatchString "${CMAKE_MATCH_2}")
|
||||
else()
|
||||
set(Major 0)
|
||||
endif()
|
||||
if ("${MinorPatchString}" MATCHES "^\\.([0-9]+)(.*)$")
|
||||
set(Minor "${CMAKE_MATCH_1}")
|
||||
set(PatchString "${CMAKE_MATCH_2}")
|
||||
else()
|
||||
set(Minor 0)
|
||||
endif()
|
||||
if ("${PatchString}" MATCHES "^\\.([0-9]+)(.*)$")
|
||||
set(Patch "${CMAKE_MATCH_1}")
|
||||
else()
|
||||
set(Patch 0)
|
||||
endif()
|
||||
math(EXPR VersionInt "(${Major}*100+${Minor})*100+${Patch}")
|
||||
set(${VersionIntVar} ${VersionInt} PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
# A handy function to add the current source directory to a local
|
||||
# filename. To be used for creating a list of sources.
|
||||
function(convert_filenames_to_full_paths NAMES)
|
||||
@@ -74,8 +50,10 @@ function(add_mfem_examples EXE_SRCS)
|
||||
|
||||
string(REPLACE ".cpp" "" EXE_NAME "${EXE_PREFIX}${SRC_FILENAME}")
|
||||
add_executable(${EXE_NAME} ${SRC_FILE})
|
||||
add_dependencies(${MFEM_ALL_EXAMPLES_TARGET_NAME} ${EXE_NAME})
|
||||
if (EXE_NEEDED_BY)
|
||||
# If given a prefix, don't add the example to the list of examples to build.
|
||||
if (NOT EXE_PREFIX)
|
||||
add_dependencies(${MFEM_ALL_EXAMPLES_TARGET_NAME} ${EXE_NAME})
|
||||
elseif (EXE_NEEDED_BY)
|
||||
add_dependencies(${EXE_NEEDED_BY} ${EXE_NAME})
|
||||
endif()
|
||||
add_dependencies(${EXE_NAME}
|
||||
@@ -83,8 +61,7 @@ function(add_mfem_examples EXE_SRCS)
|
||||
|
||||
target_link_libraries(${EXE_NAME} mfem)
|
||||
if (MFEM_USE_MPI)
|
||||
# Not needed: (mfem already links with MPI_CXX_LIBRARIES)
|
||||
# target_link_libraries(${EXE_NAME} ${MPI_CXX_LIBRARIES})
|
||||
target_link_libraries(${EXE_NAME} ${MPI_CXX_LIBRARIES})
|
||||
|
||||
# Language-specific include directories:
|
||||
if (MPI_CXX_INCLUDE_PATH)
|
||||
@@ -155,7 +132,6 @@ function(add_mfem_miniapp MFEM_EXE_NAME)
|
||||
|
||||
# Handle the MPI separately
|
||||
if (MFEM_USE_MPI)
|
||||
# Add MPI_CXX_LIBRARIES, in case this target does not link with mfem.
|
||||
if(CMAKE_VERSION VERSION_GREATER 2.8.11)
|
||||
target_link_libraries(${MFEM_EXE_NAME} PRIVATE ${MPI_CXX_LIBRARIES})
|
||||
else()
|
||||
@@ -183,26 +159,26 @@ function(mfem_find_component Prefix DirVar IncSuffixes Header LibSuffixes Lib
|
||||
|
||||
if (Lib)
|
||||
if (${DirVar} OR EnvDirVar)
|
||||
find_library(${Prefix}_LIBRARY ${Lib}
|
||||
find_library(${Prefix}_LIBRARIES ${Lib}
|
||||
HINTS ${${DirVar}} ENV ${DirVar}
|
||||
PATH_SUFFIXES ${LibSuffixes}
|
||||
NO_DEFAULT_PATH
|
||||
DOC "${LibDoc}")
|
||||
endif()
|
||||
find_library(${Prefix}_LIBRARY ${Lib}
|
||||
find_library(${Prefix}_LIBRARIES ${Lib}
|
||||
PATH_SUFFIXES ${LibSuffixes}
|
||||
DOC "${LibDoc}")
|
||||
endif()
|
||||
|
||||
if (Header)
|
||||
if (${DirVar} OR EnvDirVar)
|
||||
find_path(${Prefix}_INCLUDE_DIR ${Header}
|
||||
find_path(${Prefix}_INCLUDE_DIRS ${Header}
|
||||
HINTS ${${DirVar}} ENV ${DirVar}
|
||||
PATH_SUFFIXES ${IncSuffixes}
|
||||
NO_DEFAULT_PATH
|
||||
DOC "${IncDoc}")
|
||||
endif()
|
||||
find_path(${Prefix}_INCLUDE_DIR ${Header}
|
||||
find_path(${Prefix}_INCLUDE_DIRS ${Header}
|
||||
PATH_SUFFIXES ${IncSuffixes}
|
||||
DOC "${IncDoc}")
|
||||
endif()
|
||||
@@ -214,9 +190,8 @@ endfunction(mfem_find_component)
|
||||
# successful, optionally checks building (compile + link) one or more given
|
||||
# code snippets. Additionally, a list of required/optional/alternative
|
||||
# packages (given by ${Name}_REQUIRED_PACKAGES) are searched for and added to
|
||||
# the ${Prefix}_INCLUDE_DIRS and ${Prefix}_LIBRARIES lists. The variable
|
||||
# ${Name}_REQUIRED_LIBRARIES can be set to spcecify any additional libraries
|
||||
# that are needed. This function defines the following CACHE variables:
|
||||
# the ${Prefix}_INCLUDE_DIRS and ${Prefix}_LIBRARIES lists. The function
|
||||
# defines the following CACHE variables:
|
||||
#
|
||||
# ${Prefix}_FOUND
|
||||
# ${Prefix}_INCLUDE_DIRS
|
||||
@@ -255,14 +230,12 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
mfem_find_component("${Prefix}" "${DirVar}" "${IncSuffixes}" "${Header}"
|
||||
"${LibSuffixes}" "${Lib}" "${IncDoc}" "${LibDoc}")
|
||||
|
||||
if (((NOT Lib) OR ${Prefix}_LIBRARY) AND
|
||||
((NOT Header) OR ${Prefix}_INCLUDE_DIR))
|
||||
if (((NOT Lib) OR ${Prefix}_LIBRARIES) AND
|
||||
((NOT Header) OR ${Prefix}_INCLUDE_DIRS))
|
||||
set(Found TRUE)
|
||||
else()
|
||||
set(Found FALSE)
|
||||
endif()
|
||||
set(${Prefix}_LIBRARIES ${${Prefix}_LIBRARY})
|
||||
set(${Prefix}_INCLUDE_DIRS ${${Prefix}_INCLUDE_DIR})
|
||||
|
||||
set(ReqVars "")
|
||||
|
||||
@@ -301,22 +274,25 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
"${CompLibSuffixes}" "${CompLib}" "" "")
|
||||
if (CompRequired)
|
||||
if (CompLib)
|
||||
list(APPEND ReqVars ${FullPrefix}_LIBRARY)
|
||||
list(APPEND ReqVars ${FullPrefix}_LIBRARIES)
|
||||
endif()
|
||||
if (CompHeader)
|
||||
list(APPEND ReqVars ${FullPrefix}_INCLUDE_DIR)
|
||||
list(APPEND ReqVars ${FullPrefix}_INCLUDE_DIRS)
|
||||
endif()
|
||||
endif(CompRequired)
|
||||
if (((NOT CompLib) OR ${FullPrefix}_LIBRARY) AND
|
||||
((NOT CompHeader) OR ${FullPrefix}_INCLUDE_DIR))
|
||||
if (((NOT CompLib) OR ${FullPrefix}_LIBRARIES) AND
|
||||
((NOT CompHeader) OR ${FullPrefix}_INCLUDE_DIRS))
|
||||
# Component found
|
||||
list(APPEND ${Prefix}_LIBRARIES ${${FullPrefix}_LIBRARY})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${${FullPrefix}_INCLUDE_DIR})
|
||||
set(${FullPrefix}_FOUND TRUE CACHE BOOL
|
||||
"${Name}/${CompPrefix} was found." FORCE)
|
||||
list(APPEND ${Prefix}_LIBRARIES ${${FullPrefix}_LIBRARIES})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${${FullPrefix}_INCLUDE_DIRS})
|
||||
if (NOT ${Name}_FIND_QUIETLY)
|
||||
# message(STATUS "${Name}: ${CompPrefix}: found")
|
||||
message(STATUS
|
||||
"${Name}: ${CompPrefix}: ${${FullPrefix}_LIBRARY}")
|
||||
"${Name}: ${CompPrefix}: ${${FullPrefix}_LIBRARIES}")
|
||||
# message(STATUS
|
||||
# "${Name}: ${CompPrefix}: ${${FullPrefix}_INCLUDE_DIR}")
|
||||
# "${Name}: ${CompPrefix}: ${${FullPrefix}_INCLUDE_DIRS}")
|
||||
endif()
|
||||
else()
|
||||
# Let FindPackageHandleStandardArgs() handle errors
|
||||
@@ -369,8 +345,6 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
if (NOT ${Name}_FIND_QUIETLY)
|
||||
message(STATUS "${Name}: trying alternative package: ${ReqPackM}")
|
||||
endif()
|
||||
# Do not add ${Required} here, since that will prevent other potential
|
||||
# alternative packages from being found.
|
||||
find_package(${ReqPack} ${Quiet} COMPONENTS ${PackComps})
|
||||
string(TOUPPER ${ReqPack} ReqPACK)
|
||||
if (${ReqPack}_FOUND)
|
||||
@@ -384,7 +358,7 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
endif()
|
||||
elseif (Alternative)
|
||||
set(Alternative FALSE)
|
||||
elseif (Found)
|
||||
else()
|
||||
if (NOT ${Name}_FIND_QUIETLY)
|
||||
if (Required)
|
||||
message(STATUS "${Name}: looking for required package: ${ReqPackM}")
|
||||
@@ -394,139 +368,23 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
endif()
|
||||
string(TOUPPER ${ReqPack} ReqPACK)
|
||||
if (NOT (${ReqPack}_FOUND OR ${ReqPACK}_FOUND))
|
||||
if (NOT ${ReqPack}_TARGET_NAMES)
|
||||
find_package(${ReqPack} ${Required} ${Quiet} COMPONENTS ${PackComps})
|
||||
else()
|
||||
foreach(_target ${ReqPack} ${${ReqPack}_TARGET_NAMES})
|
||||
# Do not use ${Required} here:
|
||||
find_package(${_target} NAMES ${_target} ${ReqPack} ${Quiet}
|
||||
COMPONENTS ${PackComps})
|
||||
string(TOUPPER ${_target} _TARGET)
|
||||
if (${_target}_FOUND OR ${_TARGET}_FOUND)
|
||||
set(${ReqPack}_FOUND TRUE)
|
||||
break()
|
||||
endif()
|
||||
endforeach()
|
||||
if (${Required} AND NOT ${ReqPack}_FOUND)
|
||||
message(FATAL_ERROR " *** Required package ${ReqPack} not found."
|
||||
"Checked target names: ${ReqPack} ${${ReqPack}_TARGET_NAMES}")
|
||||
endif()
|
||||
endif()
|
||||
find_package(${ReqPack} ${Required} ${Quiet} COMPONENTS ${PackComps})
|
||||
endif()
|
||||
if (Required AND NOT (${ReqPack}_FOUND OR ${ReqPACK}_FOUND))
|
||||
message(FATAL_ERROR " --------- INTERNAL ERROR")
|
||||
endif()
|
||||
if ("${ReqPack}" STREQUAL "MPI" AND MPI_CXX_FOUND)
|
||||
if ("${ReqPack}" STREQUAL "MPI")
|
||||
list(APPEND ${Prefix}_LIBRARIES ${MPI_CXX_LIBRARIES})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${MPI_CXX_INCLUDE_PATH})
|
||||
elseif (${ReqPack}_FOUND OR ${ReqPACK}_FOUND)
|
||||
else()
|
||||
if (${ReqPack}_FOUND)
|
||||
set(_Pack ${ReqPack})
|
||||
else()
|
||||
set(_Pack ${ReqPACK})
|
||||
list(APPEND ${Prefix}_LIBRARIES ${${ReqPack}_LIBRARIES})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${${ReqPack}_INCLUDE_DIRS})
|
||||
elseif (${ReqPACK}_FOUND)
|
||||
list(APPEND ${Prefix}_LIBRARIES ${${ReqPACK}_LIBRARIES})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${${ReqPACK}_INCLUDE_DIRS})
|
||||
endif()
|
||||
set(_Pack_LIBS)
|
||||
set(_Pack_INCS)
|
||||
# - ${_Pack}_CONFIG is defined by find_package() when a config file was
|
||||
# loaded
|
||||
# - If ${ReqPack}_TARGET_NAMES is defined, use target mode
|
||||
if (NOT ((DEFINED ${_Pack}_CONFIG) OR
|
||||
(DEFINED ${ReqPack}_TARGET_NAMES)))
|
||||
# Defined variables expected:
|
||||
# - ${ReqPack}_LIB_VARS, optional, default: ${_Pack}_LIBRARIES
|
||||
# - ${ReqPack}_INCLUDE_VARS, optional, default: ${_Pack}_INCLUDE_DIRS
|
||||
set(_lib_vars ${${ReqPack}_LIB_VARS})
|
||||
if (NOT _lib_vars)
|
||||
set(_lib_vars ${_Pack}_LIBRARIES)
|
||||
endif()
|
||||
foreach (_var ${_lib_vars})
|
||||
if (${_var})
|
||||
list(APPEND _Pack_LIBS ${${_var}})
|
||||
endif()
|
||||
endforeach()
|
||||
# Includes
|
||||
set(_inc_vars ${${ReqPack}_INCLUDE_VARS})
|
||||
if (NOT _inc_vars)
|
||||
set(_inc_vars ${_Pack}_INCLUDE_DIRS)
|
||||
endif()
|
||||
foreach (_include ${_inc_vars})
|
||||
# message(STATUS "${Name}: ${ReqPack}: ${_include}")
|
||||
if (${_include})
|
||||
list(APPEND _Pack_INCS ${${_include}})
|
||||
endif()
|
||||
endforeach()
|
||||
else()
|
||||
# Target mode: check for a valid target:
|
||||
# - an entry in the variable ${ReqPack}_TARGET_NAMES (optional)
|
||||
# - ${_Pack}
|
||||
# Other optional variables:
|
||||
# - ${ReqPack}_IMPORT_CONFIG, default value: "RELEASE"
|
||||
# - ${ReqPack}_TARGET_FORCE, default value: "FALSE"
|
||||
set(TargetName)
|
||||
foreach (_target ${${ReqPack}_TARGET_NAMES} ${_Pack})
|
||||
if (TARGET ${_target})
|
||||
set(TargetName ${_target})
|
||||
break()
|
||||
endif()
|
||||
endforeach()
|
||||
if ("${TargetName}" STREQUAL "")
|
||||
message(FATAL_ERROR " *** ${ReqPack}: unknown target. "
|
||||
"Please set ${ReqPack}_TARGET_NAMES.")
|
||||
endif()
|
||||
get_target_property(IsImported ${TargetName} IMPORTED)
|
||||
if (IsImported)
|
||||
set(ImportConfig ${${ReqPack}_IMPORT_CONFIG})
|
||||
if (NOT ImportConfig)
|
||||
set(ImportConfig RELEASE)
|
||||
endif()
|
||||
get_target_property(ImpConfigs ${TargetName} IMPORTED_CONFIGURATIONS)
|
||||
list(FIND ImpConfigs ${ImportConfig} _Index)
|
||||
if (_Index EQUAL -1)
|
||||
message(FATAL_ERROR " *** ${ReqPack}: configuration "
|
||||
"${ImportConfig} not found. Set ${ReqPack}_IMPORT_CONFIG "
|
||||
"from the list: ${ImpConfigs}.")
|
||||
endif()
|
||||
endif()
|
||||
# Set _Pack_LIBS
|
||||
if (NOT IsImported OR ${ReqPack}_TARGET_FORCE)
|
||||
# Set _Pack_LIBS to be the target itself
|
||||
set(_Pack_LIBS ${TargetName})
|
||||
if (NOT ${Name}_FIND_QUIETLY)
|
||||
message(STATUS "Found ${ReqPack}: ${_Pack_LIBS} (target)")
|
||||
endif()
|
||||
else()
|
||||
# Set _Pack_LIBS from the target properties for ImportConfig
|
||||
foreach (_prop IMPORTED_LOCATION_${ImportConfig}
|
||||
IMPORTED_LINK_INTERFACE_LIBRARIES_${ImportConfig})
|
||||
get_target_property(_value ${TargetName} ${_prop})
|
||||
if (_value)
|
||||
list(APPEND _Pack_LIBS ${_value})
|
||||
endif()
|
||||
endforeach()
|
||||
if (NOT ${Name}_FIND_QUIETLY)
|
||||
message(STATUS
|
||||
"Imported ${ReqPack}[${ImportConfig}]: ${_Pack_LIBS}")
|
||||
endif()
|
||||
endif()
|
||||
# Set _Pack_INCS
|
||||
foreach (_prop INCLUDE_DIRECTORIES)
|
||||
get_target_property(_value ${TargetName} ${_prop})
|
||||
if (_value)
|
||||
list(APPEND _Pack_INCS ${_value})
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
# _Pack_LIBS and _Pack_INCS should be fully defined here
|
||||
list(APPEND ${Prefix}_LIBRARIES ${_Pack_LIBS})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${_Pack_INCS})
|
||||
endif()
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if (Found AND ${Name}_REQUIRED_LIBRARIES)
|
||||
list(APPEND ${Prefix}_LIBRARIES ${${Name}_REQUIRED_LIBRARIES})
|
||||
endif()
|
||||
|
||||
if (NOT ("${${Prefix}_INCLUDE_DIRS}" STREQUAL ""))
|
||||
list(INSERT ReqVars 0 ${Prefix}_INCLUDE_DIRS)
|
||||
set(ReqHeaders 1)
|
||||
@@ -543,6 +401,12 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
if (ReqHeaders)
|
||||
list(REMOVE_DUPLICATES ${Prefix}_INCLUDE_DIRS)
|
||||
endif()
|
||||
# Write the updated values to the cache.
|
||||
set(${Prefix}_LIBRARIES ${${Prefix}_LIBRARIES} CACHE STRING
|
||||
"${LibDoc}" FORCE)
|
||||
set(${Prefix}_INCLUDE_DIRS ${${Prefix}_INCLUDE_DIRS} CACHE STRING
|
||||
"${IncDoc}" FORCE)
|
||||
set(${Prefix}_FOUND TRUE CACHE BOOL "${Name} was found." FORCE)
|
||||
|
||||
# Check for optional "CHECK_BUILD" arguments.
|
||||
set(I 9) # 9 is the number of required arguments
|
||||
@@ -560,10 +424,6 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
set(CMAKE_REQUIRED_QUIET ${${Name}_FIND_QUIETLY})
|
||||
check_cxx_source_compiles("${TestSrc}" ${TestVar})
|
||||
if (TestReq)
|
||||
if (NOT ${TestVar})
|
||||
set(Found FALSE)
|
||||
unset(${TestVar} CACHE)
|
||||
endif()
|
||||
list(APPEND ReqVars ${TestVar})
|
||||
endif()
|
||||
elseif("${ARGV${I}}" STREQUAL "ADD_COMPONENT")
|
||||
@@ -574,35 +434,22 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
endif()
|
||||
math(EXPR I "${I}+1")
|
||||
endwhile()
|
||||
else()
|
||||
set(${Prefix}_FOUND FALSE CACHE BOOL "${Name} was not found." FORCE)
|
||||
endif()
|
||||
if ("_x_${ReqVars}" STREQUAL "_x_")
|
||||
set(${Prefix}_FOUND ${Found})
|
||||
set(ReqVars ${Prefix}_FOUND)
|
||||
endif()
|
||||
# foreach(ReqVar ${ReqVars})
|
||||
# message(STATUS " *** ${ReqVar}=${${ReqVar}}")
|
||||
# get_property(IsCached CACHE ${ReqVar} PROPERTY "VALUE" SET)
|
||||
# if (IsCached)
|
||||
# get_property(CachedVal CACHE ${ReqVar} PROPERTY "VALUE")
|
||||
# message(STATUS " *** ${ReqVar}[cached]=${CachedVal}")
|
||||
# endif()
|
||||
# message(STATUS "${ReqVar}=${${ReqVar}}")
|
||||
# endforeach()
|
||||
|
||||
include(FindPackageHandleStandardArgs)
|
||||
find_package_handle_standard_args(${Name}
|
||||
" *** ${Name} not found. Please set ${DirVar}." ${ReqVars})
|
||||
|
||||
string(TOUPPER ${Name} UName)
|
||||
if (${UName}_FOUND)
|
||||
# Write the ${Prefix}_* variables to the cache.
|
||||
set(${Prefix}_LIBRARIES ${${Prefix}_LIBRARIES} CACHE STRING
|
||||
"${LibDoc}" FORCE)
|
||||
set(${Prefix}_INCLUDE_DIRS ${${Prefix}_INCLUDE_DIRS} CACHE STRING
|
||||
"${IncDoc}" FORCE)
|
||||
set(${Prefix}_FOUND TRUE CACHE BOOL "${Name} was found." FORCE)
|
||||
if (ReqHeaders AND (NOT ${Name}_FIND_QUIETLY))
|
||||
message(STATUS "${Prefix}_INCLUDE_DIRS=${${Prefix}_INCLUDE_DIRS}")
|
||||
endif()
|
||||
if (Found AND ReqLibs AND ReqHeaders AND (NOT ${Name}_FIND_QUIETLY))
|
||||
message(STATUS "${Prefix}_INCLUDE_DIRS=${${Prefix}_INCLUDE_DIRS}")
|
||||
endif()
|
||||
|
||||
endfunction(mfem_find_package)
|
||||
|
||||
@@ -30,18 +30,7 @@
|
||||
#ifdef MFEM_USE_SUPERLU
|
||||
#error Building with SuperLU_DIST (MFEM_USE_SUPERLU=YES) requires MPI (MFEM_USE_MPI=YES)
|
||||
#endif
|
||||
#ifdef MFEM_USE_STRUMPACK
|
||||
#error Building with STRUMPACK (MFEM_USE_STRUMPACK=YES) requires MPI (MFEM_USE_MPI=YES)
|
||||
#endif
|
||||
#ifdef MFEM_USE_PETSC
|
||||
#error Building with PETSc (MFEM_USE_PETSC=YES) requires MPI (MFEM_USE_MPI=YES)
|
||||
#endif
|
||||
#endif // MFEM_USE_MPI not defined
|
||||
|
||||
// Macro that returns its first arg when MFEM_USE_BACKENDS is defined, and its
|
||||
// second arg if it is not defined.
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
#define MFEM_IF_BACKENDS(x,y) (x)
|
||||
#else
|
||||
#define MFEM_IF_BACKENDS(x,y) (y)
|
||||
#endif
|
||||
|
||||
+5
-58
@@ -12,33 +12,6 @@
|
||||
#ifndef MFEM_CONFIG_HEADER
|
||||
#define MFEM_CONFIG_HEADER
|
||||
|
||||
// MFEM version: integer of the form: (major*100 + minor)*100 + patch.
|
||||
// #define MFEM_VERSION @MFEM_VERSION@
|
||||
|
||||
// MFEM version string of the form "3.3" or "3.3.1".
|
||||
// #define MFEM_VERSION_STRING "@MFEM_VERSION_STRING@"
|
||||
|
||||
// MFEM version type, see the MFEM_VERSION_TYPE_* constants below.
|
||||
#define MFEM_VERSION_TYPE ((MFEM_VERSION)%2)
|
||||
|
||||
// MFEM version type constants.
|
||||
#define MFEM_VERSION_TYPE_RELEASE 0
|
||||
#define MFEM_VERSION_TYPE_DEVELOPMENT 1
|
||||
|
||||
// Separate MFEM version numbers for major, minor, and patch.
|
||||
#define MFEM_VERSION_MAJOR ((MFEM_VERSION)/10000)
|
||||
#define MFEM_VERSION_MINOR (((MFEM_VERSION)/100)%100)
|
||||
#define MFEM_VERSION_PATCH ((MFEM_VERSION)%100)
|
||||
|
||||
// Description of the git commit used to build MFEM.
|
||||
// #define MFEM_GIT_STRING "@MFEM_GIT_STRING@"
|
||||
|
||||
// The absolute path of the MFEM source prefix
|
||||
// #define MFEM_SOURCE_DIR "@MFEM_SOURCE_DIR@"
|
||||
|
||||
// The absolute path of the MFEM installation prefix
|
||||
// #define MFEM_INSTALL_DIR "@MFEM_INSTALL_DIR@"
|
||||
|
||||
// Build the parallel MFEM library.
|
||||
// Requires an MPI compiler, and the libraries HYPRE and METIS.
|
||||
// #define MFEM_USE_MPI
|
||||
@@ -46,18 +19,12 @@
|
||||
// Enable debug checks in MFEM.
|
||||
// #define MFEM_DEBUG
|
||||
|
||||
// Throw an exception on errors.
|
||||
// #define MFEM_USE_EXCEPTIONS
|
||||
|
||||
// Enable gzstream in MFEM.
|
||||
// #define MFEM_USE_GZSTREAM
|
||||
|
||||
// Enable backtraces for mfem_error through libunwind.
|
||||
// #define MFEM_USE_LIBUNWIND
|
||||
|
||||
// Enable MFEM features that use the METIS library (parallel MFEM).
|
||||
// #define MFEM_USE_METIS
|
||||
|
||||
// Enable this option if linking with METIS version 5 (parallel MFEM).
|
||||
// #define MFEM_USE_METIS_5
|
||||
|
||||
@@ -75,7 +42,11 @@
|
||||
// #define MFEM_USE_MEMALLOC
|
||||
|
||||
// Which library functions to use in class StopWatch for measuring time.
|
||||
// For a list of the available options, see INSTALL.
|
||||
// The available options are:
|
||||
// 0 - use std::clock from <ctime>
|
||||
// 1 - use times from <sys/times.h>
|
||||
// 2 - use high-resolution POSIX clocks
|
||||
// 3 - use QueryPerformanceCounter from <windows.h>
|
||||
// If not defined, an option is selected automatically.
|
||||
// #define MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@
|
||||
|
||||
@@ -91,9 +62,6 @@
|
||||
// Enable MFEM functionality based on the SuperLU library.
|
||||
// #define MFEM_USE_SUPERLU
|
||||
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
// #define MFEM_USE_STRUMPACK
|
||||
|
||||
// Enable functionality based on the Gecko library
|
||||
// #define MFEM_USE_GECKO
|
||||
|
||||
@@ -103,9 +71,6 @@
|
||||
// Enable Sidre support
|
||||
// #define MFEM_USE_SIDRE
|
||||
|
||||
// Enable Conduit support
|
||||
// #define MFEM_USE_CONDUIT
|
||||
|
||||
// Enable functionality based on the NetCDF library (reading CUBIT files)
|
||||
// #define MFEM_USE_NETCDF
|
||||
|
||||
@@ -115,28 +80,10 @@
|
||||
// Enable functionality based on the MPFR library.
|
||||
// #define MFEM_USE_MPFR
|
||||
|
||||
// Enable the use of MFEM backends.
|
||||
// #define MFEM_USE_BACKENDS
|
||||
|
||||
// Enable the OCCA backend.
|
||||
// #define MFEM_USE_OCCA
|
||||
|
||||
// Enable the OMP backend.
|
||||
// #define MFEM_USE_OMP
|
||||
|
||||
// Enable use of acrotensor in backends.
|
||||
// #define MFEM_USE_ACROTENSOR
|
||||
|
||||
// Enable use of unified memory.
|
||||
// #define MFEM_USE_CUDAUM
|
||||
|
||||
// Windows specific options
|
||||
#ifdef _WIN32
|
||||
// Macro needed to get defines like M_PI from <cmath>. (Visual Studio C++ only?)
|
||||
#define _USE_MATH_DEFINES
|
||||
#endif
|
||||
|
||||
// Version of HYPRE used for building MFEM.
|
||||
// #define MFEM_HYPRE_VERSION @MFEM_HYPRE_VERSION@
|
||||
|
||||
#endif // MFEM_CONFIG_HEADER
|
||||
|
||||
@@ -10,16 +10,9 @@
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Variables corresponding to defines in config.hpp (YES, NO, or value)
|
||||
MFEM_VERSION = @MFEM_VERSION@
|
||||
MFEM_VERSION_STRING = @MFEM_VERSION_STRING@
|
||||
MFEM_GIT_STRING = @MFEM_GIT_STRING@
|
||||
MFEM_SOURCE_DIR = @MFEM_SOURCE_DIR@
|
||||
MFEM_INSTALL_DIR = @MFEM_INSTALL_DIR@
|
||||
MFEM_USE_MPI = @MFEM_USE_MPI@
|
||||
MFEM_USE_METIS = @MFEM_USE_METIS@
|
||||
MFEM_USE_METIS_5 = @MFEM_USE_METIS_5@
|
||||
MFEM_DEBUG = @MFEM_DEBUG@
|
||||
MFEM_USE_EXCEPTIONS = @MFEM_USE_EXCEPTIONS@
|
||||
MFEM_USE_GZSTREAM = @MFEM_USE_GZSTREAM@
|
||||
MFEM_USE_LIBUNWIND = @MFEM_USE_LIBUNWIND@
|
||||
MFEM_USE_LAPACK = @MFEM_USE_LAPACK@
|
||||
@@ -31,19 +24,12 @@ MFEM_USE_SUNDIALS = @MFEM_USE_SUNDIALS@
|
||||
MFEM_USE_MESQUITE = @MFEM_USE_MESQUITE@
|
||||
MFEM_USE_SUITESPARSE = @MFEM_USE_SUITESPARSE@
|
||||
MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
|
||||
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
|
||||
MFEM_USE_GECKO = @MFEM_USE_GECKO@
|
||||
MFEM_USE_GNUTLS = @MFEM_USE_GNUTLS@
|
||||
MFEM_USE_NETCDF = @MFEM_USE_NETCDF@
|
||||
MFEM_USE_PETSC = @MFEM_USE_PETSC@
|
||||
MFEM_USE_MPFR = @MFEM_USE_MPFR@
|
||||
MFEM_USE_SIDRE = @MFEM_USE_SIDRE@
|
||||
MFEM_USE_CONDUIT = @MFEM_USE_CONDUIT@
|
||||
MFEM_USE_BACKENDS = @MFEM_USE_BACKENDS@
|
||||
MFEM_USE_OCCA = @MFEM_USE_OCCA@
|
||||
MFEM_USE_OMP = @MFEM_USE_OMP@
|
||||
MFEM_USE_ACROTENSOR = @MFEM_USE_ACROTENSOR@
|
||||
MFEM_USE_CUDAUM = @MFEM_USE_CUDAUM@
|
||||
|
||||
# Compiler, compile options, and link options
|
||||
MFEM_CXX = @MFEM_CXX@
|
||||
@@ -51,25 +37,17 @@ MFEM_CPPFLAGS = @MFEM_CPPFLAGS@
|
||||
MFEM_CXXFLAGS = @MFEM_CXXFLAGS@
|
||||
MFEM_TPLFLAGS = @MFEM_TPLFLAGS@
|
||||
MFEM_INCFLAGS = @MFEM_INCFLAGS@
|
||||
MFEM_PICFLAG = @MFEM_PICFLAG@
|
||||
MFEM_FLAGS = @MFEM_FLAGS@
|
||||
MFEM_EXT_LIBS = @MFEM_EXT_LIBS@
|
||||
MFEM_LIBS = @MFEM_LIBS@
|
||||
MFEM_LIB_FILE = @MFEM_LIB_FILE@
|
||||
MFEM_STATIC = @MFEM_STATIC@
|
||||
MFEM_SHARED = @MFEM_SHARED@
|
||||
MFEM_BUILD_TAG = @MFEM_BUILD_TAG@
|
||||
MFEM_PREFIX = @MFEM_PREFIX@
|
||||
MFEM_INC_DIR = @MFEM_INC_DIR@
|
||||
MFEM_LIB_DIR = @MFEM_LIB_DIR@
|
||||
|
||||
# Location of test.mk
|
||||
MFEM_TEST_MK = @MFEM_TEST_MK@
|
||||
|
||||
# Command used to launch MPI jobs
|
||||
MFEM_MPIEXEC = @MFEM_MPIEXEC@
|
||||
MFEM_MPIEXEC_NP = @MFEM_MPIEXEC_NP@
|
||||
MFEM_MPI_NP = @MFEM_MPI_NP@
|
||||
|
||||
# Optional extra configuration
|
||||
@MFEM_CONFIG_EXTRA@
|
||||
|
||||
+11
-40
@@ -18,10 +18,8 @@ if (NOT CMAKE_BUILD_TYPE)
|
||||
"Build type: Debug, Release, RelWithDebInfo, or MinSizeRel." FORCE)
|
||||
endif()
|
||||
|
||||
# MFEM options. Set to mimic the default "defaults.mk" file.
|
||||
# MFEM options. Set to mimic the default "default.mk" file.
|
||||
option(MFEM_USE_MPI "Enable MPI parallel build" OFF)
|
||||
option(MFEM_USE_METIS "Enable METIS usage" ${MFEM_USE_MPI})
|
||||
option(MFEM_USE_EXCEPTIONS "Enable the use of exceptions" OFF)
|
||||
option(MFEM_USE_GZSTREAM "Enable gzstream for compressed data streams." OFF)
|
||||
option(MFEM_USE_LIBUNWIND "Enable backtrace for errors." OFF)
|
||||
option(MFEM_USE_LAPACK "Enable LAPACK usage" OFF)
|
||||
@@ -32,20 +30,17 @@ option(MFEM_USE_SUNDIALS "Enable SUNDIALS usage" OFF)
|
||||
option(MFEM_USE_MESQUITE "Enable MESQUITE usage" OFF)
|
||||
option(MFEM_USE_SUITESPARSE "Enable SuiteSparse usage" OFF)
|
||||
option(MFEM_USE_SUPERLU "Enable SuperLU_DIST usage" OFF)
|
||||
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
|
||||
option(MFEM_USE_GECKO "Enable GECKO usage" OFF)
|
||||
option(MFEM_USE_GNUTLS "Enable GNUTLS usage" OFF)
|
||||
option(MFEM_USE_NETCDF "Enable NETCDF usage" OFF)
|
||||
option(MFEM_USE_PETSC "Enable PETSc support." OFF)
|
||||
option(MFEM_USE_MPFR "Enable MPFR usage." OFF)
|
||||
option(MFEM_USE_SIDRE "Enable Axom/Sidre usage" OFF)
|
||||
option(MFEM_USE_CONDUIT "Enable Conduit usage" OFF)
|
||||
option(MFEM_USE_SIDRE "Enable ATK/Sidre usage" OFF)
|
||||
|
||||
# Allow a user to disable testing, examples, and/or miniapps at CONFIGURE TIME
|
||||
# if they don't want/need them (e.g. if MFEM is "just a dependency" and all they
|
||||
# need is the library, building all that stuff adds unnecessary overhead). Note
|
||||
# that the examples or miniapps can always be built using the targets 'examples'
|
||||
# or 'miniapps', respectively.
|
||||
# need is the library, building all that stuff adds unnecessary overhead). To
|
||||
# match "makefile" behavior, they are all enabled by default.
|
||||
option(MFEM_ENABLE_TESTING "Enable the ctest framework for testing" ON)
|
||||
option(MFEM_ENABLE_EXAMPLES "Build all of the examples" OFF)
|
||||
option(MFEM_ENABLE_MINIAPPS "Build all of the miniapps" OFF)
|
||||
@@ -71,7 +66,7 @@ set(METIS_DIR "${MFEM_DIR}/../metis-4.0" CACHE PATH "Path to the METIS library."
|
||||
|
||||
set(LIBUNWIND_DIR "" CACHE PATH "Path to Libunwind.")
|
||||
|
||||
set(SUNDIALS_DIR "${MFEM_DIR}/../sundials-3.0.0" CACHE PATH
|
||||
set(SUNDIALS_DIR "${MFEM_DIR}/../sundials-2.7.0" CACHE PATH
|
||||
"Path to the SUNDIALS library.")
|
||||
# The following may be necessary, if SUNDIALS was built with KLU:
|
||||
# set(SUNDIALS_REQUIRED_PACKAGES "SuiteSparse/KLU/AMD/BTF/COLAMD/config"
|
||||
@@ -96,32 +91,6 @@ set(SuperLUDist_DIR "${MFEM_DIR}/../SuperLU_DIST_5.1.0" CACHE PATH
|
||||
set(SuperLUDist_REQUIRED_PACKAGES "MPI" "BLAS" "ParMETIS" CACHE STRING
|
||||
"Additional packages required by SuperLU_DIST.")
|
||||
|
||||
set(STRUMPACK_DIR "${MFEM_DIR}/../STRUMPACK-build" CACHE PATH
|
||||
"Path to the STRUMPACK library.")
|
||||
# STRUMPACK may also depend on "OpenMP", depending on how it was compiled.
|
||||
set(STRUMPACK_REQUIRED_PACKAGES "MPI" "MPI_Fortran" "ParMETIS" "METIS"
|
||||
"ScaLAPACK" "Scotch/ptscotch/ptscotcherr/scotch/scotcherr" CACHE STRING
|
||||
"Additional packages required by STRUMPACK.")
|
||||
# If the MPI package does not find all required Fortran libraries:
|
||||
# set(STRUMPACK_REQUIRED_LIBRARIES "gfortran" "mpi_mpifh" CACHE STRING
|
||||
# "Additional libraries required by STRUMPACK.")
|
||||
|
||||
# The Scotch library, required by STRUMPACK
|
||||
set(Scotch_DIR "${MFEM_DIR}/../scotch_6.0.4" CACHE PATH
|
||||
"Path to the Scotch and PT-Scotch libraries.")
|
||||
set(Scotch_REQUIRED_PACKAGES "Threads" CACHE STRING
|
||||
"Additional packages required by Scotch.")
|
||||
# Tell the "Threads" package/module to prefer pthreads.
|
||||
set(CMAKE_THREAD_PREFER_PTHREAD TRUE)
|
||||
set(Threads_LIB_VARS CMAKE_THREAD_LIBS_INIT)
|
||||
|
||||
# The ScaLAPACK library, required by STRUMPACK
|
||||
set(ScaLAPACK_DIR "${MFEM_DIR}/../scalapack-2.0.2/lib/cmake/scalapack-2.0.2"
|
||||
CACHE PATH "Path to the configuration file scalapack-config.cmake")
|
||||
set(ScaLAPACK_TARGET_NAMES scalapack)
|
||||
# set(ScaLAPACK_TARGET_FORCE)
|
||||
# set(ScaLAPACK_IMPORT_CONFIG DEBUG)
|
||||
|
||||
set(GECKO_DIR "${MFEM_DIR}/../gecko" CACHE PATH "Path to the Gecko library.")
|
||||
|
||||
set(GNUTLS_DIR "" CACHE PATH "Path to the GnuTLS library.")
|
||||
@@ -133,17 +102,19 @@ set(NetCDF_REQUIRED_PACKAGES "" CACHE STRING
|
||||
|
||||
set(PETSC_DIR "${MFEM_DIR}/../petsc" CACHE PATH
|
||||
"Path to the PETSc main directory.")
|
||||
set(PETSC_ARCH "arch-linux2-c-debug" CACHE STRING "PETSc build architecture.")
|
||||
set(PETSC_ARCH "arch-linux2-c-debug" CACHE PATH "PETSc build architecture.")
|
||||
|
||||
set(MPFR_DIR "" CACHE PATH "Path to the MPFR library.")
|
||||
|
||||
set(CONDUIT_DIR "${MFEM_DIR}/../conduit" CACHE PATH
|
||||
"Path to the Conduit library.")
|
||||
set(Conduit_REQUIRED_PACKAGES "HDF5" CACHE STRING
|
||||
"Additional packages required by Conduit.")
|
||||
|
||||
set(AXOM_DIR "${MFEM_DIR}/../axom" CACHE PATH "Path to the Axom library.")
|
||||
set(ATK_DIR "${MFEM_DIR}/../asctoolkit" CACHE PATH "Path to the ATK library.")
|
||||
# May need to add "Boost" as requirement.
|
||||
set(Axom_REQUIRED_PACKAGES "Conduit/relay" CACHE STRING
|
||||
"Additional packages required by Axom.")
|
||||
set(ATK_REQUIRED_PACKAGES "Conduit/relay" CACHE STRING
|
||||
"Additional packages required by ATK.")
|
||||
|
||||
set(BLAS_INCLUDE_DIRS "" CACHE STRING "Path to BLAS headers.")
|
||||
set(BLAS_LIBRARIES "" CACHE STRING "The BLAS library.")
|
||||
|
||||
+31
-137
@@ -30,36 +30,15 @@ PREFIX = ./mfem
|
||||
# Install program
|
||||
INSTALL = /usr/bin/install
|
||||
|
||||
STATIC = YES
|
||||
SHARED = NO
|
||||
|
||||
ifneq ($(NOTMAC),)
|
||||
AR = ar
|
||||
ARFLAGS = cruv
|
||||
RANLIB = ranlib
|
||||
PICFLAG = -fPIC
|
||||
SO_EXT = so
|
||||
SO_VER = so.$(MFEM_VERSION_STRING)
|
||||
BUILD_SOFLAGS = -shared -Wl,-soname,libmfem.$(SO_VER)
|
||||
BUILD_RPATH = -Wl,-rpath,$(BUILD_REAL_DIR)
|
||||
INSTALL_SOFLAGS = $(BUILD_SOFLAGS)
|
||||
INSTALL_RPATH = -Wl,-rpath,@MFEM_LIB_DIR@
|
||||
else
|
||||
# Silence "has no symbols" warnings on Mac OS X
|
||||
AR = ar
|
||||
ARFLAGS = Scruv
|
||||
RANLIB = ranlib -no_warning_for_no_symbols
|
||||
PICFLAG = -fPIC
|
||||
SO_EXT = dylib
|
||||
SO_VER = $(MFEM_VERSION_STRING).dylib
|
||||
MAKE_SOFLAGS = -Wl,-dylib,-install_name,$(1)/libmfem.$(SO_VER),\
|
||||
-compatibility_version,$(MFEM_VERSION_STRING),\
|
||||
-current_version,$(MFEM_VERSION_STRING),\
|
||||
-undefined,dynamic_lookup
|
||||
BUILD_SOFLAGS = $(subst $1 ,,$(call MAKE_SOFLAGS,$(BUILD_REAL_DIR)))
|
||||
BUILD_RPATH = -Wl,-undefined,dynamic_lookup
|
||||
INSTALL_SOFLAGS = $(subst $1 ,,$(call MAKE_SOFLAGS,$(MFEM_LIB_DIR)))
|
||||
INSTALL_RPATH = -Wl,-undefined,dynamic_lookup
|
||||
endif
|
||||
|
||||
# Set CXXFLAGS to overwrite the default selection of DEBUG_FLAGS/OPTIM_FLAGS
|
||||
@@ -75,50 +54,31 @@ endif
|
||||
# Command used to launch MPI jobs
|
||||
MFEM_MPIEXEC = mpirun
|
||||
MFEM_MPIEXEC_NP = -np
|
||||
# Number of mpi tasks for parallel jobs
|
||||
MFEM_MPI_NP = 4
|
||||
|
||||
# MFEM configuration options: YES/NO values, which are exported to config.mk and
|
||||
# config.hpp. The values below are the defaults for generating the actual values
|
||||
# in config.mk and config.hpp.
|
||||
|
||||
MFEM_USE_MPI = NO
|
||||
# FIXME: add MFEM_USE_BACKENDS, MFEM_USE_OCCA to the CMake build system
|
||||
MFEM_USE_BACKENDS = YES
|
||||
MFEM_USE_OCCA = YES
|
||||
MFEM_USE_METIS = $(MFEM_USE_MPI)
|
||||
MFEM_USE_METIS_5 = NO
|
||||
MFEM_DEBUG = NO
|
||||
MFEM_USE_EXCEPTIONS = NO
|
||||
MFEM_USE_GZSTREAM = NO
|
||||
MFEM_USE_LIBUNWIND = NO
|
||||
MFEM_USE_LAPACK = NO
|
||||
MFEM_THREAD_SAFE = NO
|
||||
MFEM_USE_OPENMP = NO
|
||||
MFEM_USE_MEMALLOC = YES
|
||||
MFEM_TIMER_TYPE = $(if $(NOTMAC),2,4)
|
||||
MFEM_TIMER_TYPE = $(if $(NOTMAC),2,0)
|
||||
MFEM_USE_SUNDIALS = NO
|
||||
MFEM_USE_MESQUITE = NO
|
||||
MFEM_USE_SUITESPARSE = NO
|
||||
MFEM_USE_SUPERLU = NO
|
||||
MFEM_USE_STRUMPACK = NO
|
||||
MFEM_USE_GECKO = NO
|
||||
MFEM_USE_GNUTLS = NO
|
||||
MFEM_USE_NETCDF = NO
|
||||
MFEM_USE_PETSC = NO
|
||||
MFEM_USE_MPFR = NO
|
||||
MFEM_USE_SIDRE = NO
|
||||
MFEM_USE_CONDUIT = NO
|
||||
# FIXME: add MFEM_USE_OMP and MFEM_USE_ACROTENSOR to the CMake build system
|
||||
MFEM_USE_OMP = NO
|
||||
MFEM_USE_ACROTENSOR = NO
|
||||
MFEM_USE_CUDAUM = NO
|
||||
|
||||
# Compile and link options for zlib.
|
||||
ZLIB_DIR =
|
||||
ZLIB_OPT = $(if $(ZLIB_DIR),-I$(ZLIB_DIR)/include)
|
||||
ZLIB_LIB = $(if $(ZLIB_DIR),$(ZLIB_RPATH) -L$(ZLIB_DIR)/lib ,)-lz
|
||||
ZLIB_RPATH = -Wl,-rpath,$(ZLIB_DIR)/lib
|
||||
|
||||
LIBUNWIND_OPT = -g
|
||||
LIBUNWIND_LIB = $(if $(NOTMAC),-lunwind -ldl,)
|
||||
@@ -129,7 +89,7 @@ HYPRE_OPT = -I$(HYPRE_DIR)/include
|
||||
HYPRE_LIB = -L$(HYPRE_DIR)/lib -lHYPRE
|
||||
|
||||
# METIS library configuration
|
||||
ifeq ($(MFEM_USE_SUPERLU)$(MFEM_USE_STRUMPACK),NONO)
|
||||
ifeq ($(MFEM_USE_SUPERLU),NO)
|
||||
ifeq ($(MFEM_USE_METIS_5),NO)
|
||||
METIS_DIR = @MFEM_DIR@/../metis-4.0
|
||||
METIS_OPT =
|
||||
@@ -140,7 +100,7 @@ ifeq ($(MFEM_USE_SUPERLU)$(MFEM_USE_STRUMPACK),NONO)
|
||||
METIS_LIB = -L$(METIS_DIR)/lib -lmetis
|
||||
endif
|
||||
else
|
||||
# ParMETIS: currently needed by SuperLU or STRUMPACK. We assume that METIS 5
|
||||
# ParMETIS currently needed only with SuperLU. We assume that METIS 5
|
||||
# (included with ParMETIS) is installed in the same location.
|
||||
METIS_DIR = @MFEM_DIR@/../parmetis-4.0.3
|
||||
METIS_OPT = -I$(METIS_DIR)/include
|
||||
@@ -160,10 +120,10 @@ OPENMP_LIB =
|
||||
POSIX_CLOCKS_LIB = -lrt
|
||||
|
||||
# SUNDIALS library configuration
|
||||
SUNDIALS_DIR = @MFEM_DIR@/../sundials-3.0.0
|
||||
SUNDIALS_DIR = @MFEM_DIR@/../sundials-2.7.0
|
||||
SUNDIALS_OPT = -I$(SUNDIALS_DIR)/include
|
||||
SUNDIALS_LIB = -Wl,-rpath,$(SUNDIALS_DIR)/lib -L$(SUNDIALS_DIR)/lib\
|
||||
-lsundials_arkode -lsundials_cvode -lsundials_nvecserial -lsundials_kinsol
|
||||
-lsundials_arkode -lsundials_cvode -lsundials_nvecserial -lsundials_kinsol
|
||||
|
||||
ifeq ($(MFEM_USE_MPI),YES)
|
||||
SUNDIALS_LIB += -lsundials_nvecparhyp -lsundials_nvecparallel
|
||||
@@ -180,41 +140,14 @@ MESQUITE_LIB = -L$(MESQUITE_DIR)/lib -lmesquite
|
||||
LIB_RT = $(if $(NOTMAC),-lrt,)
|
||||
SUITESPARSE_DIR = @MFEM_DIR@/../SuiteSparse
|
||||
SUITESPARSE_OPT = -I$(SUITESPARSE_DIR)/include
|
||||
SUITESPARSE_LIB = -Wl,-rpath,$(SUITESPARSE_DIR)/lib -L$(SUITESPARSE_DIR)/lib\
|
||||
-lklu -lbtf -lumfpack -lcholmod -lcolamd -lamd -lcamd -lccolamd\
|
||||
-lsuitesparseconfig $(LIB_RT) $(METIS_LIB) $(LAPACK_LIB)
|
||||
SUITESPARSE_LIB = -L$(SUITESPARSE_DIR)/lib -lklu -lbtf -lumfpack -lcholmod\
|
||||
-lcolamd -lamd -lcamd -lccolamd -lsuitesparseconfig $(LIB_RT) $(METIS_LIB)\
|
||||
$(LAPACK_LIB)
|
||||
|
||||
# SuperLU library configuration
|
||||
SUPERLU_DIR = @MFEM_DIR@/../SuperLU_DIST_5.1.0
|
||||
SUPERLU_OPT = -I$(SUPERLU_DIR)/SRC
|
||||
SUPERLU_LIB = -Wl,-rpath,$(SUPERLU_DIR)/SRC -L$(SUPERLU_DIR)/SRC -lsuperlu_dist
|
||||
|
||||
# SCOTCH library configuration (required by STRUMPACK)
|
||||
SCOTCH_DIR = @MFEM_DIR@/../scotch_6.0.4
|
||||
SCOTCH_OPT = -I$(SCOTCH_DIR)/include
|
||||
SCOTCH_LIB = -L$(SCOTCH_DIR)/lib -lptscotch -lptscotcherr -lscotch -lscotcherr\
|
||||
-lpthread
|
||||
|
||||
# SCALAPACK library configuration (required by STRUMPACK)
|
||||
SCALAPACK_DIR = @MFEM_DIR@/../scalapack-2.0.2
|
||||
SCALAPACK_OPT = -I$(SCALAPACK_DIR)/SRC
|
||||
SCALAPACK_LIB = -L$(SCALAPACK_DIR)/lib -lscalapack $(LAPACK_LIB)
|
||||
|
||||
# MPI Fortran library, needed e.g. by STRUMPACK
|
||||
# MPICH:
|
||||
MPI_FORTRAN_LIB = -lmpifort
|
||||
# OpenMPI:
|
||||
# MPI_FORTRAN_LIB = -lmpi_mpifh
|
||||
# Additional Fortan library:
|
||||
# MPI_FORTRAN_LIB += -lgfortran
|
||||
|
||||
# STRUMPACK library configuration
|
||||
STRUMPACK_DIR = @MFEM_DIR@/../STRUMPACK-build
|
||||
STRUMPACK_OPT = -I$(STRUMPACK_DIR)/include $(SCOTCH_OPT)
|
||||
# If STRUMPACK was build with OpenMP support, the following may be need:
|
||||
# STRUMPACK_OPT += $(OPENMP_OPT)
|
||||
STRUMPACK_LIB = -L$(STRUMPACK_DIR)/lib -lstrumpack $(MPI_FORTRAN_LIB)\
|
||||
$(SCOTCH_LIB) $(SCALAPACK_LIB)
|
||||
SUPERLU_LIB = -L$(SUPERLU_DIR)/SRC -lsuperlu_dist
|
||||
|
||||
# Gecko library configuration
|
||||
GECKO_DIR = @MFEM_DIR@/../gecko
|
||||
@@ -226,81 +159,42 @@ GNUTLS_OPT =
|
||||
GNUTLS_LIB = -lgnutls
|
||||
|
||||
# NetCDF library configuration
|
||||
NETCDF_DIR = $(HOME)/local
|
||||
HDF5_DIR = $(HOME)/local
|
||||
NETCDF_OPT = -I$(NETCDF_DIR)/include -I$(HDF5_DIR)/include $(ZLIB_OPT)
|
||||
NETCDF_LIB = -Wl,-rpath,$(NETCDF_DIR)/lib -L$(NETCDF_DIR)/lib\
|
||||
-Wl,-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib\
|
||||
-lnetcdf -lhdf5_hl -lhdf5 $(ZLIB_LIB)
|
||||
NETCDF_DIR = $(HOME)/local
|
||||
HDF5_DIR = $(HOME)/local
|
||||
ZLIB_DIR = $(HOME)/local
|
||||
NETCDF_OPT = -I$(NETCDF_DIR)/include
|
||||
NETCDF_LIB = -L$(NETCDF_DIR)/lib -lnetcdf -L$(HDF5_DIR)/lib -lhdf5_hl -lhdf5\
|
||||
-L$(ZLIB_DIR)/lib -lz
|
||||
|
||||
# PETSc library configuration (version greater or equal to 3.8 or the dev branch)
|
||||
PETSC_ARCH := arch-linux2-c-debug
|
||||
PETSC_DIR := $(MFEM_DIR)/../petsc/$(PETSC_ARCH)
|
||||
PETSC_VARS := $(PETSC_DIR)/lib/petsc/conf/petscvariables
|
||||
PETSC_FOUND := $(if $(wildcard $(PETSC_VARS)),YES,)
|
||||
PETSC_INC_VAR = PETSC_CC_INCLUDES
|
||||
PETSC_LIB_VAR = PETSC_EXTERNAL_LIB_BASIC
|
||||
ifeq ($(PETSC_FOUND),YES)
|
||||
PETSC_OPT := $(shell sed -n "s/$(PETSC_INC_VAR) = *//p" $(PETSC_VARS))
|
||||
PETSC_LIB := $(shell sed -n "s/$(PETSC_LIB_VAR) = *//p" $(PETSC_VARS))
|
||||
PETSC_LIB := -Wl,-rpath,$(abspath $(PETSC_DIR))/lib\
|
||||
-L$(abspath $(PETSC_DIR))/lib -lpetsc $(PETSC_LIB)
|
||||
ifeq ($(MFEM_USE_PETSC),YES)
|
||||
PETSC_DIR := $(MFEM_DIR)/../petsc/arch-linux2-c-debug
|
||||
PETSC_PC := $(PETSC_DIR)/lib/pkgconfig/PETSc.pc
|
||||
$(if $(wildcard $(PETSC_PC)),,$(error PETSc config not found - $(PETSC_PC)))
|
||||
PETSC_OPT := $(shell sed -n "s/Cflags: *//p" $(PETSC_PC))
|
||||
PETSC_LIB := $(shell sed -n "s/Libs.*: *//p" $(PETSC_PC))
|
||||
PETSC_LIB := -Wl,-rpath -Wl,$(abspath $(PETSC_DIR))/lib $(PETSC_LIB)
|
||||
endif
|
||||
|
||||
# MPFR library configuration
|
||||
MPFR_OPT =
|
||||
MPFR_LIB = -lmpfr
|
||||
|
||||
# Conduit and required libraries configuration
|
||||
CONDUIT_DIR = @MFEM_DIR@/../conduit
|
||||
CONDUIT_OPT = -I$(CONDUIT_DIR)/include/conduit
|
||||
CONDUIT_LIB = \
|
||||
-Wl,-rpath,$(CONDUIT_DIR)/lib -L$(CONDUIT_DIR)/lib \
|
||||
-lconduit -lconduit_relay -lconduit_blueprint -ldl
|
||||
|
||||
# Check if Conduit was built with hdf5 support, by looking
|
||||
# for the relay hdf5 header
|
||||
CONDUIT_HDF5_HEADER=$(CONDUIT_DIR)/include/conduit/conduit_relay_hdf5.hpp
|
||||
ifneq (,$(wildcard $(CONDUIT_HDF5_HEADER)))
|
||||
CONDUIT_OPT += -I$(HDF5_DIR)/include
|
||||
CONDUIT_LIB += -Wl,-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib \
|
||||
-lhdf5 $(ZLIB_LIB)
|
||||
endif
|
||||
|
||||
# Sidre and required libraries configuration
|
||||
# Be sure to check the HDF5_DIR (set above) is correct
|
||||
SIDRE_DIR = @MFEM_DIR@/../axom
|
||||
SIDRE_DIR = @MFEM_DIR@/../asctoolkit
|
||||
CONDUIT_DIR = @MFEM_DIR@/../conduit
|
||||
SIDRE_OPT = -I$(SIDRE_DIR)/include -I$(CONDUIT_DIR)/include/conduit\
|
||||
-I$(HDF5_DIR)/include
|
||||
SIDRE_LIB = \
|
||||
-Wl,-rpath,$(SIDRE_DIR)/lib -L$(SIDRE_DIR)/lib \
|
||||
-Wl,-rpath,$(CONDUIT_DIR)/lib -L$(CONDUIT_DIR)/lib \
|
||||
-Wl,-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib \
|
||||
-lsidre -lslic -laxom_utils -lconduit -lconduit_relay -lhdf5 $(ZLIB_LIB) -ldl
|
||||
SIDRE_LIB = -L$(SIDRE_DIR)/lib \
|
||||
-L$(CONDUIT_DIR)/lib \
|
||||
-Wl,-rpath -Wl,$(CONDUIT_DIR)/lib \
|
||||
-L$(HDF5_DIR)/lib\
|
||||
-Wl,-rpath -Wl,$(HDF5_DIR)/lib \
|
||||
-lsidre -lslic -lcommon -lconduit -lconduit_relay -lhdf5 -lz -ldl
|
||||
|
||||
OCCA_DIR = @MFEM_DIR@/../occa
|
||||
OCCA_OPT = -I$(OCCA_DIR)/include
|
||||
OCCA_LIB = -Wl,-rpath,$(OCCA_DIR)/lib -L$(OCCA_DIR)/lib -locca
|
||||
|
||||
CUDA_DIR = /usr/local/cuda
|
||||
CUDAUM_LIB = -L$(CUDA_DIR)/lib64 -lcudart
|
||||
CUDAUM_OPT = -I$(CUDA_DIR)/include
|
||||
|
||||
OMP_OPT = -qsmp=omp -qoffload
|
||||
|
||||
ACROTENSOR_DIR = @MFEM_DIR@/../acrotensor
|
||||
ACROTENSOR_OPT = -std=c++11 -I$(ACROTENSOR_DIR)/inc
|
||||
ACROTENSOR_LIB = -Wl,-rpath,$(ACROTENSOR_DIR)/lib/shared -L$(ACROTENSOR_DIR)/lib/shared -lacrotensor
|
||||
# If Acrotensor was compile with CUDA support, but MFEM_USE_CUDAUM==NO, then uncomment the lines below
|
||||
# ACROTENSOR_OPT += -I$(CUDA_DIR)/include
|
||||
# ACROTENSOR_LIB += -L$(CUDA_DIR)/lib64 -lcuda -lcudart -lnvrtc
|
||||
ifeq ($(MFEM_USE_CUDAUM),YES)
|
||||
ifeq ($(MFEM_USE_MPI),YES)
|
||||
# HYPRE needs some extra libraries in parallel on the GPU
|
||||
# FIXME: We need another solution for compilers other than XL for the
|
||||
# dlink CUDA step, but fixes need to happen elsewhere as well.
|
||||
HYPRE_LIB += -qcuda -lcublas -lcusparse -lnvToolsExt
|
||||
endif
|
||||
SIDRE_LIB += -lspio -lcommon
|
||||
endif
|
||||
|
||||
# If YES, enable some informational messages
|
||||
|
||||
@@ -1,48 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "HYPRE_config.h"
|
||||
#include <cstdio>
|
||||
|
||||
#ifdef HYPRE_RELEASE_VERSION
|
||||
#define HYPRE_VERSION_STRING HYPRE_RELEASE_VERSION
|
||||
#elif defined(HYPRE_PACKAGE_VERSION)
|
||||
#define HYPRE_VERSION_STRING HYPRE_PACKAGE_VERSION
|
||||
#endif
|
||||
|
||||
// Macros to expand a macro as a string
|
||||
#define STR_EXPAND(s) #s
|
||||
#define STR(s) STR_EXPAND(s)
|
||||
|
||||
// Convert the HYPRE_RELEASE_VERSION macro (string) to integer.
|
||||
// Examples: "2.10.0b" --> 21000, "2.11.2" --> 21102
|
||||
int main()
|
||||
{
|
||||
#ifdef HYPRE_VERSION_STRING
|
||||
const char *ptr = STR(HYPRE_VERSION_STRING);
|
||||
if (*ptr == '"') { ptr++; }
|
||||
int version = 0;
|
||||
for (int i = 0; i < 3; i++, ptr++)
|
||||
{
|
||||
int pv = 0;
|
||||
for (char d; d = *ptr, '0' <= d && d <= '9'; ptr++)
|
||||
{
|
||||
pv = 10*pv + (d - '0');
|
||||
if (pv >= 100) { return 1; }
|
||||
}
|
||||
version = 100*version + pv;
|
||||
}
|
||||
printf("%i\n", version);
|
||||
return 0;
|
||||
#else
|
||||
return 2;
|
||||
#endif
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user