Compare commits
7
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e176bb8d06 | ||
|
|
c1509fde91 | ||
|
|
44f25808b9 | ||
|
|
8296bca522 | ||
|
|
d4825be0ca | ||
|
|
ff143322bd | ||
|
|
b08c652314 |
@@ -1,49 +0,0 @@
|
||||
version: '{build}'
|
||||
|
||||
# https://www.appveyor.com/docs/build-environment/#build-worker-images
|
||||
image: Visual Studio 2017
|
||||
|
||||
install:
|
||||
|
||||
# Install MS-MPI
|
||||
- ps: Start-FileDownload 'https://download.microsoft.com/download/B/2/E/B2EB83FE-98C2-4156-834A-E1711E6884FB/MSMpiSetup.exe'
|
||||
- MSMpiSetup.exe -unattend
|
||||
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
|
||||
|
||||
# Install MS-MPI SDK
|
||||
- ps: Start-FileDownload 'https://download.microsoft.com/download/B/2/E/B2EB83FE-98C2-4156-834A-E1711E6884FB/msmpisdk.msi'
|
||||
- msmpisdk.msi /passive
|
||||
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
|
||||
|
||||
# Install METIS
|
||||
- ps: Start-FileDownload 'http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz'
|
||||
- 7z x metis-5.1.0.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd metis-5.1.0
|
||||
- ps: ( get-content "GKlib\gk_arch.h") | % { If ($_.ReadCount -ge 52) {$_ -replace "#ifdef __MSC__","#ifdef DISABLE_THIS_ANCIENT_MSC_CHECK"} Else {$_} } | set-content "GKlib\gk_arch.h"
|
||||
- cmake -H. -Bbuild
|
||||
# -DCMAKE_BUILD_TYPE=Release
|
||||
- cmake --build build
|
||||
- cd ..
|
||||
|
||||
# Install hypre
|
||||
- ps: Start-FileDownload 'https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz'
|
||||
- 7z x hypre-2.10.0b.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd hypre-2.10.0b
|
||||
- cmake -Hsrc -Bbuild -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
# - cmake -Hsrc -Bbuild -DCMAKE_BUILD_TYPE=Release -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
- cmake --build build
|
||||
- cmake --build build --target install
|
||||
- cd ..
|
||||
|
||||
# MFEM
|
||||
before_build:
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2.10.0b\src\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2.10.0b\src\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_serial -DMFEM_USE_MPI=FALSE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2.10.0b\src\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2.10.0b\src\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
|
||||
build_script:
|
||||
- cmake --build build_parallel
|
||||
- cmake --build build_serial
|
||||
|
||||
after_build:
|
||||
# - cmake --build build_parallel --target check
|
||||
- cmake --build build_serial --target check
|
||||
-167
@@ -1,167 +0,0 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# Ignore files that are generated from the repository sources by either building
|
||||
# the code or running it. These should be the same as the files erased by
|
||||
# `make distclean`.
|
||||
#
|
||||
# Also ignore OS-specific files like .DS_Store on Mac
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
# Object and library files
|
||||
*.o
|
||||
/libmfem.*
|
||||
|
||||
# CMake generated files
|
||||
CMakeCache.txt
|
||||
CMakeFiles/
|
||||
|
||||
# Backup files
|
||||
*~
|
||||
|
||||
# Default install location
|
||||
/mfem/
|
||||
|
||||
# Generated files in main directory, config/ and docs/
|
||||
/deps.mk
|
||||
config/_config.hpp
|
||||
config/config.mk
|
||||
config/sample-runs-build.log
|
||||
doc/CodeDocumentation.conf
|
||||
doc/CodeDocumentation.html
|
||||
doc/CodeDocumentation
|
||||
|
||||
# Temporary files created by the tests.
|
||||
*.stderr
|
||||
|
||||
# Totalview breakpoint files
|
||||
*.TVD.*breakpoints
|
||||
|
||||
# OS-specific: Mac
|
||||
*.dSYM
|
||||
.DS_Store
|
||||
|
||||
# Example and miniapp binaries and outputs
|
||||
|
||||
examples/ex[1-9]
|
||||
examples/ex[1-9]p
|
||||
examples/ex1[04-9]
|
||||
examples/ex1[0-9]p
|
||||
|
||||
examples/refined.mesh
|
||||
examples/displaced.mesh
|
||||
examples/mesh.*
|
||||
examples/ex5.mesh
|
||||
examples/Example5*
|
||||
examples/Example9*
|
||||
examples/Example15*
|
||||
examples/Example16*
|
||||
examples/sphere_refined.*
|
||||
examples/sol.*
|
||||
examples/sol_u.*
|
||||
examples/sol_p.*
|
||||
examples/ex9.mesh
|
||||
examples/ex9-mesh.*
|
||||
examples/ex9-init.*
|
||||
examples/ex9-final.*
|
||||
examples/deformed.*
|
||||
examples/velocity.*
|
||||
examples/elastic_energy.*
|
||||
examples/mode_*
|
||||
examples/ex16.mesh
|
||||
examples/ex16-mesh.*
|
||||
examples/ex16-init.*
|
||||
examples/ex16-final.*
|
||||
examples/vortex-mesh.*
|
||||
examples/vortex.mesh
|
||||
examples/vortex-?-init.*
|
||||
examples/vortex-?-final.*
|
||||
examples/deformation.*
|
||||
examples/pressure.*
|
||||
|
||||
examples/sundials/ex9
|
||||
examples/sundials/ex1[06]
|
||||
examples/sundials/ex9p
|
||||
examples/sundials/ex1[06]p
|
||||
|
||||
examples/sundials/ex9.mesh
|
||||
examples/sundials/ex9-mesh.*
|
||||
examples/sundials/ex9-init.*
|
||||
examples/sundials/ex9-final.*
|
||||
examples/sundials/Example9*
|
||||
examples/sundials/deformed.*
|
||||
examples/sundials/velocity.*
|
||||
examples/sundials/elastic_energy.*
|
||||
examples/sundials/ex16.mesh
|
||||
examples/sundials/ex16-mesh.*
|
||||
examples/sundials/ex16-init.*
|
||||
examples/sundials/ex16-final.*
|
||||
examples/sundials/Example16*
|
||||
|
||||
examples/petsc/ex[1-69]p
|
||||
examples/petsc/ex10p
|
||||
|
||||
examples/petsc/mesh.*
|
||||
examples/petsc/sol.*
|
||||
examples/petsc/sol_p.*
|
||||
examples/petsc/sol_u.*
|
||||
examples/petsc/Example5*
|
||||
examples/petsc/ex9-mesh.*
|
||||
examples/petsc/ex9-init.*
|
||||
examples/petsc/ex9-final.*
|
||||
examples/petsc/Example9*
|
||||
examples/petsc/deformed.*
|
||||
examples/petsc/velocity.*
|
||||
examples/petsc/elastic_energy.*
|
||||
|
||||
examples/pumi/ex1
|
||||
examples/pumi/ex[126]p
|
||||
|
||||
examples/pumi/refined.mesh
|
||||
examples/pumi/sol.gf
|
||||
examples/pumi/mesh.*
|
||||
examples/pumi/sol.*
|
||||
examples/pumi/displaced.mesh
|
||||
|
||||
miniapps/electromagnetics/volta
|
||||
miniapps/electromagnetics/tesla
|
||||
miniapps/electromagnetics/maxwell
|
||||
miniapps/electromagnetics/joule
|
||||
|
||||
miniapps/electromagnetics/Volta-AMR*
|
||||
miniapps/electromagnetics/Tesla-AMR*
|
||||
miniapps/electromagnetics/Maxwell-Parallel*
|
||||
miniapps/electromagnetics/Joule_*
|
||||
|
||||
miniapps/meshing/mobius-strip
|
||||
miniapps/meshing/klein-bottle
|
||||
miniapps/meshing/mesh-explorer
|
||||
miniapps/meshing/shaper
|
||||
miniapps/meshing/mesh-optimizer
|
||||
miniapps/meshing/pmesh-optimizer
|
||||
|
||||
miniapps/meshing/mobius-strip.mesh
|
||||
miniapps/meshing/klein-bottle.mesh
|
||||
miniapps/meshing/mesh-explorer.mesh
|
||||
miniapps/meshing/partitioning.txt
|
||||
miniapps/meshing/shaper.mesh
|
||||
miniapps/meshing/optimized*
|
||||
miniapps/meshing/perturbed*
|
||||
|
||||
miniapps/performance/ex1
|
||||
miniapps/performance/ex1p
|
||||
|
||||
miniapps/performance/refined.mesh
|
||||
miniapps/performance/mesh.*
|
||||
miniapps/performance/sol.*
|
||||
|
||||
miniapps/tools/display-basis
|
||||
miniapps/tools/load-dc
|
||||
miniapps/tools/convert-dc
|
||||
|
||||
miniapps/nurbs/ex1
|
||||
miniapps/nurbs/ex1p
|
||||
miniapps/nurbs/ex11p
|
||||
miniapps/nurbs/refined.mesh
|
||||
miniapps/nurbs/mesh.*
|
||||
miniapps/nurbs/sol.*
|
||||
miniapps/nurbs/mode_*
|
||||
miniapps/nurbs/Example1*
|
||||
+67
-230
@@ -1,207 +1,55 @@
|
||||
sudo: false
|
||||
|
||||
language: cpp
|
||||
|
||||
matrix:
|
||||
include:
|
||||
#
|
||||
# Linux
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
env: DEBUG=YES
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
env: DEBUG=NO
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=test
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
addons:
|
||||
apt:
|
||||
# sources:
|
||||
# - ubuntu-toolchain-r-test
|
||||
packages:
|
||||
# GCC 4.9
|
||||
# - g++-4.9
|
||||
# MPICH
|
||||
- mpich
|
||||
- libmpich-dev
|
||||
# OpenMPI
|
||||
# - openmpi-bin
|
||||
# - libopenmpi-dev
|
||||
env: DEBUG=YES
|
||||
MPI=YES
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
NPROCS=2
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
addons:
|
||||
apt:
|
||||
# sources:
|
||||
# - ubuntu-toolchain-r-test
|
||||
packages:
|
||||
# GCC 4.9
|
||||
# - g++-4.9
|
||||
# MPICH
|
||||
- mpich
|
||||
- libmpich-dev
|
||||
# OpenMPI
|
||||
# - openmpi-bin
|
||||
# - libopenmpi-dev
|
||||
env: DEBUG=NO
|
||||
MPI=YES
|
||||
CODECOV=YES
|
||||
MFEM_TEST_TARGET=test
|
||||
NPROCS=2
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
#
|
||||
# Mac OS X
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=YES
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=NO
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=test
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=YES
|
||||
MPI=YES
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
NPROCS=4
|
||||
TMPDIR=/tmp
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
- $HOME/local-cached
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=NO
|
||||
MPI=YES
|
||||
CODECOV=YES
|
||||
MFEM_TEST_TARGET=test
|
||||
NPROCS=4
|
||||
TMPDIR=/tmp
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
- $HOME/local-cached
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
compiler:
|
||||
- gcc
|
||||
- clang
|
||||
|
||||
os:
|
||||
- linux
|
||||
- osx
|
||||
|
||||
env:
|
||||
- DEBUG=YES MPI=YES TMPDIR=/tmp
|
||||
- DEBUG=NO MPI=YES TMPDIR=/tmp
|
||||
- DEBUG=YES MPI=NO
|
||||
- DEBUG=NO MPI=NO
|
||||
|
||||
# Test with GCC on Linux an Clang on Mac
|
||||
matrix:
|
||||
exclude:
|
||||
- compiler: clang
|
||||
os: linux
|
||||
- compiler: gcc
|
||||
os: osx
|
||||
|
||||
before_install:
|
||||
# No addon for brew yet, have to install OSX packages this way.
|
||||
# - if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
|
||||
# brew install open-mpi;
|
||||
# fi
|
||||
|
||||
# On Mac OS X, build and cache OpenMPI 2.1.1:
|
||||
- if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
|
||||
if [ ! -e $HOME/local-cached/bin/mpicc ]; then
|
||||
mkdir -p $HOME/builds && cd $HOME/builds &&
|
||||
wget https://www.open-mpi.org/software/ompi/v2.1/downloads/openmpi-2.1.1.tar.bz2 &&
|
||||
tar jxf openmpi-2.1.1.tar.bz2 &&
|
||||
mkdir openmpi-build && cd openmpi-build &&
|
||||
../openmpi-2.1.1/configure --prefix=$HOME/local-cached &&
|
||||
make -j3 all && make install;
|
||||
fi;
|
||||
PATH=$HOME/local-cached/bin:$PATH;
|
||||
cd $TRAVIS_BUILD_DIR;
|
||||
fi
|
||||
|
||||
# Update environment to find g++ 4.9 installation first.
|
||||
# - if [ $TRAVIS_OS_NAME == "linux" ]; then
|
||||
# mkdir -p latest-gcc-symlinks;
|
||||
# ln -s /usr/bin/g++-4.9 latest-gcc-symlinks/g++;
|
||||
# ln -s /usr/bin/gcc-4.9 latest-gcc-symlinks/gcc;
|
||||
# ln -s /usr/bin/gcov-4.9 latest-gcc-symlinks/gcov;
|
||||
# export PATH=$PWD/latest-gcc-symlinks:$PATH;
|
||||
# fi
|
||||
|
||||
# Install tool to upload code coverage reports to coveralls.io
|
||||
- if [ "$CODECOV" == "YES" ]; then
|
||||
export PYTHONUSERBASE=$HOME/local;
|
||||
pip install --user cpp-coveralls;
|
||||
pip install --user pyyaml;
|
||||
PATH=$HOME/local/bin:$PATH;
|
||||
fi
|
||||
- if [ $TRAVIS_OS_NAME == "linux" -a "$CXX" == "g++" ]; then sudo add-apt-repository -y ppa:ubuntu-toolchain-r/test; fi
|
||||
- if [ $TRAVIS_OS_NAME == "linux" ]; then sudo apt-get update; fi || true
|
||||
|
||||
install:
|
||||
# Set MPI compilers, print compiler version
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ "$TRAVIS_OS_NAME" == "linux" ]; then
|
||||
export MPICH_CC="$CC";
|
||||
export MPICH_CXX="$CXX";
|
||||
else
|
||||
export OMPI_CC="$CC";
|
||||
export OMPI_CXX="$CXX";
|
||||
mpic++ --showme:version;
|
||||
fi;
|
||||
mpic++ -v;
|
||||
else
|
||||
$CXX -v;
|
||||
fi
|
||||
# g++-4.9
|
||||
- if [ $TRAVIS_OS_NAME == "linux" -a "$CXX" == "g++" ]; then sudo apt-get install -qq g++-4.9; fi
|
||||
- if [ $TRAVIS_OS_NAME == "linux" -a "$CXX" == "g++" ]; then export CXX="g++-4.9"; fi
|
||||
|
||||
# Back out of the mfem directory to install the libraries
|
||||
- cd ..
|
||||
|
||||
# OpenMPI
|
||||
- if [ $TRAVIS_OS_NAME == "linux" ]; then
|
||||
sudo apt-get install openmpi-bin openmpi-common openssh-client openssh-server libopenmpi1.3 libopenmpi-dbg libopenmpi-dev;
|
||||
else
|
||||
travis_wait brew install open-mpi;
|
||||
fi
|
||||
|
||||
# hypre
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ ! -e hypre-2.10.0b/src/hypre/lib/libHYPRE.a ]; then
|
||||
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
|
||||
rm -rf hypre-2.10.0b;
|
||||
tar xvzf hypre-2.10.0b.tar.gz;
|
||||
cd hypre-2.10.0b/src;
|
||||
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
|
||||
make -j3;
|
||||
cd ../..;
|
||||
if [ ! -d hypre-2.10.0b ]; then
|
||||
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
|
||||
tar xvzf hypre-2.10.0b.tar.gz;
|
||||
cd hypre-2.10.0b/src;
|
||||
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
|
||||
make -j 4;
|
||||
cd ../..;
|
||||
else
|
||||
echo "Reusing cached hypre-2.10.0b/";
|
||||
fi;
|
||||
@@ -210,54 +58,43 @@ install:
|
||||
fi
|
||||
|
||||
# METIS
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ ! -e metis-4.0/libmetis.a ]; then
|
||||
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
|
||||
tar xvzf metis-4.0.3.tar.gz;
|
||||
make -j3 -C metis-4.0.3/Lib CC="$CC" OPTFLAGS="-O2";
|
||||
rm -rf metis-4.0;
|
||||
mv metis-4.0.3 metis-4.0;
|
||||
else
|
||||
echo "Reusing cached metis-4.0/";
|
||||
fi;
|
||||
- if [ ! -d metis-4.0 ]; then
|
||||
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
|
||||
tar xvzf metis-4.0.3.tar.gz;
|
||||
cd metis-4.0.3;
|
||||
make -j 4;
|
||||
cd ..;
|
||||
mv metis-4.0.3 metis-4.0;
|
||||
else
|
||||
echo "Reusing cached metis-4.0/";
|
||||
fi
|
||||
|
||||
# # Delete an expired cache here: https://travis-ci.org/mfem/mfem/caches
|
||||
# cache:
|
||||
# directories:
|
||||
# - $TRAVIS_BUILD_DIR/../hypre-2.10.0b
|
||||
# - $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
|
||||
script:
|
||||
# Compiler
|
||||
- if [ $MPI == "YES" ]; then
|
||||
export MYCXX=mpic++;
|
||||
export OMPI_CXX="$CXX";
|
||||
$MYCXX --showme:version;
|
||||
else
|
||||
export MYCXX="$CXX";
|
||||
fi
|
||||
|
||||
# Print the compiler version
|
||||
- $MYCXX -v
|
||||
|
||||
# Set some variables
|
||||
- cd $TRAVIS_BUILD_DIR;
|
||||
CPPFLAGS="";
|
||||
SKIP_TEST_DIRS="";
|
||||
if [ "$CODECOV" == "YES" ]; then
|
||||
CPPFLAGS="--coverage -g";
|
||||
fi;
|
||||
if [ "$CXX" == "clang++" ]; then
|
||||
export MFEM_PERF_SW=clang;
|
||||
fi
|
||||
|
||||
# Configure the library
|
||||
- make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG MFEM_CXX="$MYCXX"
|
||||
MFEM_MPI_NP=$NPROCS CPPFLAGS="$CPPFLAGS"
|
||||
# Show the configuration
|
||||
- make info
|
||||
# Build the library
|
||||
- make -j3
|
||||
# Build the examples and the miniapps
|
||||
- make -j3 all
|
||||
# Run tests
|
||||
- make $MFEM_TEST_TARGET SKIP_TEST_DIRS="$SKIP_TEST_DIRS"
|
||||
|
||||
after_success:
|
||||
- if [ "$CODECOV" == "YES" ]; then
|
||||
coveralls --include fem --include general --include linalg --include
|
||||
mesh --exclude /usr --gcov-options '\-lp' --root $TRAVIS_BUILD_DIR;
|
||||
# Build the code and do a quick check (debug mode) or a full tests run (non-debug mode)
|
||||
- if [ $DEBUG == "NO" ]; then
|
||||
export MFEM_TEST_TARGET="test";
|
||||
else
|
||||
export MFEM_TEST_TARGET="check";
|
||||
fi
|
||||
# Build and check/test MFEM, its examples and miniapps
|
||||
- cd $TRAVIS_BUILD_DIR &&
|
||||
make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG MFEM_CXX="$MYCXX" &&
|
||||
make info &&
|
||||
make all -j 4 &&
|
||||
make $MFEM_TEST_TARGET
|
||||
|
||||
@@ -8,296 +8,16 @@
|
||||
http://mfem.org
|
||||
|
||||
|
||||
Version 3.4.1 (development)
|
||||
===========================
|
||||
- Added support for reading linear and quadratic 2D quadrilateral and triangular
|
||||
Cubit meshes.
|
||||
|
||||
|
||||
Version 3.4, released on May 29, 2018
|
||||
Development version, not released yet
|
||||
=====================================
|
||||
|
||||
More general and efficient mesh adaptivity
|
||||
------------------------------------------
|
||||
- Added support for PUMI, the Parallel Unstructured Mesh Infrastructure from
|
||||
https://scorec.rpi.edu/pumi. PUMI is an unstructured, distributed mesh data
|
||||
management system that is capable of handling general non-manifold models and
|
||||
effectively supports automated adaptive analysis. PUMI enables for the first
|
||||
time support for parallel unstructured modifications of MFEM meshes.
|
||||
|
||||
- Significantly reduced MPI communication in the construction of the parallel
|
||||
prolongation matrix in ParFiniteElementSpace, for much improved parallel
|
||||
scaling of non-conforming AMR on hundreds of thousands of MPI tasks. The
|
||||
memory footprint of the ParNCMesh class has also been reduced.
|
||||
|
||||
- In FiniteElementSpace, the fully assembled refinement matrix is now replaced
|
||||
by default by a specialized refinement operator. The operator option is both
|
||||
faster and more memory efficient than using the fully assembled matrix. The
|
||||
old approach is still available and can be enabled, if needed, using the new
|
||||
method FiniteElementSpace::SetUpdateOperatorType().
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Added support for a general "high-order"-to-"low-order refined" transfer of
|
||||
GridFunction and true-dof data from a "high-order" finite element space
|
||||
defined on a coarse mesh, to a "low-order refined" space defined on a refined
|
||||
mesh. The new methods, GetTransferOperator and GetTrueTransferOperator in the
|
||||
FiniteElementSpace classes, work in both serial and parallel and support
|
||||
matrix-based as well as matrix-free transfer operator representations. They
|
||||
use a new method, GetTransferMatrix, in the FiniteElement class similar to
|
||||
GetLocalInterpolation, that allows the coarse FiniteElement to be different
|
||||
from the fine FiniteElement.
|
||||
|
||||
- Added class ComplexOperator, that implements the action of a complex operator
|
||||
through the equivalent 2x2 real formulation. Both symmetric and antisymmetric
|
||||
block structures are supported.
|
||||
|
||||
- Added classes for general block nonlinear finite element operators (deriving
|
||||
from BlockNonlinearForm and ParBlockNonlinearForm) enabling solution of
|
||||
nonlinear systems with multiple unknowns in different function spaces. Such
|
||||
operators have assemble-based action and also support assembly of the gradient
|
||||
operator to enable inversion with Newton iteration.
|
||||
|
||||
- Added variable order NURBS: for each space each knot vector in the mesh can
|
||||
have a different order. The order information is now part of the finite
|
||||
element space header in the NURBS mesh output, so NURBS meshes in the old
|
||||
format need to be updated.
|
||||
|
||||
- In the classes NonlinearForm and ParNonlinearForm, added support for
|
||||
non-conforming AMR meshes; see also the "API changes" section.
|
||||
|
||||
- New specialized time integrators: symplectic integrators of orders 1-4 for
|
||||
systems of first order ODEs derived from a Hamiltonian and generalized-alpha
|
||||
ODE solver for the filtered Navier–Stokes equations with stabilization. See
|
||||
classes SIASolver and GeneralizedAlphaSolver in linalg/ode.hpp.
|
||||
|
||||
- Inherit finite element classes from the new base class TensorBasisElement,
|
||||
whenever the basis can be represented by a tensor product of 1D bases.
|
||||
|
||||
- Added support for elimination of boundary conditions in block matrices.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added a new serial and parallel example (ex19) that solves the quasi-static
|
||||
incompressible hyperelastic equations. The example demonstrates the use of
|
||||
block nonlinear forms as well as custom block preconditioners.
|
||||
|
||||
- Added a new electromagnetics miniapp, Maxwell, for simulating time-domain
|
||||
electromagnetics phenomena as a coupled first order system of equations.
|
||||
|
||||
- A simple local refinement option has been added to the mesh-explorer miniapp
|
||||
(menu option 'r', sub-option 'l') that selects elements for refinement based
|
||||
on their spatial location - see the function 'region()' in the source file.
|
||||
|
||||
- Added a set of miniapps specifically focused on Isogeometric Analysis (IGA) on
|
||||
NURBS meshes in the miniapps/nurbs directory. Currently the directory contains
|
||||
variable order NURBS versions of examples 1, 1p and 11p.
|
||||
|
||||
- Added PUMI versions of examples ex1, ex1p, ex2 and ex6p in a new examples/pumi
|
||||
directory. The new examples demonstrate the PUMI APIs for parallel and serial
|
||||
mesh loading (ex1 and ex1p), applying BCs using classification (ex2), and
|
||||
performing parallel mesh adaptation (ex6p).
|
||||
|
||||
- Added two new miniapps related to DataCollection I/O in miniapps/tools:
|
||||
load-dc.cpp can be used to visualize fields saved via DataCollection classes;
|
||||
convert-dc.cpp demonstrates how to convert between MFEM's different concrete
|
||||
DataCollection options.
|
||||
|
||||
- Example 10p with its SUNDIALS and PETSc versions have been updated to reflect
|
||||
the change in the behavior of the method ParNonlinearForm::GetLocalGradient()
|
||||
(see the "API changes" section) and now works correctly on non-conforming AMR
|
||||
meshes. Example 10 and its SUNDIALS version have also been updated to support
|
||||
non-conforming ARM meshes.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Documented project workflow and provided contribution guidelines in the new
|
||||
top-level file, CONTRIBUTING.md.
|
||||
|
||||
- Added (optional) Conduit Mesh Blueprint support of MFEM data for both in-core
|
||||
and I/O use cases. This includes a new ConduitDataCollection that provides
|
||||
json, simple binary, and HDF5-based I/O. Support requires Conduit >= v0.3.1
|
||||
and VisIt >= v2.13.1 will read the new Data Collection outputs.
|
||||
|
||||
- Added a new developer tool, config/sample-runs.sh, that extracts the sample
|
||||
runs from all examples and miniapps and runs them. Optionally, it can save the
|
||||
output from the execution to files, allowing comparison between different
|
||||
versions and builds of the library.
|
||||
|
||||
- Support for building a shared version of the MFEM library with GNU make.
|
||||
|
||||
- Added a build option, MFEM_USE_EXCEPTIONS=YES, to throw an exception instead
|
||||
of calling abort on mfem errors.
|
||||
|
||||
- When building with the GnuTLS library, switch to using X.509 certificates for
|
||||
secure socket authentication. Support for the previously used OpenPGP keys has
|
||||
been deprecated in GnuTLS 3.5.x and removed in 3.6.0. For secure communication
|
||||
with the visualization tool GLVis, a new set of certificates can be generated
|
||||
using the latest version of the script 'glvis-keygen.sh' from GLVis.
|
||||
|
||||
- Upgraded MFEM to support Axom 0.2.8. Prior versions are no longer supported.
|
||||
|
||||
API changes
|
||||
-----------
|
||||
- Introduced a new enum, Matrix::DiagonalPolicy, that replaces the integer
|
||||
parameters in many methods that perform elimination of rows and/or columns in
|
||||
matrices. Some examples of such methods are:
|
||||
* class SparseMatrix: EliminateRow(), EliminateCol(), EliminateRowCol(), ...
|
||||
* class BilinearForm: EliminateEssentialBC(), EliminateVDofs(), ...
|
||||
* class StaticCondensation: EliminateReducedTrueDofs()
|
||||
* class BlockMatrix: EliminateRowCol()
|
||||
Calling these methods with an explicitly given (integer) constants, will now
|
||||
generate compilation errors, please use one of the new enum constants instead.
|
||||
|
||||
- Modified the virtual method AbstractSparseMatrix::EliminateZeroRows() and its
|
||||
implementations in derived classes, to accept an optional 'threshold'
|
||||
parameter, replacing previously hard-coded threshold values.
|
||||
|
||||
- In the classes NonlinearForm and ParNonlinearForm:
|
||||
* The method GetLocalGradient() no longer imposes boundary conditions. The
|
||||
motivation for the change is that, in the case of non-conforming AMR,
|
||||
performing the elimination at the local level is incorrect - it must be
|
||||
applied at the true-dof level.
|
||||
* The method SetEssentialVDofs() is now deprecated.
|
||||
|
||||
|
||||
Version 3.3.2, released on Nov 10, 2017
|
||||
=======================================
|
||||
|
||||
High-order mesh optimization
|
||||
----------------------------
|
||||
- Added support for mesh optimization via node-movement based on the Target-
|
||||
Matrix Optimization Paradigm (TMOP) developed by P.Knupp et al. A variety of
|
||||
mesh quality metrics, with their first and second derivatives have been
|
||||
implemented. The combination of targets & quality metrics is used to optimize
|
||||
the physical node positions, i.e., they must be as close as possible to the
|
||||
shape, size and/or alignment of their targets. The optimization of arbitrary
|
||||
high-order meshes in 2D, 3D, serial and parallel is supported.
|
||||
|
||||
- The new Mesh Optimizer miniapp can be used to perform mesh optimization with
|
||||
TMOP in serial and parallel versions. The miniapp also demonstrates the use of
|
||||
nonlinear operators and their coupling to Newton methods for solving
|
||||
minimization problems.
|
||||
|
||||
New and improved solvers and preconditioners
|
||||
--------------------------------------------
|
||||
- MFEM is now included in the xSDK project, the Extreme-scale Scientific
|
||||
Software Development Kit, as of xSDK-0.3.0. Various changes were made to
|
||||
comply with xSDK's community policies, https://xsdk.info/policies, including:
|
||||
xSDK-specific options in CMake, support for user-provided MPI communicators,
|
||||
runtime API for version number, and the ability to disable/redirect output.
|
||||
For more details, see general/globals.hpp and in particular the mfem::err and
|
||||
mfem::out streams replacing std::err and std::out respectively.
|
||||
|
||||
- Added (optional) support for the STRUMPACK parallel sparse direct solver and
|
||||
preconditioner. STRUMPACK uses Hierarchically Semi-Separable (HSS) compression
|
||||
in a fully algebraic manner, with interface similar to SuperLU_DIST. See
|
||||
http://portal.nersc.gov/project/sparse/strumpack for more details.
|
||||
|
||||
- Added a block lower triangular preconditioner based (only) on the actions of
|
||||
each block, see class BlockLowerTriangularPreconditioner.
|
||||
|
||||
- Added an optional operator in LOBPCG to projects vectors onto a desired
|
||||
subspace (e.g. divergence-free). Other small changes in LOBPCG include the
|
||||
ability to set the starting vectors and support for relative tolerance.
|
||||
|
||||
- The Newton solver supports an optional scaling factor, that can limit the
|
||||
increment in the Newton step, see e.g. the Mesh Optimizer miniapp.
|
||||
|
||||
- Updated MFEM integration to support the new SUNDIALS 3.0.0 interface.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added a new serial and parallel example (ex18) that solves the transient Euler
|
||||
equations on a periodic domain with explicit time integrators. In the process
|
||||
extended the NonlinearForm class to allow for integrals over faces and
|
||||
exchanging face-neighbor data in parallel.
|
||||
|
||||
- Added a new meshing miniapp, Shaper, that can be used to resolve complicated
|
||||
material interfaces by mesh refinement, e.g. as a tool for initial mesh
|
||||
generation from prescribed "material()" function. Both conforming and
|
||||
non-conforming (isotropic and anisotropic) refinements are supported.
|
||||
|
||||
- Added a new meshing miniapp, Mesh Optimizer, that demonstrates the use of TMOP
|
||||
for mesh optimization (serial and parallel version.)
|
||||
|
||||
- Added SUNDIALS version of Example 16/16p.
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Added a FindPoints method of the Mesh and ParMesh classes that returns the
|
||||
elements that contain a given set of points, together with the coordinates of
|
||||
the points in the reference space of the corresponding element. In parallel,
|
||||
if a point is shared by multiple processors, only one of them will mark that
|
||||
point as found. Note that the current implementation of this method is not
|
||||
optimal and/or 100% reliable. See the mesh-explorer miniapp for an example.
|
||||
|
||||
- Added a new class InverseElementTransformation, that supports a number of
|
||||
algorithms for inversion of general ElementTransformations. This class can be
|
||||
used as a more flexible and extensible alternative to ElementTransformation's
|
||||
TransformBack method. It is also used in the FindPoints methods as a tunable
|
||||
and customizable inversion algorithm.
|
||||
|
||||
- Memory optimizations in the NCMesh class, which now uses 50% less memory than
|
||||
before. The average cost of an element in a uniformly refined mesh (including
|
||||
the refinement hierarchy, but excluding the temporary face_list and edge_list)
|
||||
- Memory optimizations in the NCMesh class, which now uses 50% less memory.
|
||||
The average cost of an NC element in a uniformly refined mesh (including the
|
||||
refinement hierarchy, but excluding the temporary face_list and edge_list)
|
||||
is now only about 290 bytes. This also makes the class faster.
|
||||
|
||||
- Added the ability to integrate delta functions on the right-hand side (by
|
||||
sampling the test function at the center of the delta coefficient). Currently
|
||||
this is supported in the DomainLFIntegrator, VectorDomainLFIntegrator and
|
||||
VectorFEDomainLFIntegrator classes.
|
||||
|
||||
- Added five new linear interpolators in fem/bilininteg.cpp to compute products
|
||||
of scalar and vector fields or products with arbitrary coefficients.
|
||||
|
||||
- Added matrix coefficient support to CurlCurlIntegrator.
|
||||
|
||||
- Extend the method NodalFiniteElement::Project for VectorCoefficient to work
|
||||
with arbitrary number of vector components.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Added a .gitignore file that ignores all files erased by "make distclean",
|
||||
i.e. the files that can be generated from the source but we don't want to
|
||||
track in the repository, as well as a few platform-specific files.
|
||||
|
||||
- Added Linux, Mac and Windows CI testing on GitHub with Travis CI and Appveyor.
|
||||
|
||||
- Added a new macro, MFEM_VERSION, defined as a single integer of the form
|
||||
(major*100 + minor)*100 + patch. The convention is that an even number
|
||||
(i.e. even patch number) denotes a "release" version, while an odd number
|
||||
denotes a "development" version. See config/config.hpp.in.
|
||||
|
||||
- Added an option for building in parallel without a METIS dependency. This is
|
||||
used for example the Laghos miniapp, https://github.com/CEED/Laghos.
|
||||
|
||||
- Modified the installation layout: all headers, except the master headers
|
||||
(mfem.hpp and mfem-performance.hpp), are installed in <PREFIX>/include/mfem;
|
||||
the master headers are installed in both <PREFIX>/include/mfem and in
|
||||
<PREFIX>/include. The mfem configuration and testing makefiles (config.mk and
|
||||
test.mk) are installed in <PREFIX>/share/mfem, instead of <PREFIX>.
|
||||
|
||||
- Add three more options for MFEM_TIMER_TYPE.
|
||||
|
||||
- Support independent number of digits for cycle and rank in DataCollection.
|
||||
|
||||
- Converted Sidre usage from "asctoolkit" to "axom" namespace.
|
||||
|
||||
- Various small fixes and styling updates.
|
||||
|
||||
API changes
|
||||
-----------
|
||||
- The methods GetCoeff of VectorArrayCoefficient and MatrixArrayCoefficient now
|
||||
return a pointer to Coefficient (instead of reference). Note that NULL pointer
|
||||
is a valid entry for these two classes - it is treated as the zero function.
|
||||
|
||||
- When building with PETSc, the required PETSc version is now 3.8.0. Newer
|
||||
versions may work too, as long as there are no interface changes in PETSc.
|
||||
|
||||
- The class GeometryRefiner now uses the enum in Quadrature1D for its type
|
||||
specification. In particular, this will affect older versions of GLVis. A
|
||||
simple upgrade to the latest version of GLVis should resolve this issue.
|
||||
- Add a block lower triangular preconditioner in using a matrix-free
|
||||
implementation, see class BlockLowerTriangularPreconditioner.
|
||||
|
||||
|
||||
Version 3.3, released on Jan 28, 2017
|
||||
@@ -453,7 +173,7 @@ Improved file output
|
||||
- Added experimental support for an HDF5-based output file format following the
|
||||
Conduit (https://github.com/LLNL/conduit) mesh blueprint specification for
|
||||
visualization and/or restart capability. This functionality is aimed primarily
|
||||
at user of LLNL's axom project (Sidre component) that run problems at extreme
|
||||
at user of LLNL's ASC Toolkit (Sidre component) that run problems at extreme
|
||||
scales. Users desiring a small scale binary format may want to look at the
|
||||
gzstream functionality instead.
|
||||
|
||||
|
||||
+46
-149
@@ -38,14 +38,8 @@ if (NOT CMAKE_CXX_COMPILER)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# Project name and version
|
||||
#-------------------------------------------------------------------------------
|
||||
project(mfem NONE)
|
||||
# Current version of MFEM, see also `makefile`.
|
||||
# mfem_VERSION = (string)
|
||||
# MFEM_VERSION = (int) [automatically derived from mfem_VERSION]
|
||||
set(${PROJECT_NAME}_VERSION 3.4.1)
|
||||
project(mfem CXX)
|
||||
set(${PROJECT_NAME}_VERSION 3.3)
|
||||
|
||||
# Prohibit in-source build
|
||||
if (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
|
||||
@@ -53,65 +47,18 @@ if (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
|
||||
"MFEM does not support in-source CMake builds at this time.")
|
||||
endif (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
|
||||
|
||||
# Set xSDK defaults.
|
||||
set(USE_XSDK_DEFAULTS_DEFAULT OFF)
|
||||
set(XSDK_ENABLE_CXX ON)
|
||||
set(XSDK_ENABLE_C OFF)
|
||||
set(XSDK_ENABLE_Fortran OFF)
|
||||
|
||||
# Check if we need to enable C or Fortran.
|
||||
if (CMAKE_VERSION VERSION_LESS 3.2 OR
|
||||
MFEM_USE_CONDUIT OR
|
||||
MFEM_USE_SIDRE OR
|
||||
MFEM_USE_PETSC)
|
||||
if (CMAKE_VERSION VERSION_LESS 3.2 OR MFEM_USE_SIDRE)
|
||||
# This seems to be needed by:
|
||||
# * find_package(BLAS REQUIRED) and
|
||||
# * find_package(HDF5 REQUIRED) needed, in turn, by:
|
||||
# - find_package(AXOM REQUIRED)
|
||||
# * find_package(PETSc REQUIRED)
|
||||
set(XSDK_ENABLE_C ON)
|
||||
endif()
|
||||
if (MFEM_USE_STRUMPACK)
|
||||
# Just needed to find the MPI_Fortran libraries to link with
|
||||
set(XSDK_ENABLE_Fortran ON)
|
||||
endif()
|
||||
|
||||
# Include xSDK default CMake file.
|
||||
include("${CMAKE_CURRENT_SOURCE_DIR}/config/XSDKDefaults.cmake")
|
||||
|
||||
# Enable languages.
|
||||
enable_language(CXX)
|
||||
if (XSDK_ENABLE_C)
|
||||
# - find_package(ATK REQUIRED)
|
||||
enable_language(C)
|
||||
endif()
|
||||
if (XSDK_ENABLE_Fortran)
|
||||
enable_language(Fortran)
|
||||
endif()
|
||||
|
||||
# Suppress warnings about MACOSX_RPATH
|
||||
set(CMAKE_MACOSX_RPATH OFF CACHE BOOL "")
|
||||
|
||||
# CMake needs to know where to find things
|
||||
set(MFEM_CMAKE_PATH ${PROJECT_SOURCE_DIR}/config)
|
||||
set(CMAKE_MODULE_PATH ${MFEM_CMAKE_PATH}/cmake/modules)
|
||||
|
||||
# Load MFEM CMake utilities.
|
||||
include(MfemCmakeUtilities)
|
||||
|
||||
string(TOUPPER "${PROJECT_NAME}" PROJECT_NAME_UC)
|
||||
mfem_version_to_int(${${PROJECT_NAME}_VERSION} ${PROJECT_NAME_UC}_VERSION)
|
||||
set(${PROJECT_NAME_UC}_VERSION_STRING ${${PROJECT_NAME}_VERSION})
|
||||
if (EXISTS ${PROJECT_SOURCE_DIR}/.git)
|
||||
execute_process(
|
||||
COMMAND git describe --all --long --abbrev=40 --dirty --always
|
||||
WORKING_DIRECTORY "${PROJECT_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE ${PROJECT_NAME_UC}_GIT_STRING
|
||||
ERROR_QUIET OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
endif()
|
||||
if (NOT ${PROJECT_NAME_UC}_GIT_STRING)
|
||||
set(${PROJECT_NAME_UC}_GIT_STRING "(unknown)")
|
||||
endif()
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# Process configuration options
|
||||
#-------------------------------------------------------------------------------
|
||||
@@ -123,23 +70,25 @@ else()
|
||||
set(MFEM_DEBUG OFF)
|
||||
endif()
|
||||
|
||||
# MPI -> hypre; PETSc (optional)
|
||||
# MPI -> hypre, METIS
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(MPI REQUIRED)
|
||||
set(MPI_CXX_INCLUDE_DIRS ${MPI_CXX_INCLUDE_PATH})
|
||||
# Parallel MFEM depends on hypre
|
||||
include_directories(${MPI_CXX_INCLUDE_PATH})
|
||||
# Parallel MFEM depends on hypre and METIS
|
||||
find_package(HYPRE REQUIRED)
|
||||
set(MFEM_HYPRE_VERSION ${HYPRE_VERSION})
|
||||
include_directories(${HYPRE_INCLUDE_DIRS})
|
||||
find_package(METIS REQUIRED)
|
||||
include_directories(${METIS_INCLUDE_DIRS})
|
||||
if (MFEM_USE_PETSC)
|
||||
find_package(PETSc REQUIRED)
|
||||
message(STATUS "Found PETSc version ${PETSC_VERSION}")
|
||||
if (PETSC_VERSION AND (PETSC_VERSION VERSION_LESS 3.8.0))
|
||||
message(FATAL_ERROR "PETSc version >= 3.8.0 is required")
|
||||
if (PETSC_VERSION AND (PETSC_VERSION VERSION_LESS 3.7.5.99))
|
||||
message(FATAL_ERROR "PETSc version >= 3.7.5.99 is required")
|
||||
endif()
|
||||
set(PETSC_INCLUDE_DIRS ${PETSC_INCLUDES})
|
||||
include_directories(${PETSC_INCLUDES})
|
||||
endif()
|
||||
else()
|
||||
set(PKGS_NEED_MPI SUPERLU PETSC STRUMPACK PUMI)
|
||||
set(PKGS_NEED_MPI SUPERLU PETSC)
|
||||
foreach(PKG IN LISTS PKGS_NEED_MPI)
|
||||
if (MFEM_USE_${PKG})
|
||||
message(STATUS "Disabling package ${PKG} - requires MPI")
|
||||
@@ -148,19 +97,17 @@ else()
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_METIS)
|
||||
find_package(METIS REQUIRED)
|
||||
endif()
|
||||
|
||||
# GZSTREAM -> zlib
|
||||
if (MFEM_USE_GZSTREAM)
|
||||
find_package(ZLIB REQUIRED)
|
||||
include_directories(${ZLIB_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# Backtrace with libunwind
|
||||
if (MFEM_USE_LIBUNWIND)
|
||||
set(MFEMBacktrace_REQUIRED_PACKAGES "Libunwind" "LIBDL" "CXXABIDemangle")
|
||||
find_package(MFEMBacktrace REQUIRED)
|
||||
include_directories(${LIBUNWIND_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# BLAS, LAPACK
|
||||
@@ -182,6 +129,7 @@ endif()
|
||||
if (MFEM_USE_SUITESPARSE)
|
||||
find_package(SuiteSparse REQUIRED
|
||||
UMFPACK KLU AMD BTF CHOLMOD COLAMD CAMD CCOLAMD config)
|
||||
include_directories(${SuiteSparse_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# SUNDIALS
|
||||
@@ -192,82 +140,63 @@ if (MFEM_USE_SUNDIALS)
|
||||
find_package(SUNDIALS REQUIRED
|
||||
NVector_Serial NVector_Parallel NVector_ParHyp CVODE ARKODE KINSOL)
|
||||
endif()
|
||||
include_directories(${SUNDIALS_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# Mesquite
|
||||
if (MFEM_USE_MESQUITE)
|
||||
find_package(Mesquite REQUIRED)
|
||||
include_directories(${MESQUITE_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# SuperLU_DIST can only be enabled in parallel
|
||||
# SuperLU_DIST can only be enabled if parallel
|
||||
if (MFEM_USE_SUPERLU)
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(SuperLUDist REQUIRED)
|
||||
include_directories(${SuperLUDist_INCLUDE_DIRS})
|
||||
else()
|
||||
message(FATAL_ERROR " *** SuperLU_DIST requires that MPI be enabled.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# STRUMPACK can only be enabled in parallel
|
||||
if (MFEM_USE_STRUMPACK)
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(STRUMPACK REQUIRED)
|
||||
else()
|
||||
message(FATAL_ERROR " *** STRUMPACK requires that MPI be enabled.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Gecko
|
||||
if (MFEM_USE_GECKO)
|
||||
find_package(Gecko REQUIRED)
|
||||
include_directories(${GECKO_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# GnuTLS
|
||||
if (MFEM_USE_GNUTLS)
|
||||
find_package(_GnuTLS REQUIRED)
|
||||
include_directories(${GNUTLS_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# NetCDF
|
||||
if (MFEM_USE_NETCDF)
|
||||
find_package(NetCDF REQUIRED)
|
||||
include_directories(${NETCDF_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# MPFR
|
||||
if (MFEM_USE_MPFR)
|
||||
find_package(MPFR REQUIRED)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_CONDUIT)
|
||||
find_package(Conduit REQUIRED conduit relay blueprint )
|
||||
include_directories(${MPFR_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# Axom/Sidre
|
||||
if (MFEM_USE_SIDRE)
|
||||
find_package(Axom REQUIRED Sidre SLIC axom_utils)
|
||||
endif()
|
||||
|
||||
# PUMI
|
||||
if (MFEM_USE_PUMI)
|
||||
# If PUMI_DIR was specified, only link to that directory,
|
||||
# i.e. don't link to another installation in /usr/lib by mistake
|
||||
find_package(SCOREC 2.1.0 REQUIRED OPTIONAL_COMPONENTS gmi_sim
|
||||
CONFIG PATHS ${PUMI_DIR} NO_DEFAULT_PATH)
|
||||
if (SCOREC_FOUND)
|
||||
# Define a header file with the MFEM_USE_SIMMETRIX preprocessor variable
|
||||
set(MFEM_USE_SIMMETRIX ${SCOREC_gmi_sim_FOUND})
|
||||
set(PUMI_FOUND ${SCOREC_FOUND})
|
||||
get_target_property(PUMI_INCLUDE_DIRS
|
||||
SCOREC::apf INTERFACE_INCLUDE_DIRECTORIES)
|
||||
set(PUMI_LIBRARIES SCOREC::core)
|
||||
if (NOT MFEM_USE_MPI)
|
||||
find_package(ATK REQUIRED Sidre SLIC common)
|
||||
else()
|
||||
find_package(ATK REQUIRED Sidre SPIO SLIC common)
|
||||
endif()
|
||||
include_directories(${ATK_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
# MFEM_TIMER_TYPE
|
||||
if (NOT DEFINED MFEM_TIMER_TYPE)
|
||||
if (APPLE)
|
||||
# use std::clock from <ctime> for UserTime and
|
||||
# use mach_absolute_time from <mach/mach_time.h> for RealTime
|
||||
set(MFEM_TIMER_TYPE 4)
|
||||
set(MFEM_TIMER_TYPE 0) # use std::clock from <ctime>
|
||||
elseif (WIN32)
|
||||
set(MFEM_TIMER_TYPE 3) # QueryPerformanceCounter from <windows.h>
|
||||
else()
|
||||
@@ -281,27 +210,17 @@ if (NOT DEFINED MFEM_TIMER_TYPE)
|
||||
endif()
|
||||
|
||||
# List all possible libraries in order of dependencies.
|
||||
# [METIS < SuiteSparse]:
|
||||
# With newer versions of SuiteSparse which include METIS header using 64-bit
|
||||
# integers, the METIS header (with 32-bit indices, as used by mfem) needs to
|
||||
# be before SuiteSparse.
|
||||
set(MFEM_TPLS MPI_CXX OPENMP BLAS LAPACK METIS HYPRE SuiteSparse SUNDIALS PETSC
|
||||
MESQUITE SuperLUDist STRUMPACK AXOM CONDUIT GECKO GNUTLS NETCDF MPFR PUMI
|
||||
POSIXCLOCKS MFEMBacktrace ZLIB)
|
||||
set(MFEM_TPLS HYPRE OPENMP SUNDIALS MESQUITE SuiteSparse SuperLUDist
|
||||
ParMETIS METIS LAPACK BLAS GECKO GNUTLS NETCDF PETSC MPFR ATK POSIXCLOCKS
|
||||
MFEMBacktrace ZLIB)
|
||||
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
|
||||
set(TPL_LIBRARIES "")
|
||||
set(TPL_INCLUDE_DIRS "")
|
||||
foreach(TPL IN LISTS MFEM_TPLS)
|
||||
if (${TPL}_FOUND)
|
||||
message(STATUS "MFEM: using package ${TPL}")
|
||||
list(APPEND TPL_LIBRARIES ${${TPL}_LIBRARIES})
|
||||
list(APPEND TPL_INCLUDE_DIRS ${${TPL}_INCLUDE_DIRS})
|
||||
endif()
|
||||
endforeach(TPL)
|
||||
list(REMOVE_DUPLICATES TPL_LIBRARIES)
|
||||
list(REMOVE_DUPLICATES TPL_INCLUDE_DIRS)
|
||||
# message(STATUS "TPL_INCLUDE_DIRS = ${TPL_INCLUDE_DIRS}")
|
||||
include_directories(${TPL_INCLUDE_DIRS})
|
||||
|
||||
if (OPENMP_FOUND)
|
||||
message(STATUS "MFEM: using package OpenMP")
|
||||
@@ -309,8 +228,6 @@ if (OPENMP_FOUND)
|
||||
endif()
|
||||
|
||||
message(STATUS "MFEM build type: CMAKE_BUILD_TYPE = ${CMAKE_BUILD_TYPE}")
|
||||
message(STATUS "MFEM version: v${MFEM_VERSION_STRING}")
|
||||
message(STATUS "MFEM git string: ${MFEM_GIT_STRING}")
|
||||
|
||||
# Windows specific
|
||||
set(_USE_MATH_DEFINES ${WIN32})
|
||||
@@ -320,6 +237,7 @@ set(_USE_MATH_DEFINES ${WIN32})
|
||||
#-------------------------------------------------------------------------------
|
||||
|
||||
# Headers and sources
|
||||
include(MfemCmakeUtilities)
|
||||
set(SOURCES "")
|
||||
set(HEADERS "")
|
||||
set(MFEM_SOURCE_DIRS general linalg mesh fem)
|
||||
@@ -331,30 +249,21 @@ set(MASTER_HEADERS
|
||||
${PROJECT_SOURCE_DIR}/mfem.hpp
|
||||
${PROJECT_SOURCE_DIR}/mfem-performance.hpp)
|
||||
|
||||
set(_lib_path "${CMAKE_INSTALL_PREFIX}/lib")
|
||||
set(CMAKE_INSTALL_RPATH_USE_LINK_PATH ON CACHE BOOL "")
|
||||
set(CMAKE_INSTALL_RPATH "${_lib_path}" CACHE PATH "")
|
||||
set(CMAKE_INSTALL_NAME_DIR "${_lib_path}" CACHE PATH "")
|
||||
|
||||
# Declaring the library
|
||||
add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
|
||||
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
|
||||
list(REMOVE_DUPLICATES TPL_LIBRARIES)
|
||||
# message(STATUS " TPL_LIBRARIES = ${TPL_LIBRARIES}")
|
||||
if (CMAKE_VERSION VERSION_GREATER 2.8.11)
|
||||
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES})
|
||||
else()
|
||||
target_link_libraries(mfem ${TPL_LIBRARIES})
|
||||
endif()
|
||||
if (MINGW)
|
||||
target_link_libraries(mfem ws2_32)
|
||||
endif()
|
||||
set_target_properties(mfem PROPERTIES VERSION "${mfem_VERSION}")
|
||||
set_target_properties(mfem PROPERTIES SOVERSION "${mfem_VERSION}")
|
||||
|
||||
# If building out-of-source, define MFEM_BUILD_DIR to point to the build
|
||||
# directory.
|
||||
if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
|
||||
target_compile_definitions(mfem PRIVATE
|
||||
"MFEM_BUILD_DIR=${PROJECT_BINARY_DIR}")
|
||||
"-DMFEM_BUILD_DIR=${PROJECT_BINARY_DIR}")
|
||||
endif()
|
||||
|
||||
# Generate configuration file in the build directory: config/_config.hpp.
|
||||
@@ -372,11 +281,6 @@ if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
|
||||
"// Auto-generated file.
|
||||
#define MFEM_BUILD_DIR ${PROJECT_BINARY_DIR}
|
||||
#include \"${PROJECT_SOURCE_DIR}/${Header}\"
|
||||
")
|
||||
# This version will be installed in the top include directory:
|
||||
file(WRITE "${PROJECT_BINARY_DIR}/InstallHeaders/${Header}"
|
||||
"// Auto-generated file.
|
||||
#include \"mfem/${Header}\"
|
||||
")
|
||||
endforeach()
|
||||
endif()
|
||||
@@ -428,12 +332,12 @@ endif()
|
||||
# Add 'check' target - quick test
|
||||
if (NOT MFEM_USE_MPI)
|
||||
add_custom_target(check
|
||||
${CMAKE_CTEST_COMMAND} -R '^ex1_ser' -C ${CMAKE_CFG_INTDIR}
|
||||
${CMAKE_CTEST_COMMAND} -R ex1_ser -E performance -C ${CMAKE_CFG_INTDIR}
|
||||
USES_TERMINAL)
|
||||
add_dependencies(check ex1)
|
||||
else()
|
||||
add_custom_target(check
|
||||
${CMAKE_CTEST_COMMAND} -R '^ex1p' -C ${CMAKE_CFG_INTDIR}
|
||||
${CMAKE_CTEST_COMMAND} -R ex1p -E performance -C ${CMAKE_CFG_INTDIR}
|
||||
USES_TERMINAL)
|
||||
add_dependencies(check ex1p)
|
||||
endif()
|
||||
@@ -448,6 +352,7 @@ add_subdirectory(doc)
|
||||
#-------------------------------------------------------------------------------
|
||||
|
||||
message(STATUS "CMAKE_INSTALL_PREFIX = ${CMAKE_INSTALL_PREFIX}")
|
||||
string(TOUPPER "${PROJECT_NAME}" PROJECT_NAME_UC)
|
||||
set(INSTALL_INCLUDE_DIR include
|
||||
CACHE PATH "Relative path for installing header files.")
|
||||
set(INSTALL_LIB_DIR lib
|
||||
@@ -467,15 +372,11 @@ install(TARGETS ${PROJECT_NAME}
|
||||
DESTINATION ${INSTALL_LIB_DIR})
|
||||
|
||||
# Install the master headers
|
||||
foreach(Header mfem.hpp mfem-performance.hpp)
|
||||
install(FILES ${PROJECT_BINARY_DIR}/InstallHeaders/${Header}
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR})
|
||||
endforeach()
|
||||
install(FILES ${MASTER_HEADERS} DESTINATION ${INSTALL_INCLUDE_DIR}/mfem)
|
||||
install(FILES ${MASTER_HEADERS} DESTINATION ${INSTALL_INCLUDE_DIR})
|
||||
|
||||
# Install the headers; currently, the miniapps headers are excluded
|
||||
install(DIRECTORY ${MFEM_SOURCE_DIRS}
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}
|
||||
FILES_MATCHING PATTERN "*.hpp")
|
||||
|
||||
# Install ${HEADERS}
|
||||
@@ -488,11 +389,11 @@ install(DIRECTORY ${MFEM_SOURCE_DIRS}
|
||||
|
||||
# Install the configuration header files
|
||||
install(FILES ${PROJECT_BINARY_DIR}/config/_config.hpp
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem/config
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/config
|
||||
RENAME config.hpp)
|
||||
|
||||
install(FILES ${PROJECT_SOURCE_DIR}/config/tconfig.hpp
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem/config)
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/config)
|
||||
|
||||
# Package the whole thing up nicely
|
||||
include(CMakePackageConfigHelpers)
|
||||
@@ -502,15 +403,11 @@ export(TARGETS ${PROJECT_NAME}
|
||||
FILE "${PROJECT_BINARY_DIR}/MFEMTargets.cmake")
|
||||
|
||||
# Export the package for use from the build-tree (this registers the build-tree
|
||||
# with the CMake user package registry.)
|
||||
# TODO: How do we register the install-tree? Replacing the build-tree?
|
||||
# with a global CMake-registry)
|
||||
export(PACKAGE ${PROJECT_NAME})
|
||||
|
||||
# Extract the include directories required to use MFEM
|
||||
get_target_property(MFEM_TPL_INCLUDE_DIRS mfem INCLUDE_DIRECTORIES)
|
||||
if (NOT MFEM_TPL_INCLUDE_DIRS)
|
||||
set(MFEM_TPL_INCLUDE_DIRS "")
|
||||
endif()
|
||||
|
||||
# This is the build-tree version
|
||||
set(INCLUDE_INSTALL_DIRS ${PROJECT_BINARY_DIR} ${MFEM_TPL_INCLUDE_DIRS})
|
||||
|
||||
-538
@@ -1,538 +0,0 @@
|
||||
<p align="center">
|
||||
<a href="http://mfem.org/"><img alt="mfem" src="http://mfem.org/img/logo-300.png"></a>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="https://github.com/mfem/mfem/blob/master/COPYRIGHT"><img alt="License" src="https://img.shields.io/badge/License-LGPL--2.1-brightgreen.svg"></a>
|
||||
<a href="https://travis-ci.org/mfem/mfem"><img alt="Build Status" src="https://travis-ci.org/mfem/mfem.svg?branch=master"></a>
|
||||
<a href="https://ci.appveyor.com/project/mfem/mfem"><img alt="Build Status" src="https://ci.appveyor.com/api/projects/status/19non9sqm6msi2wy?svg=true"></a>
|
||||
<a href="http://mfem.github.io/doxygen/html/index.html"><img alt="Doxygen" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
|
||||
</p>
|
||||
|
||||
|
||||
# How to Contribute
|
||||
|
||||
The MFEM team welcomes contributions at all levels: bugfixes; code
|
||||
improvements; simplifications; new mesh, discretization or solver
|
||||
capabilities; improved documentation; new examples and miniapps;
|
||||
HPC performance improvements; ...
|
||||
|
||||
Use a pull request (PR) toward the `mfem:master` branch to propose your
|
||||
contribution. If you are planning significant code changes, or have any
|
||||
questions, you can also open an [issue](https://github.com/mfem/mfem/issues)
|
||||
before issuing a PR. We also welcome your [simulation
|
||||
images](http://mfem.org/gallery/), which you can submit via a pull request in
|
||||
[mfem/web](https://github.com/mfem/web).
|
||||
|
||||
See the [Quick Summary](#quick-summary) section for the main highlights of our
|
||||
GitHub workflow. For more details, consult the following sections and refer
|
||||
back to them before issuing pull requests:
|
||||
|
||||
- [Code Overview](#code-overview)
|
||||
- [GitHub Workflow](#github-workflow)
|
||||
- [MFEM Organization](#mfem-organization)
|
||||
- [New Feature Development](#new-feature-development)
|
||||
- [Developer Guidelines](#developer-guidelines)
|
||||
- [Pull Requests](#pull-requests)
|
||||
- [Pull Request Checklist](#pull-request-checklist)
|
||||
- [Master/Next Workflow](#masternext-workflow)
|
||||
- [Releases](#releases)
|
||||
- [Release Checklist](#release-checklist)
|
||||
- [LLNL Workflow](#llnl-workflow)
|
||||
- [Automated Testing](#automated-testing)
|
||||
- [Contact Information](#contact-information)
|
||||
|
||||
Contributing to MFEM requires knowledge of Git and, likely, finite elements. If
|
||||
you are new to Git, see the [GitHub learning
|
||||
resources](https://help.github.com/articles/git-and-github-learning-resources/).
|
||||
To learn more about the finite element method, see our [FEM page](http://mfem.org/fem).
|
||||
|
||||
*By submitting a pull request, you are affirming the [Developer's Certificate of
|
||||
Origin](#developers-certificate-of-origin-11) at the end of this file.*
|
||||
|
||||
|
||||
## Quick Summary
|
||||
|
||||
- We encourage you to [join the MFEM organization](#mfem-organization) and create
|
||||
development branches off `mfem:master`.
|
||||
- Please follow the [developer guidelines](#developer-guidelines), in particular
|
||||
with regards to documentation and code styling.
|
||||
- Pull requests should be issued toward `mfem:master`. Make sure
|
||||
to check the items off the [Pull Request Checklist](#pull-request-checklist).
|
||||
- After approval, MFEM developers merge the PR manually in the [mfem:next branch](#masternext-workflow).
|
||||
- After a week of testing in `mfem:next`, the original PR is merged in `mfem:master`.
|
||||
- We use [milestones](https://github.com/mfem/mfem/milestones) to coordinate the
|
||||
work on different PRs toward a release.
|
||||
- Don't hesitate to [contact us](#contact-information) if you have any questions.
|
||||
|
||||
|
||||
### Code Overview
|
||||
|
||||
- The MFEM library uses object-orient design principles which reflect, in code,
|
||||
the independent mathematical concepts of meshing, linear algebra and finite
|
||||
element spaces and operators.
|
||||
|
||||
- The MFEM source code has the following structure:
|
||||
```
|
||||
.
|
||||
├── config
|
||||
│ └── cmake
|
||||
│ └── modules
|
||||
├── data
|
||||
├── doc
|
||||
│ └── web
|
||||
│ └── examples
|
||||
├── examples
|
||||
│ ├── petsc
|
||||
│ ├── pumi
|
||||
│ └── sundials
|
||||
├── fem
|
||||
├── general
|
||||
├── linalg
|
||||
├── mesh
|
||||
└── miniapps
|
||||
├── common
|
||||
├── electromagnetics
|
||||
├── meshing
|
||||
├── nurbs
|
||||
├── performance
|
||||
└── tools
|
||||
```
|
||||
|
||||
- The main directories are `fem/`, `mesh/` and `linalg/` containing the C++
|
||||
classes implementing the finite element, mesh and linear algebra concepts
|
||||
respectively.
|
||||
|
||||
- The main mesh classes are:
|
||||
+ [`Mesh`](http://mfem.github.io/doxygen/html/classmfem_1_1Mesh.html)
|
||||
+ [`NCMesh`](http://mfem.github.io/doxygen/html/classmfem_1_1NCMesh.html)
|
||||
+ [`Element`](http://mfem.github.io/doxygen/html/classmfem_1_1Element.html)
|
||||
+ [`ElementTransformation`](http://mfem.github.io/doxygen/html/classmfem_1_1ElementTransformation.html)
|
||||
|
||||
- The main finite element classes are:
|
||||
+ [`FiniteElement`](http://mfem.github.io/doxygen/html/classmfem_1_1FiniteElement.html)
|
||||
+ [`FiniteElementCollection`](http://mfem.github.io/doxygen/html/classmfem_1_1FiniteElement.html)
|
||||
+ [`FiniteElementSpace`](http://mfem.github.io/doxygen/html/classmfem_1_1FiniteElementSpace.html)
|
||||
+ [`GridFunction`](http://mfem.github.io/doxygen/html/classmfem_1_1GridFunction.html)
|
||||
+ [`BilinearFormIntegrator`](http://mfem.github.io/doxygen/html/classmfem_1_1BilinearFormIntegrator.html) and [`LinearFormIntegrator`](http://mfem.github.io/doxygen/html/classmfem_1_1LinearFormIntegrator.html)
|
||||
+ [`LinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1LinearFormIntegrator.html), [`BilinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1MixedBilinearForm.html)
|
||||
|
||||
- The main linear algebra classes and sources are
|
||||
+ [`Operator`](http://mfem.github.io/doxygen/html/classmfem_1_1Operator.html) and [`BilinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1BilinearForm.html)
|
||||
+ [`Vector`](http://mfem.github.io/doxygen/html/classmfem_1_1BilinearForm.html) and [`LinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1LinearForm.html)
|
||||
+ [`DenseMatrix`](http://mfem.github.io/doxygen/html/classmfem_1_1DenseMatrix.html) and [`SparseMatrix`](http://mfem.github.io/doxygen/html/classmfem_1_1SparseMatrix.html)
|
||||
+ Sparse [smoothers](http://mfem.github.io/doxygen/html/sparsesmoothers_8hpp.html) and linear [solvers](http://mfem.github.io/doxygen/html/solvers_8hpp.html)
|
||||
|
||||
- Parallel MPI objects in MFEM inherit their serial counterparts, so a parallel
|
||||
mesh for example is just a serial mesh on each task plus the information on
|
||||
shared geometric entities between different tasks. The parallel source files
|
||||
have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
|
||||
|
||||
- The main parallel classes are
|
||||
+ [`ParMesh`](http://mfem.github.io/doxygen/html/solvers_8hpp.html)
|
||||
+ [`ParNCMesh`](http://mfem.github.io/doxygen/html/classmfem_1_1ParMesh.html)
|
||||
+ [`ParFiniteElementSpace`](http://mfem.github.io/doxygen/html/classmfem_1_1ParFiniteElementSpace.html)
|
||||
+ [`ParGridFunction`](http://mfem.github.io/doxygen/html/classmfem_1_1ParGridFunction.html)
|
||||
+ [`ParBilinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1ParBilinearForm.html) and [`ParLinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1ParLinearForm.html)
|
||||
+ [`HypreParMatrix`](http://mfem.github.io/doxygen/html/classmfem_1_1HypreParMatrix.html) and [`HypreParVector`](http://mfem.github.io/doxygen/html/classmfem_1_1HypreParVector.html)
|
||||
+ [`HypreSolver`](http://mfem.github.io/doxygen/html/classmfem_1_1HypreSolver.html) and other [hypre classes](http://mfem.github.io/doxygen/html/hypre_8hpp.html)
|
||||
|
||||
- The `general/` directory contains C++ classes that serve as utilities for
|
||||
communication, error handling, arrays, (Boolean) tables, timing, etc.
|
||||
|
||||
- The `config/` directory contains build-related files, both for the plain
|
||||
Makefile and the CMake build options.
|
||||
|
||||
- The `doc/` directory contains configuration for the Doxygen code documentation
|
||||
that can either be build locally, or browsed online at
|
||||
http://mfem.github.io/doxygen/html/index.html.
|
||||
|
||||
- The `data/` directory contains a collection of small mesh files, that are used
|
||||
in the simple example codes and more fully-featured mini applications in the
|
||||
`examples/` and `miniapps/` directories.
|
||||
|
||||
- See also the [code overview](http://mfem.org/code-overview/) section on the
|
||||
MFEM website.
|
||||
|
||||
## GitHub Workflow
|
||||
|
||||
The GitHub organization, https://github.com/mfem, is the main developer hub for
|
||||
the MFEM project.
|
||||
|
||||
If you plan to make contributions or will like to stay up-to-date with changes
|
||||
in the code, *we strongly encourage you to [join the MFEM organization](#mfem-organization)*.
|
||||
|
||||
This will simplify the workflow (by providing you additional permissions), and
|
||||
will allow us to reach you directly with project announcements.
|
||||
|
||||
|
||||
### MFEM Organization
|
||||
|
||||
- Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
+ Create the account at: github.com/join.
|
||||
+ For easy identification, please add your name and maybe a picture of you at: https://github.com/settings/profile.
|
||||
+ To receive notification, set a primary email at: https://github.com/settings/emails.
|
||||
+ For password-less pull/push over SSH, add your SSH keys at: https://github.com/settings/keys.
|
||||
|
||||
- [Contact us](#contact-information) for an invitation to join the MFEM GitHub
|
||||
organization.
|
||||
|
||||
- You should receive an invitation email, which you can directly accept.
|
||||
Alternatively, *after logging into GitHub*, you can accept the invitation at
|
||||
the top of https://github.com/mfem.
|
||||
|
||||
- Consider making your membership public by going to https://github.com/orgs/mfem/people
|
||||
and clicking on the organization visibility dropbox next to your name.
|
||||
|
||||
- Project discussions and announcements will be posted at
|
||||
https://github.com/orgs/mfem/teams/everyone.
|
||||
|
||||
- The MFEM source code is in the [mfem](https://github.com/mfem/mfem)
|
||||
repository.
|
||||
|
||||
- The website and corresponding documentation are in the
|
||||
[web](https://github.com/mfem/web) repository.
|
||||
|
||||
- The [PyMFEM](https://github.com/mfem/PyMFEM) repository contains a Python
|
||||
wrapper for MFEM.
|
||||
|
||||
- The [data](https://github.com/mfem/data) repository contains additional
|
||||
(large) datafiles for MFEM.
|
||||
|
||||
|
||||
### New Feature Development
|
||||
|
||||
- A new feature should be important enough that at least one person, the
|
||||
proposer, is willing to work on it and be its champion.
|
||||
|
||||
- The proposer creates a branch for the new feature (with suffix `-dev`), off
|
||||
the `master` branch, or another existing feature branch, for example:
|
||||
|
||||
```
|
||||
# Clone assuming you have setup your ssh keys on GitHub:
|
||||
git clone git@github.com:mfem/mfem.git
|
||||
|
||||
# Alternatively, clone using the "https" protocol:
|
||||
git clone https://github.com/mfem/mfem.git
|
||||
|
||||
# Create a new feature branch starting from "master":
|
||||
git checkout master
|
||||
git pull
|
||||
git checkout -b feature-dev
|
||||
|
||||
# Work on "feature-dev", add local commits
|
||||
# ...
|
||||
|
||||
# (One time only) push the branch to github and setup your local
|
||||
# branch to track the github branch (for "git pull"):
|
||||
git push -u origin feature-dev
|
||||
|
||||
```
|
||||
|
||||
- **We prefer that you create the new feature branch inside the MFEM organization
|
||||
as opposed to in a fork.** This allows everyone in the community to collaborate
|
||||
in one central place.
|
||||
|
||||
- If you prefer to work in your fork, please [enable upstream edits](https://help.github.com/articles/allowing-changes-to-a-pull-request-branch-created-from-a-fork/).
|
||||
|
||||
- Never use the `next` branch to start a new feature branch!
|
||||
|
||||
- The typical feature branch name is `new-feature-dev`, e.g. `pumi-dev`. While
|
||||
not frequent in MFEM, other suffixes are possible, e.g. `-fix`, `-doc`, etc.
|
||||
|
||||
|
||||
### Developer Guidelines
|
||||
|
||||
- *Keep the code lean and as simple as possible*
|
||||
- Well-designed simple code is frequently more general and powerful.
|
||||
- Lean code base is easier to understand by new collaborators.
|
||||
- New features should be added only if they are necessary or generally useful.
|
||||
- Introduction of language constructions not currently used in MFEM should be
|
||||
justified and generally avoided (so we can build on cutting-edge systems).
|
||||
- We prefer basic C++ and the C++03 standard, to keep the code readable by
|
||||
a large audience and to make sure it compiles anywhere.
|
||||
|
||||
- *Keep the code general and reasonably efficient*
|
||||
- Main goal is fast prototyping for research.
|
||||
- When in doubt, generality wins over efficiency.
|
||||
- Respect the needs of different users (current and/or future).
|
||||
|
||||
- *Keep things separate and logically organized*
|
||||
- General usage features go in MFEM (implemented in as much generality as
|
||||
possible), non-general features go into external apps.
|
||||
- Inside MFEM, compartmentalize between linalg, fem, mesh, GLVis, etc.
|
||||
- Contributions that are project-specific or have external dependencies are
|
||||
allowed (if they are of broader interest), but should be `#ifdef`-ed and not
|
||||
change the code by default.
|
||||
|
||||
- Code specifics
|
||||
- All significant new classes, methods and functions have Doxygen-style
|
||||
documentation in source comments.
|
||||
- Consistent code styling is enforced with `make style` in the top-level
|
||||
directory. This requires [Artistic Style](http://astyle.sourceforge.net) (we
|
||||
specifically use version 2.05.1). See also the file `config/mfem.astylerc`.
|
||||
- Use `mfem::out` and `mfem::err` instead of `std::cout` and `std::cerr` in
|
||||
internal library code. (You can use `std` in examples and miniapps.)
|
||||
- When manually resolving conflicts during a merge, make sure to mention the
|
||||
conflicted files in the commit message.
|
||||
|
||||
### Pull Requests
|
||||
|
||||
- When your branch is ready for other developers to review / comment on
|
||||
the code, create a pull request towards `mfem:master`.
|
||||
|
||||
- Pull request typically have titles like:
|
||||
|
||||
`Description [new-feature-dev]`
|
||||
|
||||
for example:
|
||||
|
||||
`Parallel Unstructured Mesh Infrastructure (PUMI) integration [pumi-dev]`
|
||||
|
||||
Note the branch name suffix (in square brackets).
|
||||
|
||||
- Titles may contain a prefix in square brackets to emphasize the type of PR.
|
||||
Common choices are: `[DON'T MERGE]`, `[WIP]` and `[DISCUSS]`, for example:
|
||||
|
||||
`[DISCUSS] Hybridized DG [hdg-dev]`
|
||||
|
||||
- Add a description, appropriate labels and assign yourself to the PR. The MFEM
|
||||
team will add reviewers as appropriate.
|
||||
|
||||
- List outstanding TODO items in the description, see PR #222 for an example.
|
||||
|
||||
- Track the Travis CI and Appveyor [continuous integration](#automated-testing)
|
||||
builds at the end of the PR. These should run clean, so address any errors as
|
||||
soon as possible.
|
||||
|
||||
|
||||
### Pull Request Checklist
|
||||
|
||||
Before a PR can be merged, it should satisfy the following:
|
||||
|
||||
- [ ] Code builds.
|
||||
- [ ] Code passes `make style`.
|
||||
- [ ] Update `CHANGELOG`:
|
||||
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
|
||||
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
|
||||
- [ ] Update `INSTALL`:
|
||||
- [ ] Has a new optional library been added? (*Make sure the external library is licensed under LGPL, not GPL!*)
|
||||
- [ ] Does `make` or `cmake` have a new target?
|
||||
- [ ] Did the requirements or the installation process change? *(rare)*.
|
||||
- [ ] Update `.gitignore`:
|
||||
- [ ] Check if `make distclean; git status` shows any files that are generated from the source but we don't want to track in the repository.
|
||||
- [ ] Add new patterns (just for the new files above) and re-run the above test.
|
||||
- [ ] New examples:
|
||||
- [ ] All sample runs at the top of the example work.
|
||||
- [ ] Update `examples/makefile`:
|
||||
- [ ] Add the example code to the appropriate `SEQ_EXAMPLES` and `PAR_EXAMPLES` variables.
|
||||
- [ ] Add any files generated by it to the `clean` target.
|
||||
- [ ] Add the example binary and any files generated by it to the top-level `.gitignore` file.
|
||||
- [ ] Update `examples/CMakeLists.txt`:
|
||||
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
|
||||
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
|
||||
- [ ] List the new example in `doc/CodeDocumentation.dox`.
|
||||
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
|
||||
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
- [ ] In `examples.md`, list the example under the appropriate categories, add new categories if necessary.
|
||||
- [ ] Add a short description of the example in the "Extensive Examples" section of `features.md`.
|
||||
- [ ] New miniapps:
|
||||
- [ ] All sample runs at the top of the miniapp work.
|
||||
- [ ] Update top-level `makefile` and `makefile` in corresponding miniapp directory.
|
||||
- [ ] Add the miniapp binary and any files generated by it to the top-level `.gitignore` file.
|
||||
- [ ] Update CMake build system:
|
||||
- [ ] Update the `CMakeLists.txt` file in the `miniapps` directory, if the new miniapp is in a new directory.
|
||||
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
|
||||
- [ ] Consider adding a new test for the new miniapp.
|
||||
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
|
||||
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
|
||||
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
|
||||
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
|
||||
- [ ] New capability:
|
||||
- [ ] All significant new classes, methods and functions have Doxygen-style documentation in source comments.
|
||||
- [ ] Consider adding new sample runs in existing examples to highlight the new capability.
|
||||
- [ ] Consider saving cool simulation pictures with the new capability in the Confluence gallery (LLNL only) or submitting them, via pull request, to the gallery section of the `mfem/web` repo.
|
||||
- [ ] If this is a major new feature, consider mentioning in the short summary inside `README` *(rare)*.
|
||||
- [ ] List major new classes in `doc/CodeDocumentation.dox` *(rare)*.
|
||||
- [ ] Update this checklist, if the new pull request affects it.
|
||||
- [ ] (LLNL only) Clone the `tests` repository and run the following tests, see `mfem/tests/README.md`:
|
||||
- [ ] `compilers`
|
||||
- [ ] `memcheck`
|
||||
- [ ] `unit-test`
|
||||
- [ ] `documentation`
|
||||
- [ ] (LLNL only) After merging:
|
||||
- [ ] Regenerate `README.html` files from companion documentation pull requests.
|
||||
- [ ] Update the `baseline` and `compiler` tests, add new tests if necessary.
|
||||
- [ ] Consider updating the script `mfem/tests/sample-runs` (`sample-runs-serial` and `sample-runs-parallel`).
|
||||
|
||||
### Master/Next Workflow
|
||||
|
||||
MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
|
||||
- The `master` branch should always be of release quality and changes should not
|
||||
be merged until they have been fully tested. This branch is protected, and
|
||||
changes can only be made through pull requests.
|
||||
|
||||
- After approval, a pull request is merged manually (by MFEM developers) in the
|
||||
`next` branch for testing and the `in-next` label is added to the PR.
|
||||
This can be done as follows:
|
||||
|
||||
```
|
||||
# Pull the latest version of the "feature-dev" branch
|
||||
git checkout feature-dev
|
||||
git pull
|
||||
|
||||
# Pull the latest version of the "next" branch
|
||||
git checkout next
|
||||
git pull
|
||||
|
||||
# Merge "feature-dev" into "next", resolving conflicts, if necessary.
|
||||
# Use the "--no-ff" flag to create a new commit with merge message.
|
||||
git merge --no-ff feature-dev
|
||||
|
||||
# Push the "next" branch to the server
|
||||
git push
|
||||
```
|
||||
|
||||
- After a week of testing in `next` (excluding bugfixes), both on GitHub, as
|
||||
well as [internally](#tests-at-llnl) at LLNL, the original PR is merged into
|
||||
`master` (provided there are no issues).
|
||||
|
||||
- After the merge, the feature branch is deleted (unless it is a long-term
|
||||
project with periodic PRs).
|
||||
|
||||
- The `next` branch is used just for integrated testing of all PRs approved for
|
||||
merging into `master` to verify that each works individually and that all of
|
||||
them work as a group. This branch can be discarded at any time, though we
|
||||
typically do that only at the end of a [release cycle](#releases).
|
||||
|
||||
|
||||
### Releases
|
||||
|
||||
- Releases are just tags in the `master` branch, e.g. https://github.com/mfem/mfem/releases/tag/v3.3.2,
|
||||
and have a version that ends in an even "patch" number, e.g. `v3.2.2` or
|
||||
`v3.4` (by convention `v3.4` is the same as `v3.4.0`.) Between releases, the
|
||||
version ends in an odd "patch" number, e.g. `v3.3.3`.
|
||||
|
||||
- We use [milestones](https://github.com/mfem/mfem/milestones) to coordinate the
|
||||
work on different PRs toward a release, see for example the
|
||||
[v3.3.2 release](https://github.com/mfem/mfem/milestone/1?closed=1).
|
||||
|
||||
- After a release is complete, the `next` branch is recreated, e.g. as follows
|
||||
(replace `3.3.2` with current release):
|
||||
- Rename the current `next` branch to `next-pre-v3.3.2`.
|
||||
- Create a new `next` branch starting from the `v3.3.2` release.
|
||||
- Local copies of `next` can then be updated with `git checkout -B next origin/next`.
|
||||
|
||||
### Release Checklist
|
||||
|
||||
- [ ] Update the MFEM version in the following files:
|
||||
- [ ] `CHANGELOG`
|
||||
- [ ] `makefile`
|
||||
- [ ] `CMakeLists.txt`
|
||||
- [ ] `doc/CodeDocumentation.conf.in`
|
||||
- [ ] (LLNL only) Make sure all `README.html` files in the source repo are up to date.
|
||||
- [ ] Tag the repository:
|
||||
|
||||
```
|
||||
git tag -a v3.1 -m "Official release v3.1"
|
||||
git push origin v3.1
|
||||
```
|
||||
- [ ] Create the release tarball and push to `mfem/releases`.
|
||||
- [ ] Recreate the `next` branch as described in previous section.
|
||||
- [ ] Update and push documentation to `mfem/doxygen`.
|
||||
- [ ] Update URL shorlinks:
|
||||
- [ ] Create a shortlink at [https://goo.gl/](https://goo.gl/) for the release tarball, e.g. http://mfem.github.io/releases/mfem-3.1.tgz.
|
||||
- [ ] (LLNL only) Add and commit the new shorlink in the `links` and `links-mfem` files of the internal `mfem/downloads` repo.
|
||||
- [ ] Add the new shortlinks to the MFEM packages in `spack`, `homebrew/science`, `VisIt`, etc.
|
||||
- [ ] Update website in `mfem/web` repo:
|
||||
- Update version and shortlinks in `src/index.md` and `src/download.md`.
|
||||
- Use [cloc-1.62.pl](http://cloc.sourceforge.net/) and `ls -lh` to estimate the SLOC and the tarball size in `src/download.md`.
|
||||
|
||||
|
||||
## LLNL Workflow
|
||||
|
||||
- The GitHub `master` and `next` branches are mirrored to the LLNL institutional
|
||||
Bitbucket repository as `gh-master` and `gh-next`.
|
||||
|
||||
- `gh-master` is merged into LLNL's internal `master` through pull requests; write
|
||||
permissions to `master` are restricted to ensure this is the only way in which it
|
||||
gets updated.
|
||||
|
||||
- We never push directly from LLNL to GitHub.
|
||||
|
||||
- Versions of the code on LLNL's internal server, from most to least stable:
|
||||
- MFEM official release on mfem.org -- Most stable, tested in many apps.
|
||||
- `mfem:master` -- Recent development version, guaranteed to work.
|
||||
- `mfem:gh-master` -- Stable development version, passed testing, you can use
|
||||
it to build your code between releases.
|
||||
- `mfem:gh-next` -- Bleeding-edge development version, may be broken, use at
|
||||
your own risk.
|
||||
|
||||
|
||||
## Automated Testing
|
||||
|
||||
MFEM has several levels of automated testing running on GitHub, as well as on
|
||||
local Mac and Linux workstations, and Livermore Computing clusters at LLNL.
|
||||
|
||||
### Linux and Mac smoke tests
|
||||
We use Travis CI to drive the default tests on the `master` and `next`
|
||||
branches. See the `.travis` file and the logs at
|
||||
[https://travis-ci.org/mfem/mfem](https://travis-ci.org/mfem/mfem).
|
||||
|
||||
Testing using Travis CI should be kept lightweight, as there is a 50 minute time
|
||||
constraint on jobs. Two virtual machines are configured - Mac (OS X) and Linux.
|
||||
|
||||
- Tests on the `master` branch are triggered whenever a PR is issued on this branch.
|
||||
- Tests on the `next` branch are currently scheduled to run each night.
|
||||
|
||||
### Windows smoke test
|
||||
We use Appveyor to test building with the MS Visual C++ compiler in a Windows
|
||||
environment, as well as to test the CMake build. See the `.appveyor` file and the
|
||||
build logs at
|
||||
[https://ci.appveyor.com/project/mfem/mfem](https://ci.appveyor.com/project/mfem/mfem).
|
||||
|
||||
CMake is used to generate the MSVC Project files and drive the build. A release
|
||||
and debug build is performed with a simple run of `ex1` to verify the executable.
|
||||
|
||||
### Tests at LLNL
|
||||
At LLNL, we mirror the `master` and `next` branches internally (to `gh-master`
|
||||
and `gh-next`) and run longer nightly tests via cron. On the weekends, a more
|
||||
extensive test is run which extracts and executes all the different sample runs
|
||||
from each example.
|
||||
|
||||
|
||||
## Contact Information
|
||||
|
||||
- Contact the MFEM team by posting to the [GitHub issue tracker](https://github.com/mfem/mfem).
|
||||
Please perform a search to make sure your question has not been answered already.
|
||||
|
||||
- Email communications should be sent to the MFEM developers mailing list,
|
||||
mfem-dev@llnl.gov.
|
||||
|
||||
|
||||
## [Developer's Certificate of Origin 1.1](https://developercertificate.org/)
|
||||
|
||||
By making a contribution to this project, I certify that:
|
||||
|
||||
(a) The contribution was created in whole or in part by me and I have the right
|
||||
to submit it under the open source license indicated in the file; or
|
||||
|
||||
(b) The contribution is based upon previous work that, to the best of my
|
||||
knowledge, is covered under an appropriate open source license and I have
|
||||
the right under that license to submit that work with modifications, whether
|
||||
created in whole or in part by me, under the same open source license
|
||||
(unless I am permitted to submit under a different license), as indicated in
|
||||
the file; or
|
||||
|
||||
(c) The contribution was provided directly to me by some other person who
|
||||
certified (a), (b) or (c) and I have not modified it.
|
||||
|
||||
(d) I understand and agree that this project and the contribution are public and
|
||||
that a record of the contribution (including all personal information I
|
||||
submit with it, including my sign-off) is maintained indefinitely and may be
|
||||
redistributed consistent with this project or the open source license(s)
|
||||
involved.
|
||||
@@ -18,23 +18,14 @@ requires an MPI C++ compiler, as well as the following external libraries:
|
||||
- METIS (a family of multilevel partitioning algorithms)
|
||||
http://glaros.dtc.umn.edu/gkhome/metis/metis/overview
|
||||
|
||||
The METIS dependency can be disabled but that is not generally recommended, see
|
||||
the option MFEM_USE_METIS.
|
||||
|
||||
The library supports two build systems: one based on GNU make, and a second one
|
||||
based on CMake. Both build systems are described below. Some hints for building
|
||||
without GNU make or CMake can be found at the end of this file.
|
||||
|
||||
In addition to the native build systems, MFEM packages are also available in the
|
||||
following package managers:
|
||||
In addition to the native build systems, MFEM packages are also available in
|
||||
the Homebrew/Science, https://github.com/Homebrew/homebrew-science, and the
|
||||
Spack, https://github.com/LLNL/spack, package managers.
|
||||
|
||||
- Spack, https://github.com/spack/spack
|
||||
- OpenHPC, http://openhpc.community
|
||||
- Homebrew/Science, https://github.com/Homebrew/homebrew-science
|
||||
|
||||
We also recommend downloading and building the MFEM-based GLVis visualization
|
||||
tool which can be used to visualize the meshes and solution in MFEM's examples
|
||||
and miniapps. See http://glvis.org and http://mfem.org/building.
|
||||
|
||||
Quick start with GNU make
|
||||
=========================
|
||||
@@ -147,7 +138,7 @@ check the results from all the serial/parallel MFEM examples and miniapps use:
|
||||
|
||||
Note that by default MFEM uses "mpirun -np" in its test runs (this is also what
|
||||
is used in the sample runs of its examples and miniapps). The MPI launcher can
|
||||
be changed by the user as described in the "Specifying an MPI job launcher"
|
||||
be changed by the user as described in the "Specifying a MPI job launcher"
|
||||
section at the end of this file.
|
||||
|
||||
Running all the tests may take a while. Implementation details about the check
|
||||
@@ -159,8 +150,8 @@ An optional installation of the library and the headers can be performed with
|
||||
make install [PREFIX=<dir>]
|
||||
|
||||
The library will be installed in $(PREFIX)/lib, the headers in
|
||||
$(PREFIX)/include, and the configuration makefile (config.mk) in
|
||||
$(PREFIX)/share/mfem. The PREFIX option can also be set during configuration.
|
||||
$(PREFIX)/include, and the configuration makefile (config.mk) in $(PREFIX).
|
||||
The PREFIX option can also be set during configuration.
|
||||
|
||||
Information about the current build configuration can be viewed using
|
||||
|
||||
@@ -190,8 +181,6 @@ examples/ directory.
|
||||
|
||||
Configuration options (GNU make)
|
||||
================================
|
||||
See the configuration file config/defaults.mk for the default settings.
|
||||
|
||||
Compilers:
|
||||
CXX - C++ compiler, serial build
|
||||
MPICXX - MPI C++ compiler, parallel build
|
||||
@@ -202,14 +191,10 @@ Compiler options:
|
||||
CXXFLAGS - If not set, defined based on the above optimized/debug flags
|
||||
CPPFLAGS - Additional compiler options
|
||||
|
||||
Build options:
|
||||
STATIC - Build a static version of the library (YES/NO), default = YES
|
||||
SHARED - Build a shared version of the library (YES/NO), default = NO
|
||||
|
||||
Installation options:
|
||||
PREFIX - Specify the installation directory. The library (libmfem.a) will be
|
||||
installed in $(PREFIX)/lib, the headers in $(PREFIX)/include, and
|
||||
the configuration makefile (config.mk) in $(PREFIX)/share/mfem.
|
||||
the configuration makefile (config.mk) in $(PREFIX).
|
||||
INSTALL - Specify the install program, e.g /usr/bin/install
|
||||
|
||||
MFEM library features/options (GNU make)
|
||||
@@ -218,21 +203,10 @@ MFEM_USE_MPI = YES/NO
|
||||
Choose parallel/serial build. The parallel build requires proper setup of the
|
||||
HYPRE_* and METIS_* library options, see below.
|
||||
|
||||
MFEM_USE_METIS = YES/NO
|
||||
Enable/disable the use of the METIS library. By default, this option is set
|
||||
to the value of MFEM_USE_MPI. If this option is explicitly disabled in a
|
||||
parallel build, then the only parallel partitioning (domain decomposition)
|
||||
option in the library will be Cartesian partitioning with box meshes, and
|
||||
thus most of the parallel examples and miniapps will fail.
|
||||
|
||||
MFEM_DEBUG = YES/NO
|
||||
Choose debug/optimized build. The debug build enables a number of messages
|
||||
and consistency checks that may simplify bug-hunting.
|
||||
|
||||
MFEM_USE_EXCEPTIONS = YES/NO
|
||||
Enable the use of exceptions. In particular, modifies the default bahavior
|
||||
when errors are encountered: throw an exception, instead of aborting.
|
||||
|
||||
MFEM_USE_LIBUNWIND = YES/NO
|
||||
Use libunwind to print a stacktrace whenever mfem_error is raised. The
|
||||
information printed is enough to determine the line numbers where the
|
||||
@@ -257,16 +231,13 @@ MFEM_USE_MEMALLOC = YES/NO
|
||||
Internal MFEM option: enable batch allocation for some small objects.
|
||||
Recommended value is YES.
|
||||
|
||||
MFEM_TIMER_TYPE = 0/1/2/3/4/5/6/NO
|
||||
MFEM_TIMER_TYPE = 0/1/2/3/NO
|
||||
Specify which library functions to use in the class StopWatch used for
|
||||
measuring time. The available options are:
|
||||
0 - use std::clock from <ctime>, standard C++
|
||||
1 - use times from <sys/times.h>
|
||||
2 - use high-resolution POSIX clocks (see option POSIX_CLOCKS_LIB)
|
||||
3 - use QueryPerformanceCounter from <windows.h>
|
||||
4 - use mach_absolute_time from <mach/mach_time.h> + std::clock (Mac)
|
||||
5 - use gettimeofday from <sys/time.h>
|
||||
6 - use MPI_Wtime from <mpi.h>
|
||||
NO - use option 3 if the compiler macro _WIN32 is defined, 0 otherwise
|
||||
|
||||
MFEM_USE_SUNDIALS = YES/NO
|
||||
@@ -290,12 +261,6 @@ MFEM_USE_SUPERLU = YES/NO
|
||||
SuperLURowLocMatrix a distributed CSR matrix class needed by SuperLU. When
|
||||
enabled, this option uses the SUPERLU_* library options, see below.
|
||||
|
||||
MFEM_USE_STRUMPACK = YES/NO
|
||||
Enable MFEM functionality based on the STRUMPACK sparse direct solver and
|
||||
preconditioner through the STRUMPACKSolver and STRUMPACKRowLocMatrix
|
||||
classes. When enabled, this option uses the STRUMPACK_* library options, see
|
||||
below.
|
||||
|
||||
MFEM_USE_GNUTLS = YES/NO
|
||||
Enable secure socket support in class socketstream, using the auxiliary
|
||||
GnuTLS_* classes, based on the GnuTLS library. This option may be useful in
|
||||
@@ -307,9 +272,6 @@ MFEM_USE_GNUTLS = YES/NO
|
||||
the script 'glvis-keygen.sh' in the main GLVis directory can be used to do
|
||||
that:
|
||||
bash glvis-keygen.sh ["Your Name"] ["Your Email"]
|
||||
In MFEM v3.3.2 and earlier, the secure authentication is based on OpenPGP
|
||||
keys, while later versions use X.509 certificates. The latest version of the
|
||||
script 'glvis-keygen.sh' can be used to generate both types of keys.
|
||||
When MFEM_USE_GNUTLS is enabled, the additional build options, GNUTLS_*, are
|
||||
also used, see below.
|
||||
|
||||
@@ -336,13 +298,6 @@ MFEM_USE_SIDRE = YES/NO
|
||||
specification. When enabled, this option requires installation of HDF5 (see
|
||||
also MFEM_USE_NETCDF), Conduit and LLNL's axom project.
|
||||
|
||||
MFEM_USE_CONDUIT = YES/NO
|
||||
Enables support for converting MFEM Mesh and Grid Function objects to and
|
||||
from Conduit Mesh Blueprint Descriptions (https://github.com/LLNL/conduit/)
|
||||
and support for JSON and Binary I/O via Conduit Relay. This option requires
|
||||
an installation of Conduit. If Conduit was built with HDF5 support, it also
|
||||
requires an installation of HDF5 (see also MFEM_USE_NETCDF).
|
||||
|
||||
MFEM_USE_GZSTREAM = YES/NO
|
||||
Enables use of on-the-fly gzip compressed streams. With this feature enabled
|
||||
(YES), MFEM can compress its output files on-the-fly. In addition, it can
|
||||
@@ -353,14 +308,6 @@ MFEM_USE_GZSTREAM = YES/NO
|
||||
able to properly read an input file if it is gzip compressed. In that case,
|
||||
the solution is to uncompress the file with an external tool (such as gunzip)
|
||||
before attempting to use it with MFEM.
|
||||
When enabled, this option uses the ZLIB_* library options, see below.
|
||||
|
||||
MFEM_USE_PUMI = YES/NO
|
||||
Enable the usage of PUMI (https://scorec.rpi.edu/pumi/) in MFEM. The Parallel
|
||||
Unstructured Mesh Infrastructure (PUMI) is an unstructured, distributed mesh
|
||||
data management system that is capable of handling general non-manifold
|
||||
models and effectively supports automated adaptive analysis. PUMI enables
|
||||
support for parallel unstructured mesh modifications in MFEM.
|
||||
|
||||
MFEM_BUILD_TAG = (any value)
|
||||
An optional tag to characterize the build. Exported to config/config.mk.
|
||||
@@ -386,8 +333,8 @@ The specific libraries and their options are:
|
||||
URL: http://www.llnl.gov/CASC/hypre
|
||||
Options: HYPRE_OPT, HYPRE_LIB.
|
||||
|
||||
- METIS, used when MFEM_USE_METIS = YES. If using METIS 5, set
|
||||
MFEM_USE_METIS_5 = YES (default is to use METIS 4).
|
||||
- METIS, required for the parallel build, i.e. when MFEM_USE_MPI = YES. If using
|
||||
METIS 5, set MFEM_USE_METIS_5 = YES (default is to use METIS 4).
|
||||
URL: http://glaros.dtc.umn.edu/gkhome/metis/metis/overview
|
||||
Options: METIS_OPT, METIS_LIB.
|
||||
|
||||
@@ -405,10 +352,7 @@ The specific libraries and their options are:
|
||||
Option: POSIX_CLOCKS_LIB (default = -lrt).
|
||||
|
||||
- SUNDIALS (optional), used when MFEM_USE_SUNDIALS = YES.
|
||||
Beginning with MFEM v3.3, SUNDIALS v2.7.0 is supported.
|
||||
Beginning with MFEM v3.3.2, SUNDIALS v3.0.0 is also supported.
|
||||
If MFEM_USE_MPI is enabled, we expect that SUNDIALS is built with support for
|
||||
both MPI and hypre.
|
||||
In parallel we expect that SUNDIALS is built with support for MPI and hypre.
|
||||
URL: http://computation.llnl.gov/projects/sundials/sundials-software
|
||||
Options: SUNDIALS_OPT, SUNDIALS_LIB.
|
||||
|
||||
@@ -427,14 +371,6 @@ The specific libraries and their options are:
|
||||
URL: http://crd-legacy.lbl.gov/~xiaoye/SuperLU
|
||||
Options: SUPERLU_OPT, SUPERLU_LIB.
|
||||
|
||||
- STRUMPACK (optional), used when MFEM_USE_STRUMPACK = YES. Note that STRUMPACK
|
||||
requires the PT-Scotch and Scalapack libraries as well as ParMETIS, which
|
||||
includes METIS 5 in its distribution.
|
||||
The support for STRUMPACK was added in MFEM v3.3.2 and it requires STRUMPACK
|
||||
2.0.0 or later.
|
||||
URL: http://portal.nersc.gov/project/sparse/strumpack
|
||||
Options: STRUMPACK_OPT, STRUMPACK_LIB.
|
||||
|
||||
- GnuTLS (optional), used when MFEM_USE_GNUTLS = YES. On most Linux systems,
|
||||
GnuTLS is available as a development package, e.g. gnutls-devel. On Mac OS X,
|
||||
one can get the library through the Homebrew package manager (http://brew.sh).
|
||||
@@ -448,7 +384,7 @@ The specific libraries and their options are:
|
||||
URL: www.unidata.ucar.edu/software/netcdf
|
||||
Options: NETCDF_OPT, NETCDF_LIB.
|
||||
|
||||
- PETSc (optional), used when MFEM_USE_PETSC = YES. Version 3.8 or higher of
|
||||
- PETSc (optional), used when MFEM_USE_PETSC = YES. Version 3.8 or higher of
|
||||
the PETSC dev branch is required. The MFEM and PETSc builds can share common
|
||||
libraries, e.g., hypre and SUNDIALS. Here's an example configuration, assuming
|
||||
PETSc has been cloned on the same level as mfem and hypre:
|
||||
@@ -465,16 +401,6 @@ The specific libraries and their options are:
|
||||
https://support.hdfgroup.org/HDF5 (HDF5)
|
||||
Options: SIDRE_OPT, SIDRE_LIB.
|
||||
|
||||
- Conduit, used when MFEM_USE_CONDUIT = YES. Direct Conduit Mesh Blueprint
|
||||
support requires Conduit >= v0.3.1 and VisIt >= v2.13.1 to read the output.
|
||||
URL: https://github.com/LLNL/conduit (Conduit)
|
||||
https://support.hdfgroup.org/HDF5 (HDF5)
|
||||
Options: CONDUIT_OPT, CONDUIT_LIB.
|
||||
|
||||
- PUMI, used when MFEM_USE_PUMI = YES.
|
||||
URL: https://scorec.rpi.edu/pumi
|
||||
Options: PUMI_OPT, PUMI_LIB.
|
||||
|
||||
- MPFR (optional), used when MFEM_USE_MPFR = YES.
|
||||
URL: http://mpfr.org, it depends on the GMP library: https://gmplib.org
|
||||
Options: MPFR_OPT, MPFR_LIB.
|
||||
@@ -485,11 +411,6 @@ The specific libraries and their options are:
|
||||
URL: http://www.nongnu.org/libunwind
|
||||
Options: LIBUNWIND_OPT, LIBUNWIND_LIB.
|
||||
|
||||
- ZLIB (optional), used when MFEM_USE_GZSTREAM = YES, or when MFEM_USE_NETCDF =
|
||||
YES (in the default settings for NETCDF_OPT and NETCDF_LIB).
|
||||
URL: https://zlib.net
|
||||
Options: ZLIB_OPT, ZLIB_LIB.
|
||||
|
||||
|
||||
Building with CMake
|
||||
===================
|
||||
@@ -521,12 +442,6 @@ Debug and optimization options are controlled through the CMake variable
|
||||
CMAKE_BUILD_TYPE which can be set to standard values like "Debug", and "Release"
|
||||
(default).
|
||||
|
||||
To use a specific generator use the "-G <generator>" option of cmake:
|
||||
|
||||
cmake <mfem-source-dir> -G "Xcode"
|
||||
cmake <mfem-source-dir> -G "Visual Studio 12 2013"
|
||||
cmake <mfem-source-dir> -G "MinGW Makefiles"
|
||||
|
||||
With CMake it is possible to build MFEM as a shared library using the standard
|
||||
CMake option -DBUILD_SHARED_LIBS=1.
|
||||
|
||||
@@ -534,30 +449,15 @@ Once configured, the library can be built simply with (assuming a UNIX type
|
||||
system, where the default is to generate "UNIX Makefiles")
|
||||
|
||||
make -j 4
|
||||
or
|
||||
cmake --build .
|
||||
or
|
||||
cmake --build . --config Release [Visual Studio, Xcode]
|
||||
|
||||
The build can be quick-tested by running
|
||||
|
||||
make check
|
||||
or
|
||||
cmake --build . --target check
|
||||
or
|
||||
cmake --build . --config Release --target check [Visual Studio, Xcode]
|
||||
|
||||
which will simply compile and run Example 1/1p. For more extensive tests that
|
||||
check the results from all the serial/parallel MFEM examples and miniapps use:
|
||||
|
||||
make exec -j 4
|
||||
make test
|
||||
or
|
||||
cmake --build . --target exec
|
||||
cmake --build . --target test
|
||||
or
|
||||
cmake --build . --config Release --target exec [Visual Studio, Xcode]
|
||||
cmake --build . --config Release --target RUN_TESTS [Visual Studio, Xcode]
|
||||
|
||||
Note that running all the tests may take a while.
|
||||
|
||||
@@ -565,11 +465,6 @@ Installation prefix can be configured by setting the standard CMake variable
|
||||
CMAKE_INSTALL_PREFIX. To install the library, use
|
||||
|
||||
make install
|
||||
or
|
||||
cmake --build . --target install
|
||||
or
|
||||
cmake --build . --config Release --target install [Xcode]
|
||||
cmake --build . --config Release --target INSTALL [Visual Studio]
|
||||
|
||||
The library will be installed in <PREFIX>/lib, the headers in <PREFIX>/include,
|
||||
and the configuration CMake files in <PREFIX>/lib/cmake/mfem.
|
||||
@@ -577,8 +472,6 @@ and the configuration CMake files in <PREFIX>/lib/cmake/mfem.
|
||||
|
||||
Configuration variables (CMake)
|
||||
===============================
|
||||
See the configuration file config/defaults.cmake for the default settings.
|
||||
|
||||
Non-standard CMake variables for compilers:
|
||||
CXX - If set, overwrite the auto-detected C++ compiler, serial build
|
||||
MPICXX - If set, overwrite the auto-detected MPI C++ compiler, parallel build
|
||||
@@ -592,7 +485,6 @@ The following options are equivalent to the GNU make options with the same name:
|
||||
[see "MFEM library features/options (GNU make)" above]
|
||||
|
||||
MFEM_USE_MPI
|
||||
MFEM_USE_METIS - Set to ${MFEM_USE_MPI}, can be overwritten.
|
||||
MFEM_USE_LIBUNWIND
|
||||
MFEM_USE_LAPACK
|
||||
MFEM_THREAD_SAFE
|
||||
@@ -602,12 +494,10 @@ MFEM_TIMER_TYPE - Set automatically, can be overwritten.
|
||||
MFEM_USE_MESQUITE
|
||||
MFEM_USE_SUITESPARSE
|
||||
MFEM_USE_SUPERLU
|
||||
MFEM_USE_STRUMPACK
|
||||
MFEM_USE_GNUTLS
|
||||
MFEM_USE_NETCDF
|
||||
MFEM_USE_MPFR
|
||||
MFEM_USE_GZSTREAM
|
||||
MFEM_USE_PUMI
|
||||
|
||||
The following options are CMake specific:
|
||||
|
||||
@@ -646,14 +536,13 @@ The CMake build system adds auto-detection for the following packages/libraries:
|
||||
- METIS - The option MFEM_USE_METIS_5 is auto-detected.
|
||||
- MESQUITE
|
||||
- SuiteSparse
|
||||
- SuperLUDist, STRUMPACK
|
||||
- SuperLUDist
|
||||
- ParMETIS
|
||||
- GNUTLS - Extends the built-in CMake support, to search GNUTLS_DIR as well.
|
||||
- NETCDF
|
||||
- MPFR
|
||||
- LIBUNWIND
|
||||
- POSIXCLOCKS
|
||||
- PUMI
|
||||
|
||||
The following built-in CMake packages are also used:
|
||||
|
||||
@@ -669,15 +558,15 @@ Before using another build system (e.g. Visual Studio) it is necessary to create
|
||||
a proper configuration header file, config/config.hpp, using the template from
|
||||
config/config.hpp.in:
|
||||
|
||||
cp config/config.hpp.in config/_config.hpp
|
||||
cp config/config.hpp.in config/config.hpp
|
||||
|
||||
The file config/_config.hpp can then be edited to enable desired options. The
|
||||
The file config/config.hpp can then be edited to enable desired options. The
|
||||
MFEM library is simply a combination of all object files obtained by compiling
|
||||
the .cpp source files in the source directories: general, linalg, mesh, and fem.
|
||||
|
||||
|
||||
Specifying an MPI job launcher
|
||||
==============================
|
||||
Specifying a MPI job launcher
|
||||
=============================
|
||||
By default, MFEM will use 'mpirun -np #' to launch any of its parallel tests or
|
||||
miniapps, where # is the number of MPI tasks. An alternate MPI launcher can be
|
||||
provided by setting the MFEM_MPIEXEC and MFEM_MPIEXEC_NP config variables.
|
||||
|
||||
@@ -12,15 +12,11 @@ to enable the research and development of scalable finite element discretization
|
||||
and solver algorithms through general finite element abstractions, accurate and
|
||||
flexible visualization, and tight integration with the hypre library.
|
||||
|
||||
* For building instructions, see the file INSTALL, or type "make help".
|
||||
For building instructions, see the file INSTALL, or type "make help". Copyright
|
||||
information and licensing restrictions can be found in the file COPYRIGHT.
|
||||
|
||||
* Copyright and licensing information can be found in the file COPYRIGHT.
|
||||
|
||||
* The best starting point for new users interested in MFEM's features is the
|
||||
interactive documentation in examples/README.html.
|
||||
|
||||
* Developers interested in contributing to the library, should read the
|
||||
instructions and documentation in the CONTRIBUTING.md file.
|
||||
The best starting point for new users interested in MFEM's features is the
|
||||
interactive documentation in examples/README.html.
|
||||
|
||||
Conceptually, MFEM can be viewed as a finite element toolbox that provides the
|
||||
building blocks for developing finite element algorithms in a manner similar to
|
||||
@@ -60,8 +56,7 @@ time integrators, etc.
|
||||
For examples of using MFEM, see the examples/ and miniapps/ directories, as well
|
||||
as the OpenGL visualization tool GLVis which is available at http://glvis.org.
|
||||
|
||||
This project is released under the LGPL v2.1 license with static linking
|
||||
exception. See files COPYRIGHT and LICENSE file for full details.
|
||||
This project is released under the LGPL v2.1 license. See LICENSE file for full
|
||||
details.
|
||||
|
||||
LLNL Release Number: LLNL-CODE-443211
|
||||
DOI: 10.11578/dc.20171025.1248
|
||||
|
||||
@@ -1,27 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_ALL_HPP
|
||||
#define MFEM_BACKENDS_ALL_HPP
|
||||
|
||||
#include "../config/config.hpp"
|
||||
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "base/backend.hpp"
|
||||
|
||||
#ifdef MFEM_USE_OCCA
|
||||
#include "occa/backend.hpp"
|
||||
#endif
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_ALL_HPP
|
||||
@@ -1,213 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_ARRAY_HPP
|
||||
#define MFEM_BACKENDS_BASE_ARRAY_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "layout.hpp"
|
||||
#include "utils.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Extension to the template class Array<T>
|
||||
class PArray : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Layout with shared ownership (smart pointer)
|
||||
DLayout layout;
|
||||
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/** @brief Create and return a new array (in @a *clone) of the same dynamic
|
||||
type as this array using the same layout and ItemSize().
|
||||
|
||||
Set @a *clone to NULL if allocation fails.
|
||||
|
||||
If @a copy_data is true, the contents of this array is copied to the new
|
||||
array; otherwise, the new array remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the array data of the newly created
|
||||
object (in @a *buffer) , if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
virtual PArray *DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const = 0;
|
||||
|
||||
/// Resize the array, reallocating its data if necessary.
|
||||
/** If @a buffer is not NULL, return the array data (in @a *buffer), if it
|
||||
is stored as a contiguous array on the host; otherwise, set @a *buffer to
|
||||
NULL. Returns 0 on success and non-zero otherwise, e.g. if memory
|
||||
allocation fails.
|
||||
|
||||
If the @a new_layout is not supported, a non-zero error code will be
|
||||
returned.
|
||||
|
||||
The @a new_layout has to be valid, i.e. new_layout != NULL and
|
||||
new_layout->HasEngine() == true.
|
||||
|
||||
@note If reallocation is performed, the previous content of the array is
|
||||
NOT copied to the new location. */
|
||||
virtual int DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size) = 0;
|
||||
|
||||
/** @brief Get access to the contents of the array in host memory, as a
|
||||
contiguous array. */
|
||||
/** If the array data is stored as a contiguous array in host memory, return
|
||||
a pointer to it. Otherwise, copy the data to @a buffer (if @a buffer is
|
||||
not NULL) and return @a buffer.
|
||||
@note If not NULL, @a buffer is assumed to be of size greater than or
|
||||
equal to Size(). */
|
||||
virtual void *DoPullData(void *buffer, std::size_t item_size) = 0;
|
||||
|
||||
/** @brief Set all entries of the array to the (single) value pointed to by
|
||||
@a value_ptr. */
|
||||
virtual void DoFill(const void *value_ptr, std::size_t item_size) = 0;
|
||||
|
||||
/** @brief Set all Size() entries of the array from the given contiguous
|
||||
array, @a src_buffer, on the host. */
|
||||
virtual void DoPushData(const void *src_buffer, std::size_t item_size) = 0;
|
||||
|
||||
/// Copy the data from @a src to @a *this.
|
||||
/** Both arrays must have the same dynamic type, layout, and item_size. */
|
||||
virtual void DoAssign(const PArray &src, std::size_t item_size) = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
public:
|
||||
/** @brief The @a layout parameter will be reference counted and therefore it
|
||||
should be dynamically allocated. */
|
||||
/** The @a layout must be valid in the sense that layout != NULL and
|
||||
layout->HasEngine() == true. */
|
||||
PArray(PLayout &p_layout)
|
||||
: layout(&p_layout)
|
||||
{
|
||||
MFEM_ASSERT(layout && layout->HasEngine(), "invalid layout");
|
||||
}
|
||||
|
||||
virtual ~PArray() { }
|
||||
|
||||
/// Get the current size of the array.
|
||||
std::size_t Size() const { return layout->Size(); }
|
||||
|
||||
/// Get the current layout of the array.
|
||||
PLayout &GetLayout() const { return *layout; }
|
||||
|
||||
/// TODO
|
||||
/// Note: we cannot use static_cast for class PArray.
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return dynamic_cast<derived_t&>(*this); }
|
||||
|
||||
/// TODO
|
||||
/// Note: we cannot use static_cast for class PArray.
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return dynamic_cast<const derived_t&>(*this); }
|
||||
|
||||
|
||||
// TODO: Error handling ... handle errors at the Engine level, at the class
|
||||
// level, or at the method level?
|
||||
|
||||
// TODO: Asynchronous execution interface ...
|
||||
|
||||
|
||||
/**
|
||||
@name Public virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/** @brief Create and return a new array (in @a *clone) of the same dynamic
|
||||
type as this array using the same layout and ItemSize().
|
||||
|
||||
Set @a *clone to NULL if allocation fails.
|
||||
|
||||
If @a copy_data is true, the contents of this array is copied to the new
|
||||
array; otherwise, the new array remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the array data of the newly created
|
||||
object (in @a *buffer) , if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
template <typename T>
|
||||
DArray Clone(bool copy_data, T **buffer) const
|
||||
{ return DArray(DoClone(copy_data, (void**)buffer, sizeof(T))); }
|
||||
|
||||
/// Resize the array, reallocating its data if necessary.
|
||||
/** If @a buffer is not NULL, return the array data (in @a *buffer), if it
|
||||
is stored as a contiguous array on the host; otherwise, set @a *buffer to
|
||||
NULL. Returns 0 on success and non-zero otherwise, e.g. if memory
|
||||
allocation fails.
|
||||
|
||||
If the @a new_layout is not supported, a non-zero error code will be
|
||||
returned.
|
||||
|
||||
The @a new_layout has to be valid, i.e. new_layout != NULL and
|
||||
new_layout->HasEngine() == true.
|
||||
|
||||
@note If reallocation is performed, the previous content of the array is
|
||||
NOT copied to the new location. */
|
||||
template <typename T>
|
||||
int Resize(PLayout &new_layout, T **buffer)
|
||||
{ return DoResize(new_layout, (void**)buffer, sizeof(T)); }
|
||||
|
||||
/// Shortcut for Resize(*layout, buffer).
|
||||
/** This method is useful for updating the array after its layout is changed
|
||||
externally. */
|
||||
template <typename T>
|
||||
int Update(T **buffer)
|
||||
{ return DoResize(*layout, (void**)buffer, sizeof(T)); }
|
||||
|
||||
/// Shortcut for layout->Resize(new_size) followed by Update()
|
||||
template <typename T>
|
||||
int Resize(std::size_t new_size, T **buffer)
|
||||
{ layout->Resize(new_size); return Update(buffer); }
|
||||
|
||||
/** @brief Get access to the contents of the array in host memory, as a
|
||||
contiguous array. */
|
||||
/** If the array data is stored as a contiguous array in host memory, return
|
||||
a pointer to it. Otherwise, copy the data to @a buffer (if @a buffer is
|
||||
not NULL) and return @a buffer.
|
||||
@note If not NULL, @a buffer is assumed to be of size greater than or
|
||||
equal to Size(). */
|
||||
template <typename T>
|
||||
T *PullData(T *buffer)
|
||||
{ return Size() ? (T*)DoPullData((void*)buffer, sizeof(T)) : NULL; }
|
||||
|
||||
/** @brief Set all entries of the array to the (single) value pointed to by
|
||||
@a value_ptr. */
|
||||
template <typename T>
|
||||
void Fill(const T &value) { if (Size()) { DoFill(&value, sizeof(T)); } }
|
||||
|
||||
/** @brief Set all Size() entries of the array from the given contiguous
|
||||
array, @a src_buffer, on the host. */
|
||||
template <typename T>
|
||||
void PushData(const T *src_buffer)
|
||||
{ if (Size()) { DoPushData(src_buffer, sizeof(T)); } }
|
||||
|
||||
/// Copy the data from @a src to @a *this.
|
||||
/** Both arrays must have the same dynamic type, layout, and entry type. */
|
||||
template <typename T>
|
||||
void Assign(const PArray &src) { if (Size()) { DoAssign(src, sizeof(T)); } }
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_ARRAY_HPP
|
||||
@@ -1,57 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_BACKEND_HPP
|
||||
#define MFEM_BACKENDS_BASE_BACKEND_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "memory_resource.hpp"
|
||||
#include "engine.hpp"
|
||||
#include "array.hpp"
|
||||
#include "vector.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
|
||||
#include <string>
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include <mpi.h>
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// TODO
|
||||
class Backend
|
||||
{
|
||||
public:
|
||||
/// TODO
|
||||
virtual ~Backend() { }
|
||||
|
||||
/// TODO
|
||||
virtual bool Supports(const std::string &engine_spec) const = 0;
|
||||
|
||||
/// TODO
|
||||
virtual Engine *Create(const std::string &engine_spec) = 0;
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// TODO
|
||||
virtual Engine *Create(MPI_Comm comm, const std::string &engine_spec) = 0;
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_BACKEND_HPP
|
||||
@@ -1,72 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_BILINEARFORM_HPP
|
||||
#define MFEM_BACKENDS_BASE_BILINEARFORM_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class Vector;
|
||||
class OperatorHandle;
|
||||
class BilinearForm;
|
||||
|
||||
/// TODO: doxygen
|
||||
class PBilinearForm : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Engine with shared ownership
|
||||
SharedPtr<const Engine> engine;
|
||||
/// Not owned.
|
||||
BilinearForm *bform;
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
PBilinearForm(const Engine &e, BilinearForm &bf)
|
||||
: engine(&e), bform(&bf) { }
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~PBilinearForm() { }
|
||||
|
||||
/// Get the associated Engine
|
||||
const Engine &GetEngine() const { return *engine; }
|
||||
|
||||
/// Assemble the PBilinearForm.
|
||||
/** This method is called from the method BilinearForm::Assemble() of the
|
||||
associated BilinearForm #bform.
|
||||
@returns True, if the host assembly should be skipped. */
|
||||
virtual bool Assemble() = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
OperatorHandle &A) = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
Vector &x, Vector &b,
|
||||
OperatorHandle &A, Vector &X, Vector &B,
|
||||
int copy_interior) = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void RecoverFEMSolution(const Vector &X, const Vector &b,
|
||||
Vector &x) = 0;
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_BILINEARFORM_HPP
|
||||
@@ -1,49 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
Engine::Engine(Backend *b, int n_mem, int n_workers)
|
||||
: backend(b),
|
||||
#ifdef MFEM_USE_MPI
|
||||
comm(MPI_COMM_NULL),
|
||||
#endif
|
||||
num_mem_res(n_mem),
|
||||
num_workers(n_workers),
|
||||
memory_resources(new MemoryResource*[num_mem_res]()),
|
||||
workers_weights(new double[num_workers]()),
|
||||
workers_mem_res(new int[num_workers]())
|
||||
{
|
||||
// Note: all arrays are value-initialized with zeros.
|
||||
}
|
||||
|
||||
Engine::~Engine()
|
||||
{
|
||||
delete [] workers_mem_res;
|
||||
delete [] workers_weights;
|
||||
for (int i = 0; i < num_mem_res; i++)
|
||||
{
|
||||
delete memory_resources[i];
|
||||
}
|
||||
delete [] memory_resources;
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
@@ -1,190 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_ENGINE_HPP
|
||||
#define MFEM_BACKENDS_BASE_ENGINE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "../../general/scalars.hpp"
|
||||
#include "memory_resource.hpp"
|
||||
#include "smart_pointers.hpp"
|
||||
#include "utils.hpp"
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include <mpi.h>
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// Forward declarations.
|
||||
class Backend;
|
||||
template <typename T> class Array;
|
||||
class Operator;
|
||||
class FiniteElementSpace;
|
||||
class LinearForm;
|
||||
class BilinearForm;
|
||||
class MixedBilinearForm;
|
||||
class NonlinearForm;
|
||||
|
||||
|
||||
/// In parallel, each MPI rank will usually create a single engine.
|
||||
class Engine : public RefCounted
|
||||
{
|
||||
protected:
|
||||
Backend *backend; ///< Backend that created the engine. Not owned.
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
MPI_Comm comm; ///< Associated MPI communicator (may be MPI_COMM_NULL).
|
||||
#endif
|
||||
|
||||
/// Number of memory resources used by the Engine.
|
||||
int num_mem_res;
|
||||
/// Number of workers used by the Engine.
|
||||
int num_workers;
|
||||
|
||||
/// Memory resources used by the engine - array of pointers.
|
||||
/** Both the array and the entries are owned. */
|
||||
MemoryResource **memory_resources;
|
||||
|
||||
/// Relative computational speed of the workers. Owned.
|
||||
double *workers_weights;
|
||||
|
||||
/// For each worker, which memory resource it uses.
|
||||
int *workers_mem_res;
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
Engine(Backend *b, int n_mem, int n_workers);
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual ~Engine();
|
||||
|
||||
|
||||
/**
|
||||
@name Machine resources interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// Get the associated MPI_Comm
|
||||
MPI_Comm GetComm() const { return comm; }
|
||||
#endif
|
||||
|
||||
/// TODO
|
||||
int GetNumMemRes() const { return num_mem_res; }
|
||||
|
||||
/// TODO
|
||||
MemoryResource &GetMemRes(int idx) const { return *memory_resources[idx]; }
|
||||
|
||||
/// TODO
|
||||
int GetNumWorkers() const { return num_workers; }
|
||||
|
||||
/// TODO
|
||||
const double *GetWorkersWeights() const { return workers_weights; }
|
||||
|
||||
/// TODO
|
||||
const int *GetWorkersMemRes() const { return workers_mem_res; }
|
||||
|
||||
///@}
|
||||
// End: Machine resources interface
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
|
||||
|
||||
// TODO: Error handling ... handle errors at the Engine level, at the class
|
||||
// level, or at the method level?
|
||||
|
||||
|
||||
/**
|
||||
@name Virtual interface: finite element data structures and algorithms
|
||||
*/
|
||||
///@{
|
||||
|
||||
// TODO: Asynchronous execution in this class ...
|
||||
|
||||
/// Allocate and return a new layout for the given @a size.
|
||||
/** The layout decomposition (in the case of multiple workers) is determined
|
||||
automatically by the Engine using a deterministic algorithm: calls to
|
||||
this method with the same @a size will produce the same result, as long
|
||||
as the Engine remains unmodified between the calls.
|
||||
|
||||
The returned object is allocated with operator new and must be
|
||||
deallocated by the caller.
|
||||
|
||||
TODO: Returns NULL if memory allocation fails?
|
||||
*/
|
||||
virtual DLayout MakeLayout(std::size_t size) const = 0;
|
||||
|
||||
/// Allocate and return a new layout for the given worker decomposition.
|
||||
/** The returned object is allocated with operator new and must be
|
||||
deallocated by the caller.
|
||||
|
||||
TODO: Returns NULL if memory allocation fails?
|
||||
|
||||
The @a offsets should satisfy: offsets.Size() == number of workers + 1,
|
||||
offsets[0] == 0, and offsets[i] <= offsets[i+1], for i: 0 <= i < number
|
||||
of workers. */
|
||||
virtual DLayout MakeLayout(const Array<std::size_t> &offsets) const = 0;
|
||||
|
||||
// Note: There may be other ways to construct layouts in the future, e.g.
|
||||
// block-vector layouts, or multi-vector layouts.
|
||||
|
||||
/// TODO
|
||||
virtual DArray MakeArray(PLayout &layout, std::size_t item_size) const = 0;
|
||||
|
||||
/// Allocate and return a new vector using the given @a layout.
|
||||
/** The returned object is a smart pointer that will automatically deallocate
|
||||
the vector.
|
||||
|
||||
TODO: Produce an error if memory allocation fails?
|
||||
|
||||
Only layouts returned by this Engine are guaranteed to be supported.
|
||||
Using a type that is not supported will produce an error. */
|
||||
virtual DVector MakeVector(PLayout &layout,
|
||||
int type_id = ScalarId<double>::value) const = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual DFiniteElementSpace MakeFESpace(FiniteElementSpace &fes) const = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual DBilinearForm MakeBilinearForm(BilinearForm &bf) const = 0;
|
||||
|
||||
|
||||
// Question: How do we construct coefficients?
|
||||
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual void AssembleLinearForm(LinearForm &l_form) const = 0;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual Operator *MakeOperator(const MixedBilinearForm &mbl_form) const = 0;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual Operator *MakeOperator(const NonlinearForm &nl_form) const = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_ENGINE_HPP
|
||||
@@ -1,98 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_FE_SPACE_HPP
|
||||
#define MFEM_BACKENDS_BASE_FE_SPACE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "utils.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class FiniteElementSpace;
|
||||
class QuadratureSpace;
|
||||
|
||||
/// TODO: doxygen
|
||||
class PFiniteElementSpace : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Engine with shared ownership
|
||||
SharedPtr<const Engine> engine;
|
||||
/// Not owned.
|
||||
mfem::FiniteElementSpace *fes;
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
PFiniteElementSpace(const Engine &e, mfem::FiniteElementSpace &fespace)
|
||||
: engine(&e), fes(&fespace) { }
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~PFiniteElementSpace() { }
|
||||
|
||||
/// Get the associated engine
|
||||
const Engine &GetEngine() const { return *engine; }
|
||||
|
||||
/// Return the associated mfem::FiniteElementSpace
|
||||
mfem::FiniteElementSpace *GetFESpace() const { return fes; }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
|
||||
|
||||
/**
|
||||
@name Virtual interface: finite element space functionality
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// TODO
|
||||
/** Return the operator mapping T-vectors to L-vectors. If a NULL pointer is
|
||||
returned then the mapping is the idenity. */
|
||||
virtual const mfem::Operator *GetProlongationOperator() const = 0;
|
||||
|
||||
/// TODO
|
||||
/** Return the operator mapping L-vectors to T-vectors that extracts the
|
||||
subset of all true dofs, i.e. no assembly is performed. If a NULL pointer
|
||||
is returned then the mapping is the idenity. */
|
||||
virtual const mfem::Operator *GetRestrictionOperator() const = 0;
|
||||
|
||||
/// TODO
|
||||
/** Return the operator mapping L-vectors to Q-vectors that evaluates the
|
||||
values of a GridFunction as a QuadratureFunction on the given
|
||||
QuadratureSpace. If the returned pointer is NULL, then the mapping is the
|
||||
identity. */
|
||||
virtual const mfem::Operator *GetInterpolationOperator(
|
||||
const mfem::QuadratureSpace &qspace) const = 0;
|
||||
|
||||
/// TODO
|
||||
/** Return the operator mapping L-vectors to Q-vectors that evaluates the
|
||||
_reference element_ gradients of a GridFunction as a QuadratureFunction
|
||||
on the given QuadratureSpace. */
|
||||
virtual const mfem::Operator *GetGradientOperator(
|
||||
const mfem::QuadratureSpace &qspace) const = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_FE_SPACE_HPP
|
||||
@@ -1,110 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_LAYOUT_HPP
|
||||
#define MFEM_BACKENDS_BASE_LAYOUT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "smart_pointers.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Polymorphic layout (array/vector layout descriptor)
|
||||
class PLayout : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Engine with shared ownership
|
||||
SharedPtr<const Engine> engine;
|
||||
std::size_t size;
|
||||
|
||||
template <typename DObject>
|
||||
struct Maker
|
||||
{
|
||||
template <typename entry_t>
|
||||
static DObject MakeNew(PLayout &layout);
|
||||
};
|
||||
|
||||
public:
|
||||
explicit PLayout(std::size_t s = 0) : engine(NULL), size(s) { }
|
||||
|
||||
explicit PLayout(const Engine &e, std::size_t s = 0)
|
||||
: engine(&e), size(s) { }
|
||||
|
||||
virtual ~PLayout() { }
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// Resize the layout
|
||||
virtual void Resize(std::size_t new_size) { size = new_size; }
|
||||
|
||||
/// Resize the layout based on the given worker offsets
|
||||
virtual void Resize(const Array<std::size_t> &offsets)
|
||||
{ MFEM_ABORT("method not supported"); }
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
/// Layouts without engine cannot create DArray, DVector, etc.
|
||||
bool HasEngine() const { return engine != NULL; }
|
||||
|
||||
/// TODO: doxygen
|
||||
const Engine &GetEngine() const { return *engine; }
|
||||
|
||||
/// TODO: doxygen
|
||||
std::size_t Size() const { return size; }
|
||||
|
||||
/// TODO
|
||||
/** Useful in backends for down-casting to a backend-specific layout type.
|
||||
|
||||
When MFEM_DEBUG=YES, performs a type check using dynamic_cast. */
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
/// TODO
|
||||
/** Useful in backends for down-casting to a backend-specific layout type.
|
||||
|
||||
When MFEM_DEBUG=YES, performs a type check using dynamic_cast. */
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
|
||||
/// TODO: doxygen
|
||||
template <typename DObject, typename entry_t>
|
||||
DObject Make()
|
||||
{
|
||||
MFEM_ASSERT(HasEngine(), "this method requires an Engine");
|
||||
return Maker<DObject>::template MakeNew<entry_t>(*this);
|
||||
}
|
||||
};
|
||||
|
||||
template <> struct PLayout::Maker<DArray>
|
||||
{
|
||||
template <typename entry_t> static DArray MakeNew(PLayout &layout)
|
||||
{ return layout.GetEngine().MakeArray(layout, sizeof(entry_t)); }
|
||||
};
|
||||
|
||||
template <> struct PLayout::Maker<DVector>
|
||||
{
|
||||
template <typename entry_t> static DVector MakeNew(PLayout &layout)
|
||||
{ return layout.GetEngine().MakeVector(layout, ScalarId<entry_t>::value); }
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_LAYOUT_HPP
|
||||
@@ -1,59 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "memory_resource.hpp"
|
||||
#include "../../general/error.hpp"
|
||||
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <cerrno>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
void *NewDeleteMemoryResource::DoAllocate(std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
void *p = ::operator new[](bytes);
|
||||
MFEM_VERIFY(!alignment || (std::size_t)(p) % alignment == 0,
|
||||
"invalid alignment");
|
||||
return p;
|
||||
}
|
||||
|
||||
void NewDeleteMemoryResource::DoDeallocate(void *p, std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
::operator delete[](p);
|
||||
}
|
||||
|
||||
|
||||
void *AlignedMemoryResource::DoAllocate(std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
void *p;
|
||||
if (!alignment) { alignment = sizeof(long double); }
|
||||
MFEM_VERIFY(posix_memalign(&p, alignment, bytes) == 0,
|
||||
"error in posix_memalign(): " << strerror(errno));
|
||||
return p;
|
||||
}
|
||||
|
||||
void AlignedMemoryResource::DoDeallocate(void *p, std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
free(p);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
@@ -1,70 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
|
||||
#define MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include <cstddef>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Polymorphic memory resource. Similar to C++17's std::pmr::memory_resource.
|
||||
class MemoryResource
|
||||
{
|
||||
protected:
|
||||
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment) = 0;
|
||||
virtual void DoDeallocate(void* p, std::size_t bytes,
|
||||
std::size_t alignment) = 0;
|
||||
|
||||
public:
|
||||
// Implicitly defined default & copy constructors
|
||||
|
||||
/// Virtual destructor.
|
||||
virtual ~MemoryResource() { }
|
||||
|
||||
/// If alignment == 0, use default alignment.
|
||||
void *Allocate(std::size_t bytes, std::size_t alignment = 0)
|
||||
{ return DoAllocate(bytes, alignment); }
|
||||
|
||||
/// If alignment == 0, use default alignment.
|
||||
void Deallocate(void *p, std::size_t bytes, std::size_t alignment = 0)
|
||||
{ DoDeallocate(p, bytes, alignment); }
|
||||
};
|
||||
|
||||
|
||||
/** @brief Dynamic host memory resource using operator new[](std::size_t) for
|
||||
allocation and operator delete[](void*) for deallocation. */
|
||||
class NewDeleteMemoryResource : public MemoryResource
|
||||
{
|
||||
protected:
|
||||
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment);
|
||||
virtual void DoDeallocate(void *p, std::size_t bytes, std::size_t alignment);
|
||||
};
|
||||
|
||||
|
||||
/** @brief Dynamic host memory resource using posix_memalign() for aligned
|
||||
allocation and free() for deallocation. */
|
||||
class AlignedMemoryResource : public MemoryResource
|
||||
{
|
||||
protected:
|
||||
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment);
|
||||
virtual void DoDeallocate(void *p, std::size_t bytes, std::size_t alignment);
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
|
||||
@@ -1,234 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
|
||||
#define MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "utils.hpp"
|
||||
#include "../../general/error.hpp"
|
||||
#include <cstddef>
|
||||
|
||||
// #define MFEM_TRACE_SHARED_PTR
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
#include "../../general/globals.hpp"
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Base class for classes with simple reference counting.
|
||||
/** Reference counting is performed by the class SharedPtr. */
|
||||
class RefCounted
|
||||
{
|
||||
private:
|
||||
mutable unsigned ref_count;
|
||||
|
||||
/// Only class SharedPtr can access ref_count.
|
||||
template <typename T> friend class SharedPtr;
|
||||
|
||||
public:
|
||||
RefCounted() : ref_count(0) { }
|
||||
|
||||
/** @brief Prevent SharedPtr objects from deleting this object by
|
||||
incrementing the reference counter by one. */
|
||||
void DontDelete() const { ++ref_count; }
|
||||
};
|
||||
|
||||
|
||||
/** @brief Smart pointer class that manages objects of type T derived from class
|
||||
RefCounted. */
|
||||
/** This class is generally meant to work with dynamically allocated object,
|
||||
specifically objects allocated with operator new(). It will invoke operator
|
||||
delete() to destroy the managed object when its reference counter reaches
|
||||
zero. This behavior can be overriden by calling RefCounted::DontDelete() to
|
||||
ensure that an object will not be deleted by a SharedPtr that holds a
|
||||
pointer to it.
|
||||
@note This class is NOT thread-safe and does not support circular ownership.
|
||||
*/
|
||||
template <typename T>
|
||||
class SharedPtr
|
||||
{
|
||||
public:
|
||||
typedef T stored_type;
|
||||
|
||||
private:
|
||||
T *ptr;
|
||||
|
||||
void Init(T *new_ptr)
|
||||
{
|
||||
ptr = new_ptr;
|
||||
if (ptr) { ++ptr->RefCounted::ref_count; }
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
#elif 0
|
||||
mfem::out << " [" << _MFEM_FUNC_NAME << "]: ptr = " << ptr;
|
||||
if (ptr)
|
||||
{
|
||||
mfem::out << ", new ref_count = " << ptr->RefCounted::ref_count;
|
||||
}
|
||||
mfem::out << '\n';
|
||||
#endif
|
||||
}
|
||||
void Destroy()
|
||||
{
|
||||
MFEM_ASSERT(!ptr || ptr->RefCounted::ref_count >= 1, "invalid use");
|
||||
if (ptr && --ptr->RefCounted::ref_count == 0) { delete ptr; }
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
#elif 0
|
||||
mfem::out << " [" << _MFEM_FUNC_NAME << "]: ptr = " << ptr;
|
||||
if (ptr)
|
||||
{
|
||||
mfem::out << ", new ref_count = " << ptr->RefCounted::ref_count;
|
||||
}
|
||||
mfem::out << '\n';
|
||||
#endif
|
||||
}
|
||||
|
||||
public:
|
||||
SharedPtr() : ptr(NULL)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]: ptr = " << ptr << '\n';
|
||||
#endif
|
||||
}
|
||||
|
||||
SharedPtr(const SharedPtr &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Init(other.ptr);
|
||||
}
|
||||
|
||||
template <typename U>
|
||||
SharedPtr(const SharedPtr<U> &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Init(other.Get());
|
||||
}
|
||||
|
||||
explicit SharedPtr(T *p)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Init(p);
|
||||
}
|
||||
|
||||
~SharedPtr()
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Destroy();
|
||||
}
|
||||
|
||||
SharedPtr &operator=(const SharedPtr &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Reset(other.ptr); return *this;
|
||||
}
|
||||
|
||||
template <typename U>
|
||||
SharedPtr &operator=(const SharedPtr<U> &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Reset(other.Get()); return *this;
|
||||
}
|
||||
|
||||
T &operator*() const { return *ptr; }
|
||||
T *operator->() const { return ptr; }
|
||||
|
||||
operator bool() const { return ptr; }
|
||||
bool operator!() const { return !ptr; }
|
||||
|
||||
template <typename U>
|
||||
bool operator==(const SharedPtr<U> &other) const
|
||||
{ return ptr == other.Get(); }
|
||||
template <typename U>
|
||||
bool operator!=(const SharedPtr<U> &other) const
|
||||
{ return ptr != other.Get(); }
|
||||
|
||||
// Comparison to any type convertible to void *, e.g. the type of NULL.
|
||||
template <typename U>
|
||||
bool operator==(const U &p) const { return ptr == (void*) p; }
|
||||
template <typename U>
|
||||
bool operator!=(const U &p) const { return ptr != (void*) p; }
|
||||
|
||||
T *Get() const { return ptr; }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t *As() const { return util::As<derived_t>(ptr); }
|
||||
|
||||
unsigned UseCount() const { return ptr ? ptr->RefCounted::ref_count : 0; }
|
||||
|
||||
void Reset()
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Destroy();
|
||||
ptr = NULL;
|
||||
}
|
||||
|
||||
/// The type U* needs to be implicitly convertible to T*
|
||||
template <typename U>
|
||||
void Reset(U *new_ptr)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
if (ptr != new_ptr) { Destroy(); Init(new_ptr); }
|
||||
}
|
||||
|
||||
void Swap(SharedPtr &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
std::swap(ptr, other.ptr);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template <class T>
|
||||
inline void Swap(SharedPtr<T> &a, SharedPtr<T> &b) { a.Swap(b); }
|
||||
|
||||
|
||||
class PLayout;
|
||||
typedef SharedPtr<PLayout> DLayout;
|
||||
|
||||
class PArray;
|
||||
typedef SharedPtr<PArray> DArray;
|
||||
|
||||
class PVector;
|
||||
typedef SharedPtr<PVector> DVector;
|
||||
|
||||
class PFiniteElementSpace;
|
||||
typedef SharedPtr<PFiniteElementSpace> DFiniteElementSpace;
|
||||
|
||||
class PBilinearForm;
|
||||
typedef SharedPtr<PBilinearForm> DBilinearForm;
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
|
||||
@@ -1,52 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_UTILS_HPP
|
||||
#define MFEM_BACKENDS_BASE_UTILS_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "../../general/error.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace util
|
||||
{
|
||||
|
||||
//
|
||||
// Inline methods
|
||||
//
|
||||
|
||||
/// TODO: doxygen
|
||||
template <typename derived_t, typename base_t>
|
||||
inline derived_t *As(base_t *base_obj)
|
||||
{
|
||||
MFEM_ASSERT(dynamic_cast<derived_t*>(base_obj) != NULL,
|
||||
"invalid object type");
|
||||
return static_cast<derived_t*>(base_obj);
|
||||
}
|
||||
|
||||
/// TODO: doxygen
|
||||
template <typename derived_t, typename base_t>
|
||||
inline derived_t *Is(base_t *base_obj)
|
||||
{
|
||||
return dynamic_cast<derived_t*>(base_obj);
|
||||
}
|
||||
|
||||
} // namespace mfem::util
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_UTILS_HPP
|
||||
@@ -1,153 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_VECTOR_HPP
|
||||
#define MFEM_BACKENDS_BASE_VECTOR_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "../../general/scalars.hpp"
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Polymorphic vector - array of scalars.
|
||||
class PVector : virtual public PArray
|
||||
{
|
||||
protected:
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/** @brief Create and return a new vector of the same dynamic type as this
|
||||
vector using the same layout with entries specified by @a buffer_type_id
|
||||
which should be a constant defined by the `value` field in a
|
||||
specialization of the template class mfem::ScalarId.
|
||||
|
||||
Returns NULL if allocation fails.
|
||||
|
||||
If @a copy_data is true, the contents of this vector is copied to the new
|
||||
vector; otherwise, the new vector remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the vector data of the newly created
|
||||
object (in @a *buffer), if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
virtual PVector *DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const = 0;
|
||||
|
||||
/** @brief Compute and return the dot product of @a *this and @a x. In the
|
||||
case of an MPI-parallel vector, the result must be the MPI-global dot
|
||||
product. */
|
||||
/** Both vectors must have the same dynamic type and layout. */
|
||||
virtual void DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const = 0;
|
||||
|
||||
// TODO: add reduction operations: min, max, sum
|
||||
|
||||
/// Perform the operation @a *this = @a a @a x + @a b @a y.
|
||||
/** Rules:
|
||||
- the dynamic type of both @a x and @a y is the same as that of @a *this
|
||||
- if @a a == 0, neither @a x nor its data are accessed
|
||||
- if @a b == 0, neither @a y nor its data are accessed
|
||||
- @a x's data is never the same as @a y's data, unless @a a == 0, or
|
||||
@a b == 0
|
||||
- @a x's data or @a y's data may be the same as the data of @a *this
|
||||
- all accessed vectors, @a x, @a y, and @a *this have the same layout. */
|
||||
virtual void DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id) = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
public:
|
||||
/** @brief Create a PVector. */
|
||||
/** The @a layout must be valid in the sense that layout != NULL and
|
||||
layout->HasEngine() == true. */
|
||||
PVector(PLayout &p_layout)
|
||||
: PArray(p_layout) { }
|
||||
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
|
||||
|
||||
// TODO: Error handling ... handle errors at the Engine level, at the class
|
||||
// level, or at the method level?
|
||||
|
||||
// TODO: Asynchronous execution interface ...
|
||||
|
||||
// TODO: Multi-vector interface ...
|
||||
|
||||
|
||||
/**
|
||||
@name Public virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/** @brief Create and return a new vector of the same dynamic type as this
|
||||
vector using the same layout with entries of type @a scalar_t.
|
||||
|
||||
If @a copy_data is true, the contents of this vector is copied to the new
|
||||
vector; otherwise, the new vector remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the vector data of the newly created
|
||||
object (in @a *buffer) , if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
template <typename scalar_t>
|
||||
DVector Clone(bool copy_data, scalar_t **buffer) const
|
||||
{
|
||||
return DVector(DoVectorClone(copy_data, (void**)buffer,
|
||||
ScalarId<scalar_t>::value));
|
||||
}
|
||||
|
||||
/** @brief Compute and return the dot product of @a *this and @a x. In the
|
||||
case of an MPI-parallel vector, the result must be the MPI-global dot
|
||||
product. */
|
||||
/** Both vectors must have the same dynamic type and layout. */
|
||||
template <typename scalar_t>
|
||||
scalar_t DotProduct(const PVector &x) const
|
||||
{
|
||||
scalar_t result;
|
||||
DoDotProduct(x, &result, ScalarId<scalar_t>::value);
|
||||
return result;
|
||||
}
|
||||
|
||||
// TODO: add reduction operations: min, max, sum
|
||||
|
||||
/// Perform the operation @a *this = @a a @a x + @a b @a y.
|
||||
/** Rules:
|
||||
- the dynamic type of both @a x and @a y is the same as that of @a *this
|
||||
- if @a a == 0, neither @a x nor its data are accessed
|
||||
- if @a b == 0, neither @a y nor its data are accessed
|
||||
- @a x's data is never the same as @a y's data, unless @a a == 0, or
|
||||
@a b == 0
|
||||
- @a x's data or @a y's data may be the same as the data of @a *this
|
||||
- all accessed vectors, @a x, @a y, and @a *this have the same layout. */
|
||||
template <typename scalar_t>
|
||||
void Axpby(const scalar_t &a, const PVector &x,
|
||||
const scalar_t &b, const PVector &y)
|
||||
{ if (Size()) { DoAxpby(&a, x, &b, y, ScalarId<scalar_t>::value); } }
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_VECTOR_HPP
|
||||
@@ -1,66 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
ELEMENT_BATCH : How many elements are in each
|
||||
. computation batch
|
||||
NUM_DOFS_1D : Dofs in the 1D segments
|
||||
NUM_DOFS_2D : Dofs in the 2D faces
|
||||
NUM_DOFS_3D : Dofs in the 3D domain
|
||||
NUM_QUAD_1D : Dofs in the 1D segments
|
||||
NUM_QUAD_2D : Dofs in the 2D faces
|
||||
NUM_QUAD_3D : Dofs in the 3D domain
|
||||
NUM_MAX_1D : max(NUM_QUAD_1D, NUM_DOFS_1D)
|
||||
NUM_QUAD_DOFS_1D: NUM_QUAD_1D * NUM_DOFS_1D
|
||||
COEFF_ARGS : Code that passes required arguments to the kernel
|
||||
COEFF : Code that computes the coefficient
|
||||
================================================
|
||||
|
||||
[MISSING]
|
||||
- Add support to auto-pick @dim and use @idxOrder on stack arrays
|
||||
| double a[2][2];
|
||||
| a[0][1]; <-- regular index
|
||||
| a(0,1); <-- uses @idxOrder a[0][1] or a[1][0]
|
||||
- Add support for @idxOrder to change indexing order after allocation
|
||||
| double a[2][2] @idxOrder(0,1);
|
||||
| a(0,1) -> a[1][0]
|
||||
| @set(a, idxOrder(1,0));
|
||||
| a(0,1) -> a[0][1]
|
||||
- Add support to iterate over loop depending on mode
|
||||
| for(i; @inner) {
|
||||
| for(0 < j < N) {} <-- ++j or j += block?
|
||||
| }
|
||||
*/
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# ifdef OCCA_USING_GPU
|
||||
# if USING_LOW_ORDER
|
||||
# include "mfem-occa://diffusion/tensor/gpuHighOrder.okl"
|
||||
# else
|
||||
# include "mfem-occa://diffusion/tensor/gpuHighOrder.okl"
|
||||
# endif
|
||||
# else
|
||||
# include "mfem-occa://diffusion/tensor/cpu.okl"
|
||||
# endif
|
||||
#else
|
||||
# ifdef OCCA_USING_GPU
|
||||
# if USING_LOW_ORDER
|
||||
# include "mfem-occa://diffusion/simplex/gpuHighOrder.okl"
|
||||
# else
|
||||
# include "mfem-occa://diffusion/simplex/gpuHighOrder.okl"
|
||||
# endif
|
||||
# else
|
||||
# include "mfem-occa://diffusion/simplex/cpu.okl"
|
||||
# endif
|
||||
#endif
|
||||
@@ -1,66 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
ELEMENT_BATCH : How many elements are in each
|
||||
. computation batch
|
||||
NUM_DOFS_1D : Dofs in the 1D segments
|
||||
NUM_DOFS_2D : Dofs in the 2D faces
|
||||
NUM_DOFS_3D : Dofs in the 3D domain
|
||||
NUM_QUAD_1D : Dofs in the 1D segments
|
||||
NUM_QUAD_2D : Dofs in the 2D faces
|
||||
NUM_QUAD_3D : Dofs in the 3D domain
|
||||
NUM_MAX_1D : max(NUM_QUAD_1D, NUM_DOFS_1D)
|
||||
NUM_QUAD_DOFS_1D: NUM_QUAD_1D * NUM_DOFS_1D
|
||||
COEFF_ARGS : Code that passes required arguments to the kernel
|
||||
COEFF : Code that computes the coefficient
|
||||
================================================
|
||||
|
||||
[MISSING]
|
||||
- Add support to auto-pick @dim and use @idxOrder on stack arrays
|
||||
| double a[2][2];
|
||||
| a[0][1]; <-- regular index
|
||||
| a(0,1); <-- uses @idxOrder a[0][1] or a[1][0]
|
||||
- Add support for @idxOrder to change indexing order after allocation
|
||||
| double a[2][2] @idxOrder(0,1);
|
||||
| a(0,1) -> a[1][0]
|
||||
| @set(a, idxOrder(1,0));
|
||||
| a(0,1) -> a[0][1]
|
||||
- Add support to iterate over loop depending on mode
|
||||
| for(i; @inner) {
|
||||
| for(0 < j < N) {} <-- ++j or j += block?
|
||||
| }
|
||||
*/
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# ifdef OCCA_USING_GPU
|
||||
# if USING_LOW_ORDER
|
||||
# include "mfem-occa://mass/tensor/gpuHighOrder.okl"
|
||||
# else
|
||||
# include "mfem-occa://mass/tensor/gpuHighOrder.okl"
|
||||
# endif
|
||||
# else
|
||||
# include "mfem-occa://mass/tensor/cpu.okl"
|
||||
# endif
|
||||
#else
|
||||
# ifdef OCCA_USING_GPU
|
||||
# if USING_LOW_ORDER
|
||||
# include "mfem-occa://mass/simplex/gpuHighOrder.okl"
|
||||
# else
|
||||
# include "mfem-occa://mass/simplex/gpuHighOrder.okl"
|
||||
# endif
|
||||
# else
|
||||
# include "mfem-occa://mass/simplex/cpu.okl"
|
||||
# endif
|
||||
#endif
|
||||
@@ -1,38 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
ELEMENT_BATCH : How many elements are in each
|
||||
. computation batch
|
||||
NUM_DOFS_1D : Dofs in the 1D segments
|
||||
NUM_DOFS_2D : Dofs in the 2D faces
|
||||
NUM_DOFS_3D : Dofs in the 3D domain
|
||||
NUM_QUAD_1D : Dofs in the 1D segments
|
||||
NUM_QUAD_2D : Dofs in the 2D faces
|
||||
NUM_QUAD_3D : Dofs in the 3D domain
|
||||
NUM_MAX_1D : max(NUM_QUAD_1D, NUM_DOFS_1D)
|
||||
NUM_QUAD_DOFS_1D: NUM_QUAD_1D * NUM_DOFS_1D
|
||||
CONST_COEFF : If the coefficient is constant, pass it
|
||||
. as a define
|
||||
================================================
|
||||
|
||||
[MISSING]
|
||||
See kernels/DiffusionIntegrator.okl
|
||||
*/
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# ifndef OCCA_USING_GPU
|
||||
# include "mfem-occa://vmass/tensor/cpu.okl"
|
||||
# endif
|
||||
#endif
|
||||
@@ -1,121 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
PArray *Array::DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
Array *new_array = new Array(OccaLayout(), item_size);
|
||||
if (copy_data)
|
||||
{
|
||||
new_array->slice.copyFrom(slice);
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
*buffer = new_array->GetBuffer();
|
||||
}
|
||||
return new_array;
|
||||
}
|
||||
|
||||
int Array::DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size)
|
||||
{
|
||||
MFEM_ASSERT(dynamic_cast<Layout *>(&new_layout) != NULL,
|
||||
"new_layout is not an OCCA Layout");
|
||||
Layout *lt = static_cast<Layout *>(&new_layout);
|
||||
int err = OccaResize(lt, item_size);
|
||||
if (!err && buffer)
|
||||
{
|
||||
*buffer = GetBuffer();
|
||||
}
|
||||
return err;
|
||||
}
|
||||
|
||||
void *Array::DoPullData(void *buffer, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
if (!slice.getDevice().hasSeparateMemorySpace())
|
||||
{
|
||||
return slice.ptr();
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
slice.copyTo(buffer);
|
||||
}
|
||||
return buffer;
|
||||
}
|
||||
|
||||
void Array::DoFill(const void *value_ptr, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
switch (item_size)
|
||||
{
|
||||
case sizeof(int8_t):
|
||||
OccaFill(*(const int8_t *)value_ptr);
|
||||
break;
|
||||
case sizeof(int16_t):
|
||||
OccaFill(*(const int16_t *)value_ptr);
|
||||
break;
|
||||
case sizeof(int32_t):
|
||||
OccaFill(*(const int32_t *)value_ptr);
|
||||
break;
|
||||
// case sizeof(int64_t):
|
||||
// OccaFill(*(const int64_t *)value_ptr);
|
||||
// break;
|
||||
case sizeof(double):
|
||||
OccaFill(*(const double *)value_ptr);
|
||||
break;
|
||||
// case sizeof(::occa::double2):
|
||||
// OccaFill(*(const ::occa::double2 *)value_ptr);
|
||||
// break;
|
||||
default:
|
||||
MFEM_ABORT("item_size = " << item_size << " is not supported");
|
||||
}
|
||||
}
|
||||
|
||||
void Array::DoPushData(const void *src_buffer, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
if (slice.getDevice().hasSeparateMemorySpace() || slice.ptr() != src_buffer)
|
||||
{
|
||||
slice.copyFrom(src_buffer);
|
||||
}
|
||||
}
|
||||
|
||||
void Array::DoAssign(const PArray &src, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
// Note: static_cast can not be used here since PArray is a virtual base
|
||||
// class.
|
||||
const Array *source = dynamic_cast<const Array *>(&src);
|
||||
MFEM_ASSERT(source != NULL, "invalid source Array type");
|
||||
OccaAssign(*source);
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,175 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_ARRAY_HPP
|
||||
#define MFEM_BACKENDS_OCCA_ARRAY_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "layout.hpp"
|
||||
#include "../base/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Array : public virtual PArray
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// DLayout layout;
|
||||
|
||||
// Always true: Size()*item_size == slice.size() <= data.size()
|
||||
mutable ::occa::memory data, slice;
|
||||
|
||||
//
|
||||
// Virtual interface
|
||||
//
|
||||
|
||||
virtual PArray *DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const;
|
||||
|
||||
virtual int DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size);
|
||||
|
||||
virtual void *DoPullData(void *buffer, std::size_t item_size);
|
||||
|
||||
virtual void DoFill(const void *value_ptr, std::size_t item_size);
|
||||
|
||||
virtual void DoPushData(const void *src_buffer, std::size_t item_size);
|
||||
|
||||
virtual void DoAssign(const PArray &src, std::size_t item_size);
|
||||
|
||||
//
|
||||
// Auxiliary methods
|
||||
//
|
||||
|
||||
inline void *GetBuffer() const;
|
||||
|
||||
public:
|
||||
Array(const Engine &e)
|
||||
: PArray(*(new Layout(e, 0))),
|
||||
data(e.Alloc(0)),
|
||||
slice(data)
|
||||
{ }
|
||||
|
||||
Array(Layout <, std::size_t item_size)
|
||||
: PArray(lt),
|
||||
data(lt.OccaEngine().Alloc(lt.Size()*item_size)),
|
||||
slice(data)
|
||||
{ }
|
||||
|
||||
virtual ~Array() { }
|
||||
|
||||
inline void MakeRef(Array &master);
|
||||
|
||||
Layout &OccaLayout() const { return layout->As<Layout>(); }
|
||||
|
||||
const Engine &OccaEngine() const { return OccaLayout().OccaEngine(); }
|
||||
|
||||
::occa::memory &OccaMem() { return slice; }
|
||||
const ::occa::memory &OccaMem() const { return slice; }
|
||||
|
||||
inline int OccaResize(Layout *lt, std::size_t item_size);
|
||||
|
||||
inline int OccaResize(std::size_t new_size, std::size_t item_size);
|
||||
|
||||
template <typename T>
|
||||
inline void OccaFill(const T val);
|
||||
|
||||
inline void OccaAssign(const Array &src);
|
||||
|
||||
inline void OccaPush(const void *src);
|
||||
};
|
||||
|
||||
|
||||
//
|
||||
// Inline methods
|
||||
//
|
||||
|
||||
inline void *Array::GetBuffer() const
|
||||
{
|
||||
if (!slice.getDevice().hasSeparateMemorySpace())
|
||||
{
|
||||
return slice.ptr();
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
inline int Array::OccaResize(Layout *lt, std::size_t item_size)
|
||||
{
|
||||
layout.Reset(lt); // Reset() checks if the pointer is the same
|
||||
const std::size_t new_bytes = lt->Size()*item_size;
|
||||
if (data.size() < new_bytes ||
|
||||
data.getDevice() != lt->OccaEngine().GetDevice())
|
||||
{
|
||||
data = lt->OccaEngine().Alloc(new_bytes);
|
||||
slice = data;
|
||||
// If memory allocation fails - an exception is thrown.
|
||||
}
|
||||
else if (slice.size() != new_bytes)
|
||||
{
|
||||
slice = data.slice(0, new_bytes);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
inline void Array::MakeRef(Array &master)
|
||||
{
|
||||
layout = master.layout;
|
||||
data = master.data;
|
||||
slice = master.slice;
|
||||
}
|
||||
|
||||
inline int Array::OccaResize(std::size_t new_size, std::size_t item_size)
|
||||
{
|
||||
Layout &ol = OccaLayout();
|
||||
ol.OccaResize(new_size);
|
||||
return OccaResize(&ol, item_size);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
inline void Array::OccaFill(const T val)
|
||||
{
|
||||
::occa::linalg::operator_eq<T>(slice, val);
|
||||
}
|
||||
|
||||
inline void Array::OccaAssign(const Array &src)
|
||||
{
|
||||
if (slice != src.slice && slice.size() != 0)
|
||||
{
|
||||
MFEM_ASSERT(slice.size() == src.slice.size(), "");
|
||||
slice.copyFrom(src.slice);
|
||||
}
|
||||
}
|
||||
|
||||
inline void Array::OccaPush(const void *src)
|
||||
{
|
||||
if (slice.size() != 0)
|
||||
{
|
||||
slice.copyFrom(src);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_ARRAY_HPP
|
||||
@@ -1,47 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
bool Backend::Supports(const std::string &engine_spec) const
|
||||
{
|
||||
// TODO: check if 'engine_spec' is valid OCCA string.
|
||||
return true;
|
||||
}
|
||||
|
||||
mfem::Engine *Create(const std::string &engine_spec)
|
||||
{
|
||||
return new Engine(engine_spec);
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec)
|
||||
{
|
||||
return new Engine(comm, engine_spec);
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,49 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_BACKEND_HPP
|
||||
#define MFEM_BACKENDS_OCCA_BACKEND_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
// Only the Backend and Engine classes should be exposed through "backend.hpp"
|
||||
#include "../base/backend.hpp"
|
||||
#include "engine.hpp"
|
||||
#include <occa.hpp>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Backend : public mfem::Backend
|
||||
{
|
||||
public:
|
||||
virtual ~Backend();
|
||||
|
||||
virtual bool Supports(const std::string &engine_spec) const;
|
||||
|
||||
virtual mfem::Engine *Create(const std::string &engine_spec);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
virtual mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec);
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_BACKEND_HPP
|
||||
@@ -1,538 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "../../fem/bilinearform.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
OccaBilinearForm::OccaBilinearForm(FiniteElementSpace *ofespace_) :
|
||||
Operator(ofespace_->OccaVLayout()),
|
||||
localX(ofespace_->OccaEVLayout()),
|
||||
localY(ofespace_->OccaEVLayout())
|
||||
{
|
||||
Init(ofespace_->OccaEngine(), ofespace_, ofespace_);
|
||||
}
|
||||
|
||||
OccaBilinearForm::OccaBilinearForm(FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_) :
|
||||
Operator(otrialFESpace_->OccaVLayout(),
|
||||
otestFESpace_->OccaVLayout()),
|
||||
localX(otrialFESpace_->OccaEVLayout()),
|
||||
localY(otestFESpace_->OccaEVLayout())
|
||||
{
|
||||
Init(otrialFESpace_->OccaEngine(), otrialFESpace_, otestFESpace_);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::Init(const Engine &e,
|
||||
FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_)
|
||||
{
|
||||
engine.Reset(&e);
|
||||
|
||||
otrialFESpace = otrialFESpace_;
|
||||
trialFESpace = otrialFESpace_->GetFESpace();
|
||||
|
||||
otestFESpace = otestFESpace_;
|
||||
testFESpace = otestFESpace_->GetFESpace();
|
||||
|
||||
mesh = trialFESpace->GetMesh();
|
||||
|
||||
const int elements = GetNE();
|
||||
|
||||
const int trialVDim = trialFESpace->GetVDim();
|
||||
|
||||
const int trialLocalDofs = otrialFESpace->GetLocalDofs();
|
||||
const int testLocalDofs = otestFESpace->GetLocalDofs();
|
||||
|
||||
// First-touch policy when running with OpenMP
|
||||
if (GetDevice().mode() == "OpenMP")
|
||||
{
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
::occa::kernel initLocalKernel =
|
||||
GetDevice().buildKernel(okl_path + "utils.okl",
|
||||
"InitLocalVector");
|
||||
|
||||
const std::size_t sd = sizeof(double);
|
||||
const uint64_t trialEntries = sd * (elements * trialLocalDofs);
|
||||
const uint64_t testEntries = sd * (elements * testLocalDofs);
|
||||
for (int v = 0; v < trialVDim; ++v)
|
||||
{
|
||||
const uint64_t trialOffset = v * trialEntries;
|
||||
const uint64_t testOffset = v * testEntries;
|
||||
|
||||
initLocalKernel(elements, trialLocalDofs,
|
||||
localX.OccaMem().slice(trialOffset, trialEntries));
|
||||
initLocalKernel(elements, testLocalDofs,
|
||||
localY.OccaMem().slice(testOffset, testEntries));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int OccaBilinearForm::BaseGeom() const
|
||||
{
|
||||
return mesh->GetElementBaseGeometry();
|
||||
}
|
||||
|
||||
int OccaBilinearForm::GetDim() const
|
||||
{
|
||||
return mesh->Dimension();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetNE() const
|
||||
{
|
||||
return mesh->GetNE();
|
||||
}
|
||||
|
||||
Mesh& OccaBilinearForm::GetMesh() const
|
||||
{
|
||||
return *mesh;
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaBilinearForm::GetTrialOccaFESpace() const
|
||||
{
|
||||
return *otrialFESpace;
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaBilinearForm::GetTestOccaFESpace() const
|
||||
{
|
||||
return *otestFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaBilinearForm::GetTrialFESpace() const
|
||||
{
|
||||
return *trialFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaBilinearForm::GetTestFESpace() const
|
||||
{
|
||||
return *testFESpace;
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTrialNDofs() const
|
||||
{
|
||||
return trialFESpace->GetNDofs();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTestNDofs() const
|
||||
{
|
||||
return testFESpace->GetNDofs();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTrialVDim() const
|
||||
{
|
||||
return trialFESpace->GetVDim();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTestVDim() const
|
||||
{
|
||||
return testFESpace->GetVDim();
|
||||
}
|
||||
|
||||
const FiniteElement& OccaBilinearForm::GetTrialFE(const int i) const
|
||||
{
|
||||
return *(trialFESpace->GetFE(i));
|
||||
}
|
||||
|
||||
const FiniteElement& OccaBilinearForm::GetTestFE(const int i) const
|
||||
{
|
||||
return *(testFESpace->GetFE(i));
|
||||
}
|
||||
|
||||
// Adds new Domain Integrator.
|
||||
void OccaBilinearForm::AddDomainIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, DomainIntegrator);
|
||||
}
|
||||
|
||||
// Adds new Boundary Integrator.
|
||||
void OccaBilinearForm::AddBoundaryIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, BoundaryIntegrator);
|
||||
}
|
||||
|
||||
// Adds new interior Face Integrator.
|
||||
void OccaBilinearForm::AddInteriorFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, InteriorFaceIntegrator);
|
||||
}
|
||||
|
||||
// Adds new boundary Face Integrator.
|
||||
void OccaBilinearForm::AddBoundaryFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, BoundaryFaceIntegrator);
|
||||
}
|
||||
|
||||
// Adds Integrator based on OccaIntegratorType
|
||||
void OccaBilinearForm::AddIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props,
|
||||
const OccaIntegratorType itype)
|
||||
{
|
||||
if (integrator == NULL)
|
||||
{
|
||||
std::stringstream error_ss;
|
||||
error_ss << "OccaBilinearForm::";
|
||||
switch (itype)
|
||||
{
|
||||
case DomainIntegrator : error_ss << "AddDomainIntegrator"; break;
|
||||
case BoundaryIntegrator : error_ss << "AddBoundaryIntegrator"; break;
|
||||
case InteriorFaceIntegrator: error_ss << "AddInteriorFaceIntegrator"; break;
|
||||
case BoundaryFaceIntegrator: error_ss << "AddBoundaryFaceIntegrator"; break;
|
||||
}
|
||||
error_ss << " (...):\n"
|
||||
<< " Integrator is NULL";
|
||||
const std::string error = error_ss.str();
|
||||
mfem_error(error.c_str());
|
||||
}
|
||||
integrator->SetupIntegrator(*this, baseKernelProps + props, itype);
|
||||
integrators.push_back(integrator);
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTrialProlongation() const
|
||||
{
|
||||
return otrialFESpace->GetProlongationOperator();
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTestProlongation() const
|
||||
{
|
||||
return otestFESpace->GetProlongationOperator();
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTrialRestriction() const
|
||||
{
|
||||
return otrialFESpace->GetRestrictionOperator();
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTestRestriction() const
|
||||
{
|
||||
return otestFESpace->GetRestrictionOperator();
|
||||
}
|
||||
|
||||
void OccaBilinearForm::Assemble()
|
||||
{
|
||||
// [MISSING] Find geometric information that is needed by intergrators
|
||||
// to share between integrators.
|
||||
const int integratorCount = (int) integrators.size();
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
integrators[i]->Assemble();
|
||||
}
|
||||
}
|
||||
|
||||
void OccaBilinearForm::FormLinearSystem(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *&Aout,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior)
|
||||
{
|
||||
FormOperator(constraintList, Aout);
|
||||
InitRHS(constraintList, x, b, Aout, X, B, copy_interior);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::FormOperator(const mfem::Array<int> &constraintList,
|
||||
mfem::Operator *&Aout)
|
||||
{
|
||||
const mfem::Operator *trialP = GetTrialProlongation();
|
||||
const mfem::Operator *testP = GetTestProlongation();
|
||||
mfem::Operator *rap = this;
|
||||
|
||||
if (trialP)
|
||||
{
|
||||
rap = new RAPOperator(*testP, *this, *trialP);
|
||||
}
|
||||
|
||||
Aout = new OccaConstrainedOperator(rap, constraintList,
|
||||
rap != this);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::InitRHS(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *A,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior)
|
||||
{
|
||||
// FIXME: move these kernels to the Backend?
|
||||
static ::occa::kernelBuilder get_subvector_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_get_subvector",
|
||||
|
||||
"const int dof_i = v2[i];"
|
||||
"v0[i] = dof_i >= 0 ? v1[dof_i] : -v1[-dof_i - 1];",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}");
|
||||
|
||||
static ::occa::kernelBuilder set_subvector_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_set_subvector",
|
||||
"const int dof_i = v2[i];"
|
||||
"if (dof_i >= 0) { v0[dof_i] = v1[i]; }"
|
||||
"else { v0[-dof_i - 1] = -v1[i]; }",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}");
|
||||
|
||||
const mfem::Operator *P = GetTrialProlongation();
|
||||
const mfem::Operator *R = GetTrialRestriction();
|
||||
|
||||
if (P)
|
||||
{
|
||||
// Variational restriction with P
|
||||
B.Resize(P->InLayout());
|
||||
P->MultTranspose(b, B);
|
||||
X.Resize(R->OutLayout());
|
||||
R->Mult(x, X);
|
||||
}
|
||||
else
|
||||
{
|
||||
// rap, X and B point to the same data as this, x and b
|
||||
X.MakeRef(x);
|
||||
B.MakeRef(b);
|
||||
}
|
||||
|
||||
if (!copy_interior && constraintList.Size() > 0)
|
||||
{
|
||||
::occa::kernel get_subvector_kernel =
|
||||
get_subvector_builder.build(GetDevice());
|
||||
::occa::kernel set_subvector_kernel =
|
||||
set_subvector_builder.build(GetDevice());
|
||||
|
||||
const Array &constrList = constraintList.Get_PArray()->As<Array>();
|
||||
Vector subvec(constrList.OccaLayout());
|
||||
|
||||
get_subvector_kernel(constraintList.Size(),
|
||||
subvec.OccaMem(),
|
||||
X.Get_PVector()->As<Vector>().OccaMem(),
|
||||
constrList.OccaMem());
|
||||
|
||||
X.Fill(0.0);
|
||||
|
||||
set_subvector_kernel(constraintList.Size(),
|
||||
X.Get_PVector()->As<Vector>().OccaMem(),
|
||||
subvec.OccaMem(),
|
||||
constrList.OccaMem());
|
||||
}
|
||||
|
||||
// FIXME: add case for HypreParMatrix here
|
||||
OccaConstrainedOperator *cA = dynamic_cast<OccaConstrainedOperator*>(A);
|
||||
if (cA)
|
||||
{
|
||||
cA->EliminateRHS(X.Get_PVector()->As<Vector>(),
|
||||
B.Get_PVector()->As<Vector>());
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem_error("OccaBilinearForm::InitRHS expects an OccaConstrainedOperator");
|
||||
}
|
||||
}
|
||||
|
||||
// Matrix vector multiplication.
|
||||
void OccaBilinearForm::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
otrialFESpace->GlobalToLocal(x, localX);
|
||||
localY.OccaFill<double>(0.0);
|
||||
|
||||
const int integratorCount = (int) integrators.size();
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
integrators[i]->MultAdd(localX, localY);
|
||||
}
|
||||
|
||||
otestFESpace->LocalToGlobal(localY, y);
|
||||
}
|
||||
|
||||
// Matrix transpose vector multiplication.
|
||||
void OccaBilinearForm::MultTranspose_(const Vector &x, Vector &y) const
|
||||
{
|
||||
otestFESpace->GlobalToLocal(x, localX);
|
||||
localY.OccaFill<double>(0.0);
|
||||
|
||||
const int integratorCount = (int) integrators.size();
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
integrators[i]->MultTransposeAdd(localX, localY);
|
||||
}
|
||||
|
||||
otrialFESpace->LocalToGlobal(localY, y);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::OccaRecoverFEMSolution(const mfem::Vector &X,
|
||||
const mfem::Vector &b,
|
||||
mfem::Vector &x)
|
||||
{
|
||||
const mfem::Operator *P = this->GetTrialProlongation();
|
||||
if (P)
|
||||
{
|
||||
// Apply conforming prolongation
|
||||
x.Resize(P->OutLayout());
|
||||
P->Mult(X, x);
|
||||
}
|
||||
// Otherwise X and x point to the same data
|
||||
}
|
||||
|
||||
// Frees memory bilinear form.
|
||||
OccaBilinearForm::~OccaBilinearForm()
|
||||
{
|
||||
// Make sure all integrators free their data
|
||||
IntegratorVector::iterator it = integrators.begin();
|
||||
while (it != integrators.end())
|
||||
{
|
||||
delete *it;
|
||||
++it;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void BilinearForm::InitOccaBilinearForm()
|
||||
{
|
||||
// Init 'obform' using 'bform'
|
||||
MFEM_ASSERT(bform != NULL, "");
|
||||
MFEM_ASSERT(obform == NULL, "");
|
||||
|
||||
FiniteElementSpace &ofes =
|
||||
bform->FESpace()->Get_PFESpace()->As<FiniteElementSpace>();
|
||||
obform = new OccaBilinearForm(&ofes);
|
||||
|
||||
// Transfer domain integrators
|
||||
mfem::Array<mfem::BilinearFormIntegrator*> &dbfi = *bform->GetDBFI();
|
||||
for (int i = 0; i < dbfi.Size(); i++)
|
||||
{
|
||||
std::string integ_name(dbfi[i]->Name());
|
||||
Coefficient *scal_coeff = dbfi[i]->GetScalarCoefficient();
|
||||
ConstantCoefficient *const_coeff =
|
||||
dynamic_cast<ConstantCoefficient*>(scal_coeff);
|
||||
GridFunctionCoefficient *gridfunc_coeff =
|
||||
dynamic_cast<GridFunctionCoefficient*>(scal_coeff);
|
||||
// TODO: other types of coefficients ...
|
||||
|
||||
OccaCoefficient *ocoeff = NULL;
|
||||
if (const_coeff)
|
||||
{
|
||||
ocoeff = new OccaCoefficient(obform->OccaEngine(),
|
||||
const_coeff->constant);
|
||||
}
|
||||
else if (gridfunc_coeff)
|
||||
{
|
||||
ocoeff = new OccaCoefficient(obform->OccaEngine(),
|
||||
*gridfunc_coeff->GetGridFunction(), true);
|
||||
}
|
||||
else if (!scal_coeff)
|
||||
{
|
||||
ocoeff = new OccaCoefficient(obform->OccaEngine(), 1.0);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Coefficient type not supported");
|
||||
}
|
||||
|
||||
OccaIntegrator *ointeg = NULL;
|
||||
if (integ_name == "(undefined)")
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator does not define Name()");
|
||||
}
|
||||
else if (integ_name == "mass")
|
||||
{
|
||||
ointeg = new OccaMassIntegrator(*ocoeff);
|
||||
}
|
||||
else if (integ_name == "diffusion")
|
||||
{
|
||||
ointeg = new OccaDiffusionIntegrator(*ocoeff);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator [Name() = " << integ_name
|
||||
<< "] is not supported");
|
||||
}
|
||||
|
||||
// NOTE: The integrators copy ocoeff, so it can be deleted here so there
|
||||
// is no memory leak.
|
||||
delete ocoeff;
|
||||
|
||||
const mfem::IntegrationRule *ir = dbfi[i]->GetIntRule();
|
||||
if (ir) { ointeg->SetIntegrationRule(*ir); }
|
||||
|
||||
obform->AddDomainIntegrator(ointeg);
|
||||
}
|
||||
|
||||
// TODO: other types of integrators ...
|
||||
}
|
||||
|
||||
bool BilinearForm::Assemble()
|
||||
{
|
||||
if (obform == NULL) { InitOccaBilinearForm(); }
|
||||
|
||||
obform->Assemble();
|
||||
|
||||
return true; // --> host assembly is not needed
|
||||
}
|
||||
|
||||
void BilinearForm::FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::OperatorHandle &A)
|
||||
{
|
||||
if (A.Type() == mfem::Operator::ANY_TYPE)
|
||||
{
|
||||
mfem::Operator *Aout = NULL;
|
||||
obform->FormOperator(ess_tdof_list, Aout);
|
||||
A.Reset(Aout);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Operator::Type is not supported, type = " << A.Type());
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::OperatorHandle &A,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior)
|
||||
{
|
||||
FormSystemMatrix(ess_tdof_list, A);
|
||||
obform->InitRHS(ess_tdof_list, x, b, A.Ptr(), X, B, copy_interior);
|
||||
}
|
||||
|
||||
void BilinearForm::RecoverFEMSolution(const mfem::Vector &X,
|
||||
const mfem::Vector &b,
|
||||
mfem::Vector &x)
|
||||
{
|
||||
obform->OccaRecoverFEMSolution(X, b, x);
|
||||
}
|
||||
|
||||
BilinearForm::~BilinearForm()
|
||||
{
|
||||
delete obform;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,213 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
|
||||
#define MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "fespace.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
enum OccaIntegratorType
|
||||
{
|
||||
DomainIntegrator = 0,
|
||||
BoundaryIntegrator = 1,
|
||||
InteriorFaceIntegrator = 2,
|
||||
BoundaryFaceIntegrator = 3
|
||||
};
|
||||
|
||||
class OccaIntegrator;
|
||||
|
||||
|
||||
/** Class for bilinear form - "Matrix" with associated FE space and
|
||||
BLFIntegrators. */
|
||||
class OccaBilinearForm : public Operator
|
||||
{
|
||||
friend class OccaIntegrator;
|
||||
|
||||
protected:
|
||||
typedef std::vector<OccaIntegrator*> IntegratorVector;
|
||||
|
||||
SharedPtr<const Engine> engine;
|
||||
|
||||
// State information
|
||||
mutable mfem::Mesh *mesh;
|
||||
|
||||
mutable FiniteElementSpace *otrialFESpace;
|
||||
mutable mfem::FiniteElementSpace *trialFESpace;
|
||||
|
||||
mutable FiniteElementSpace *otestFESpace;
|
||||
mutable mfem::FiniteElementSpace *testFESpace;
|
||||
|
||||
IntegratorVector integrators;
|
||||
|
||||
// Device data
|
||||
::occa::properties baseKernelProps;
|
||||
|
||||
// The input and output vectors are mapped to local nodes for efficient
|
||||
// operations. In other words, they are E-vectors.
|
||||
// The size is: (number of elements) * (nodes in element) * (vector dim)
|
||||
mutable Vector localX, localY;
|
||||
|
||||
public:
|
||||
OccaBilinearForm(FiniteElementSpace *ofespace_);
|
||||
|
||||
OccaBilinearForm(FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_);
|
||||
|
||||
void Init(const Engine &e,
|
||||
FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_);
|
||||
|
||||
const Engine &OccaEngine() const { return *engine; }
|
||||
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return engine->GetDevice(idx); }
|
||||
|
||||
// Useful mesh Information
|
||||
int BaseGeom() const;
|
||||
int GetDim() const;
|
||||
int64_t GetNE() const;
|
||||
|
||||
mfem::Mesh& GetMesh() const;
|
||||
|
||||
FiniteElementSpace& GetTrialOccaFESpace() const;
|
||||
FiniteElementSpace& GetTestOccaFESpace() const;
|
||||
|
||||
mfem::FiniteElementSpace& GetTrialFESpace() const;
|
||||
mfem::FiniteElementSpace& GetTestFESpace() const;
|
||||
|
||||
// Useful FE information
|
||||
int64_t GetTrialNDofs() const;
|
||||
int64_t GetTestNDofs() const;
|
||||
|
||||
int64_t GetTrialVDim() const;
|
||||
int64_t GetTestVDim() const;
|
||||
|
||||
const mfem::FiniteElement& GetTrialFE(const int i) const;
|
||||
const mfem::FiniteElement& GetTestFE(const int i) const;
|
||||
|
||||
// Adds new Domain Integrator.
|
||||
void AddDomainIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds new Boundary Integrator.
|
||||
void AddBoundaryIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds new interior Face Integrator.
|
||||
void AddInteriorFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds new boundary Face Integrator.
|
||||
void AddBoundaryFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds Integrator based on OccaIntegratorType
|
||||
void AddIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props,
|
||||
const OccaIntegratorType itype);
|
||||
|
||||
virtual const mfem::Operator *GetTrialProlongation() const;
|
||||
virtual const mfem::Operator *GetTestProlongation() const;
|
||||
|
||||
virtual const mfem::Operator *GetTrialRestriction() const;
|
||||
virtual const mfem::Operator *GetTestRestriction() const;
|
||||
|
||||
// Assembles the form i.e. sums over all domain/bdr integrators.
|
||||
virtual void Assemble();
|
||||
|
||||
void FormLinearSystem(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *&Aout,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior = 0);
|
||||
|
||||
void FormOperator(const mfem::Array<int> &constraintList,
|
||||
mfem::Operator *&Aout);
|
||||
|
||||
void InitRHS(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *Aout,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior = 0);
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
|
||||
void OccaRecoverFEMSolution(const mfem::Vector &X, const mfem::Vector &b,
|
||||
mfem::Vector &x);
|
||||
|
||||
// Destroys bilinear form.
|
||||
~OccaBilinearForm();
|
||||
};
|
||||
|
||||
|
||||
/// TODO: doxygen
|
||||
class BilinearForm : public mfem::PBilinearForm
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// mfem::BilinearForm *bform;
|
||||
OccaBilinearForm *obform;
|
||||
|
||||
// Called from Assemble() if obform is NULL to initialize obform.
|
||||
void InitOccaBilinearForm();
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
BilinearForm(const Engine &e, mfem::BilinearForm &bf)
|
||||
: mfem::PBilinearForm(e, bf), obform(NULL) { }
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~BilinearForm();
|
||||
|
||||
/// Assemble the PBilinearForm.
|
||||
/** This method is called from the method mfem::BilinearForm::Assemble() of
|
||||
the associated mfem::BilinearForm, #bform.
|
||||
@returns True, if the host assembly should NOT be performed. */
|
||||
virtual bool Assemble();
|
||||
|
||||
virtual void FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::OperatorHandle &A);
|
||||
|
||||
virtual void FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::OperatorHandle &A,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior);
|
||||
|
||||
virtual void RecoverFEMSolution(const mfem::Vector &X, const mfem::Vector &b,
|
||||
mfem::Vector &x);
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
|
||||
@@ -1,954 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "bilininteg.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
std::map<std::string, OccaDofQuadMaps> OccaDofQuadMaps::AllDofQuadMaps;
|
||||
|
||||
OccaGeometry OccaGeometry::Get(::occa::device device,
|
||||
FiniteElementSpace &ofespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const int flags)
|
||||
{
|
||||
OccaGeometry geom;
|
||||
|
||||
mfem::Mesh &mesh = *(ofespace.GetMesh());
|
||||
if (!mesh.GetNodes())
|
||||
{
|
||||
mesh.SetCurvature(1, false, -1, mfem::Ordering::byVDIM);
|
||||
}
|
||||
mfem::GridFunction &nodes = *(mesh.GetNodes());
|
||||
const mfem::FiniteElementSpace &fespace = *(nodes.FESpace());
|
||||
const mfem::FiniteElement &fe = *(fespace.GetFE(0));
|
||||
|
||||
const int dims = fe.GetDim();
|
||||
const int elements = fespace.GetNE();
|
||||
const int numDofs = fe.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
MFEM_ASSERT(dims == mesh.SpaceDimension(), "");
|
||||
|
||||
geom.meshNodes.allocate(device,
|
||||
dims, numDofs, elements);
|
||||
|
||||
const mfem::Table &e2dTable = fespace.GetElementToDofTable();
|
||||
const int *elementMap = e2dTable.GetJ();
|
||||
nodes.Pull();
|
||||
for (int e = 0; e < elements; ++e)
|
||||
{
|
||||
for (int dof = 0; dof < numDofs; ++dof)
|
||||
{
|
||||
const int gid = elementMap[dof + numDofs*e];
|
||||
for (int dim = 0; dim < dims; ++dim)
|
||||
{
|
||||
geom.meshNodes(dim, dof, e) = nodes[fespace.DofToVDof(gid,dim)];
|
||||
}
|
||||
}
|
||||
}
|
||||
geom.meshNodes.keepInDevice();
|
||||
|
||||
if (flags & Jacobian)
|
||||
{
|
||||
geom.J.allocate(device,
|
||||
dims, dims, numQuad, elements);
|
||||
}
|
||||
else
|
||||
{
|
||||
geom.J.allocate(device, 1);
|
||||
}
|
||||
if (flags & JacobianInv)
|
||||
{
|
||||
geom.invJ.allocate(device,
|
||||
dims, dims, numQuad, elements);
|
||||
}
|
||||
else
|
||||
{
|
||||
geom.invJ.allocate(device, 1);
|
||||
}
|
||||
if (flags & JacobianDet)
|
||||
{
|
||||
geom.detJ.allocate(device,
|
||||
numQuad, elements);
|
||||
}
|
||||
else
|
||||
{
|
||||
geom.detJ.allocate(device, 1);
|
||||
}
|
||||
|
||||
geom.J.stopManaging();
|
||||
geom.invJ.stopManaging();
|
||||
geom.detJ.stopManaging();
|
||||
|
||||
OccaDofQuadMaps &maps = OccaDofQuadMaps::GetSimplexMaps(device, fe, ir);
|
||||
|
||||
::occa::properties props;
|
||||
props["defines/NUM_DOFS"] = numDofs;
|
||||
props["defines/NUM_QUAD"] = numQuad;
|
||||
props["defines/STORE_JACOBIAN"] = (flags & Jacobian);
|
||||
props["defines/STORE_JACOBIAN_INV"] = (flags & JacobianInv);
|
||||
props["defines/STORE_JACOBIAN_DET"] = (flags & JacobianDet);
|
||||
|
||||
const std::string &okl_path = ofespace.OccaEngine().GetOklPath();
|
||||
::occa::kernel init = device.buildKernel(okl_path + "geometry.okl",
|
||||
stringWithDim("InitGeometryInfo",
|
||||
fe.GetDim()),
|
||||
props);
|
||||
init(elements,
|
||||
maps.dofToQuadD,
|
||||
geom.meshNodes,
|
||||
geom.J, geom.invJ, geom.detJ);
|
||||
|
||||
return geom;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps::OccaDofQuadMaps() :
|
||||
hash() {}
|
||||
|
||||
OccaDofQuadMaps::OccaDofQuadMaps(const OccaDofQuadMaps &maps)
|
||||
{
|
||||
*this = maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::operator = (const OccaDofQuadMaps &maps)
|
||||
{
|
||||
hash = maps.hash;
|
||||
dofToQuad = maps.dofToQuad;
|
||||
dofToQuadD = maps.dofToQuadD;
|
||||
quadToDof = maps.quadToDof;
|
||||
quadToDofD = maps.quadToDofD;
|
||||
quadWeights = maps.quadWeights;
|
||||
return *this;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return Get(device,
|
||||
*fespace.GetFE(0),
|
||||
*fespace.GetFE(0),
|
||||
ir,
|
||||
transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return Get(device, fe, fe, ir, transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const FiniteElementSpace &trialFESpace,
|
||||
const FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return Get(device,
|
||||
*trialFESpace.GetFE(0),
|
||||
*testFESpace.GetFE(0),
|
||||
ir,
|
||||
transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return (dynamic_cast<const mfem::TensorBasisElement*>(&trialFE)
|
||||
? GetTensorMaps(device, trialFE, testFE, ir, transpose)
|
||||
: GetSimplexMaps(device, trialFE, testFE, ir, transpose));
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return GetTensorMaps(device,
|
||||
fe, fe,
|
||||
ir, transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
const mfem::TensorBasisElement &trialTFE =
|
||||
dynamic_cast<const mfem::TensorBasisElement&>(trialFE);
|
||||
const mfem::TensorBasisElement &testTFE =
|
||||
dynamic_cast<const mfem::TensorBasisElement&>(testFE);
|
||||
|
||||
std::stringstream ss;
|
||||
ss << ::occa::hash(device)
|
||||
<< "Tensor"
|
||||
<< "O1:" << trialFE.GetOrder()
|
||||
<< "O2:" << testFE.GetOrder()
|
||||
<< "BT1:" << trialTFE.GetBasisType()
|
||||
<< "BT2:" << testTFE.GetBasisType()
|
||||
<< "Q:" << ir.GetNPoints();
|
||||
std::string hash = ss.str();
|
||||
|
||||
// If we've already made the dof-quad maps, reuse them
|
||||
OccaDofQuadMaps &maps = AllDofQuadMaps[hash];
|
||||
if (!maps.hash.size())
|
||||
{
|
||||
// Create the dof-quad maps
|
||||
maps.hash = hash;
|
||||
|
||||
OccaDofQuadMaps trialMaps = GetD2QTensorMaps(device, trialFE, ir);
|
||||
OccaDofQuadMaps testMaps = GetD2QTensorMaps(device, testFE , ir, true);
|
||||
|
||||
maps.dofToQuad = trialMaps.dofToQuad;
|
||||
maps.dofToQuadD = trialMaps.dofToQuadD;
|
||||
maps.quadToDof = testMaps.dofToQuad;
|
||||
maps.quadToDofD = testMaps.dofToQuadD;
|
||||
maps.quadWeights = testMaps.quadWeights;
|
||||
}
|
||||
return maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps OccaDofQuadMaps::GetD2QTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
const mfem::TensorBasisElement &tfe =
|
||||
dynamic_cast<const mfem::TensorBasisElement&>(fe);
|
||||
|
||||
const mfem::Poly_1D::Basis &basis = tfe.GetBasis1D();
|
||||
const int order = fe.GetOrder();
|
||||
// [MISSING] Get 1D dofs
|
||||
const int dofs = order + 1;
|
||||
const int dims = fe.GetDim();
|
||||
|
||||
// Create the dof -> quadrature point map
|
||||
const mfem::IntegrationRule &ir1D =
|
||||
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir.GetOrder());
|
||||
const int quadPoints = ir1D.GetNPoints();
|
||||
const int quadPoints2D = quadPoints*quadPoints;
|
||||
const int quadPoints3D = quadPoints2D*quadPoints;
|
||||
const int quadPointsND = ((dims == 1) ? quadPoints :
|
||||
((dims == 2) ? quadPoints2D : quadPoints3D));
|
||||
|
||||
OccaDofQuadMaps maps;
|
||||
// Initialize the dof -> quad mapping
|
||||
maps.dofToQuad.allocate(device,
|
||||
quadPoints, dofs);
|
||||
maps.dofToQuadD.allocate(device,
|
||||
quadPoints, dofs);
|
||||
|
||||
double *quadWeights1DData = NULL;
|
||||
|
||||
if (transpose)
|
||||
{
|
||||
maps.dofToQuad.reindex(1,0);
|
||||
maps.dofToQuadD.reindex(1,0);
|
||||
// Initialize quad weights only for transpose
|
||||
maps.quadWeights.allocate(device,
|
||||
quadPointsND);
|
||||
quadWeights1DData = new double[quadPoints];
|
||||
}
|
||||
|
||||
mfem::Vector d2q(dofs);
|
||||
mfem::Vector d2qD(dofs);
|
||||
for (int q = 0; q < quadPoints; ++q)
|
||||
{
|
||||
const mfem::IntegrationPoint &ip = ir1D.IntPoint(q);
|
||||
basis.Eval(ip.x, d2q, d2qD);
|
||||
if (transpose)
|
||||
{
|
||||
quadWeights1DData[q] = ip.weight;
|
||||
}
|
||||
for (int d = 0; d < dofs; ++d)
|
||||
{
|
||||
maps.dofToQuad(q, d) = d2q[d];
|
||||
maps.dofToQuadD(q, d) = d2qD[d];
|
||||
}
|
||||
}
|
||||
|
||||
maps.dofToQuad.keepInDevice();
|
||||
maps.dofToQuadD.keepInDevice();
|
||||
|
||||
if (transpose)
|
||||
{
|
||||
for (int q = 0; q < quadPointsND; ++q)
|
||||
{
|
||||
const int qx = q % quadPoints;
|
||||
const int qz = q / quadPoints2D;
|
||||
const int qy = (q - qz*quadPoints2D) / quadPoints;
|
||||
double w = quadWeights1DData[qx];
|
||||
if (dims > 1)
|
||||
{
|
||||
w *= quadWeights1DData[qy];
|
||||
}
|
||||
if (dims > 2)
|
||||
{
|
||||
w *= quadWeights1DData[qz];
|
||||
}
|
||||
maps.quadWeights[q] = w;
|
||||
}
|
||||
maps.quadWeights.keepInDevice();
|
||||
delete [] quadWeights1DData;
|
||||
}
|
||||
|
||||
return maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return GetSimplexMaps(device,
|
||||
fe, fe,
|
||||
ir, transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
std::stringstream ss;
|
||||
ss << ::occa::hash(device)
|
||||
<< "Simplex"
|
||||
<< "O1:" << trialFE.GetOrder()
|
||||
<< "O2:" << testFE.GetOrder()
|
||||
<< "Q:" << ir.GetNPoints();
|
||||
std::string hash = ss.str();
|
||||
|
||||
// If we've already made the dof-quad maps, reuse them
|
||||
OccaDofQuadMaps &maps = AllDofQuadMaps[hash];
|
||||
if (!maps.hash.size())
|
||||
{
|
||||
// Create the dof-quad maps
|
||||
maps.hash = hash;
|
||||
|
||||
OccaDofQuadMaps trialMaps = GetD2QSimplexMaps(device, trialFE, ir);
|
||||
OccaDofQuadMaps testMaps = GetD2QSimplexMaps(device, testFE , ir, true);
|
||||
|
||||
maps.dofToQuad = trialMaps.dofToQuad;
|
||||
maps.dofToQuadD = trialMaps.dofToQuadD;
|
||||
maps.quadToDof = testMaps.dofToQuad;
|
||||
maps.quadToDofD = testMaps.dofToQuadD;
|
||||
maps.quadWeights = testMaps.quadWeights;
|
||||
}
|
||||
return maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps OccaDofQuadMaps::GetD2QSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
const int dims = fe.GetDim();
|
||||
const int numDofs = fe.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
OccaDofQuadMaps maps;
|
||||
// Initialize the dof -> quad mapping
|
||||
maps.dofToQuad.allocate(device,
|
||||
numQuad, numDofs);
|
||||
maps.dofToQuadD.allocate(device,
|
||||
dims, numQuad, numDofs);
|
||||
|
||||
if (transpose)
|
||||
{
|
||||
maps.dofToQuad.reindex(1,0);
|
||||
maps.dofToQuadD.reindex(1,0);
|
||||
// Initialize quad weights only for transpose
|
||||
maps.quadWeights.allocate(device,
|
||||
numQuad);
|
||||
}
|
||||
|
||||
mfem::Vector d2q(numDofs);
|
||||
mfem::DenseMatrix d2qD(numDofs, dims);
|
||||
for (int q = 0; q < numQuad; ++q)
|
||||
{
|
||||
const mfem::IntegrationPoint &ip = ir.IntPoint(q);
|
||||
if (transpose)
|
||||
{
|
||||
maps.quadWeights[q] = ip.weight;
|
||||
}
|
||||
fe.CalcShape(ip, d2q);
|
||||
fe.CalcDShape(ip, d2qD);
|
||||
for (int d = 0; d < numDofs; ++d)
|
||||
{
|
||||
const double w = d2q[d];
|
||||
maps.dofToQuad(q, d) = w;
|
||||
for (int dim = 0; dim < dims; ++dim)
|
||||
{
|
||||
const double wD = d2qD(d, dim);
|
||||
maps.dofToQuadD(dim, q, d) = wD;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
maps.dofToQuad.keepInDevice();
|
||||
maps.dofToQuadD.keepInDevice();
|
||||
if (transpose)
|
||||
{
|
||||
maps.quadWeights.keepInDevice();
|
||||
}
|
||||
|
||||
return maps;
|
||||
}
|
||||
|
||||
//---[ Integrator Defines ]-----------
|
||||
std::string stringWithDim(const std::string &s, const int dim)
|
||||
{
|
||||
std::string ret = s;
|
||||
ret += ('0' + (char) dim);
|
||||
ret += 'D';
|
||||
return ret;
|
||||
}
|
||||
|
||||
int closestWarpBatchTo(const int value)
|
||||
{
|
||||
return ((value + 31) / 32) * 32;
|
||||
}
|
||||
|
||||
int closestMultipleWarpBatch(const int multiple, const int maxSize)
|
||||
{
|
||||
if (multiple > maxSize)
|
||||
{
|
||||
return maxSize;
|
||||
}
|
||||
int batch = (32 / multiple);
|
||||
int minDiff = 32 - (multiple * batch);
|
||||
for (int i = 64; i <= maxSize; i += 32)
|
||||
{
|
||||
const int newDiff = i - (multiple * (i / multiple));
|
||||
if (newDiff < minDiff)
|
||||
{
|
||||
batch = (i / multiple);
|
||||
minDiff = newDiff;
|
||||
}
|
||||
}
|
||||
return batch;
|
||||
}
|
||||
|
||||
void SetProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
SetProperties(fespace, fespace, ir, props);
|
||||
}
|
||||
|
||||
void SetProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["defines/TRIAL_VDIM"] = trialFESpace.GetVDim();
|
||||
props["defines/TEST_VDIM"] = testFESpace.GetVDim();
|
||||
props["defines/NUM_DIM"] = trialFESpace.GetDim();
|
||||
|
||||
if (trialFESpace.hasTensorBasis())
|
||||
{
|
||||
SetTensorProperties(trialFESpace, testFESpace, ir, props);
|
||||
}
|
||||
else
|
||||
{
|
||||
SetSimplexProperties(trialFESpace, testFESpace, ir, props);
|
||||
}
|
||||
}
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
SetTensorProperties(fespace, fespace, ir, props);
|
||||
}
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace.GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace.GetFE(0));
|
||||
|
||||
const mfem::IntegrationRule &ir1D =
|
||||
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir.GetOrder());
|
||||
|
||||
const int trialDofs = trialFE.GetDof();
|
||||
const int testDofs = testFE.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
const int trialDofs1D = trialFE.GetOrder() + 1;
|
||||
const int testDofs1D = testFE.GetOrder() + 1;
|
||||
const int quad1D = ir1D.GetNPoints();
|
||||
int trialDofsND = trialDofs1D;
|
||||
int testDofsND = testDofs1D;
|
||||
int quadND = quad1D;
|
||||
|
||||
const bool trialByVDIM = (trialFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
const bool testByVDIM = (testFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
|
||||
props["defines/ORDERING_BY_NODES"] = 0;
|
||||
props["defines/ORDERING_BY_VDIM"] = 1;
|
||||
props["defines/VDIM_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TRIAL_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TEST_ORDERING"] = (int) testByVDIM;
|
||||
|
||||
props["defines/USING_TENSOR_OPS"] = 1;
|
||||
props["defines/NUM_DOFS"] = trialDofs;
|
||||
props["defines/NUM_QUAD"] = numQuad;
|
||||
|
||||
props["defines/TRIAL_DOFS"] = trialDofs;
|
||||
props["defines/TEST_DOFS"] = testDofs;
|
||||
|
||||
for (int d = 1; d <= 3; ++d)
|
||||
{
|
||||
if (d > 1)
|
||||
{
|
||||
trialDofsND *= trialDofs1D;
|
||||
testDofsND *= testDofs1D;
|
||||
quadND *= quad1D;
|
||||
}
|
||||
props["defines"][stringWithDim("NUM_DOFS_", d)] = trialDofsND;
|
||||
props["defines"][stringWithDim("NUM_QUAD_", d)] = quadND;
|
||||
|
||||
props["defines"][stringWithDim("TRIAL_DOFS_", d)] = trialDofsND;
|
||||
props["defines"][stringWithDim("TEST_DOFS_" , d)] = testDofsND;
|
||||
}
|
||||
|
||||
// 1D Defines
|
||||
const int m1InnerBatch = 32 * ((quad1D + 31) / 32);
|
||||
props["defines/A1_ELEMENT_BATCH"] = closestMultipleWarpBatch(quad1D, 512);
|
||||
props["defines/M1_OUTER_ELEMENT_BATCH"] = closestMultipleWarpBatch(m1InnerBatch,
|
||||
512);
|
||||
props["defines/M1_INNER_ELEMENT_BATCH"] = m1InnerBatch;
|
||||
|
||||
// 2D Defines
|
||||
props["defines/A2_ELEMENT_BATCH"] = 1;
|
||||
props["defines/A2_QUAD_BATCH"] = 1;
|
||||
props["defines/M2_ELEMENT_BATCH"] = 32;
|
||||
|
||||
// 3D Defines
|
||||
const int a3QuadBatch = closestMultipleWarpBatch(quadND, 512);
|
||||
props["defines/A3_ELEMENT_BATCH"] = closestMultipleWarpBatch(a3QuadBatch, 512);
|
||||
props["defines/A3_QUAD_BATCH"] = a3QuadBatch;
|
||||
}
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
SetSimplexProperties(fespace, fespace, ir, props);
|
||||
}
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace.GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace.GetFE(0));
|
||||
|
||||
const int trialDofs = trialFE.GetDof();
|
||||
const int testDofs = testFE.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
const int maxDQ = std::max(std::max(trialDofs, testDofs), numQuad);
|
||||
|
||||
const bool trialByVDIM = (trialFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
const bool testByVDIM = (testFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
|
||||
props["defines/ORDERING_BY_NODES"] = 0;
|
||||
props["defines/ORDERING_BY_VDIM"] = 1;
|
||||
props["defines/VDIM_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TRIAL_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TEST_ORDERING"] = (int) testByVDIM;
|
||||
|
||||
props["defines/USING_TENSOR_OPS"] = 0;
|
||||
props["defines/NUM_DOFS"] = trialDofs;
|
||||
props["defines/NUM_QUAD"] = numQuad;
|
||||
|
||||
props["defines/TRIAL_DOFS"] = trialDofs;
|
||||
props["defines/TEST_DOFS"] = testDofs;
|
||||
|
||||
// 2D Defines
|
||||
const int quadBatch = closestWarpBatchTo(numQuad);
|
||||
props["defines/A2_ELEMENT_BATCH"] = closestMultipleWarpBatch(quadBatch, 2048);
|
||||
props["defines/A2_QUAD_BATCH"] = quadBatch;
|
||||
props["defines/M2_INNER_BATCH"] = closestWarpBatchTo(maxDQ);
|
||||
|
||||
// 3D Defines
|
||||
props["defines/A3_ELEMENT_BATCH"] = closestMultipleWarpBatch(quadBatch, 2048);
|
||||
props["defines/A3_QUAD_BATCH"] = quadBatch;
|
||||
props["defines/M3_INNER_BATCH"] = closestWarpBatchTo(maxDQ);
|
||||
}
|
||||
|
||||
|
||||
//---[ Base Integrator ]--------------
|
||||
OccaIntegrator::OccaIntegrator(const Engine &e)
|
||||
: engine(&e),
|
||||
bform(),
|
||||
mesh(),
|
||||
otrialFESpace(),
|
||||
otestFESpace(),
|
||||
trialFESpace(),
|
||||
testFESpace(),
|
||||
itype(DomainIntegrator),
|
||||
ir(NULL),
|
||||
hasTensorBasis(false) { }
|
||||
|
||||
OccaIntegrator::~OccaIntegrator() {}
|
||||
|
||||
void OccaIntegrator::SetupMaps()
|
||||
{
|
||||
maps = OccaDofQuadMaps::Get(GetDevice(),
|
||||
*otrialFESpace,
|
||||
*otestFESpace,
|
||||
*ir);
|
||||
|
||||
mapsTranspose = OccaDofQuadMaps::Get(GetDevice(),
|
||||
*otestFESpace,
|
||||
*otrialFESpace,
|
||||
*ir);
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaIntegrator::GetTrialOccaFESpace() const
|
||||
{
|
||||
return *otrialFESpace;
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaIntegrator::GetTestOccaFESpace() const
|
||||
{
|
||||
return *otestFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaIntegrator::GetTrialFESpace() const
|
||||
{
|
||||
return *trialFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaIntegrator::GetTestFESpace() const
|
||||
{
|
||||
return *testFESpace;
|
||||
}
|
||||
|
||||
void OccaIntegrator::SetIntegrationRule(const mfem::IntegrationRule &ir_)
|
||||
{
|
||||
ir = &ir_;
|
||||
}
|
||||
|
||||
const mfem::IntegrationRule& OccaIntegrator::GetIntegrationRule() const
|
||||
{
|
||||
return *ir;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaIntegrator::GetDofQuadMaps()
|
||||
{
|
||||
return maps;
|
||||
}
|
||||
|
||||
void OccaIntegrator::SetupIntegrator(OccaBilinearForm &bform_,
|
||||
const ::occa::properties &props_,
|
||||
const OccaIntegratorType itype_)
|
||||
{
|
||||
MFEM_ASSERT(engine == &bform_.OccaEngine(), "");
|
||||
bform = &bform_;
|
||||
mesh = &(bform_.GetMesh());
|
||||
|
||||
otrialFESpace = &(bform_.GetTrialOccaFESpace());
|
||||
otestFESpace = &(bform_.GetTestOccaFESpace());
|
||||
|
||||
trialFESpace = &(bform_.GetTrialFESpace());
|
||||
testFESpace = &(bform_.GetTestFESpace());
|
||||
|
||||
hasTensorBasis = otrialFESpace->hasTensorBasis();
|
||||
|
||||
props = props_;
|
||||
itype = itype_;
|
||||
|
||||
if (ir == NULL)
|
||||
{
|
||||
SetupIntegrationRule();
|
||||
}
|
||||
|
||||
SetupMaps();
|
||||
|
||||
SetProperties(*otrialFESpace,
|
||||
*otestFESpace,
|
||||
*ir,
|
||||
props);
|
||||
|
||||
Setup();
|
||||
}
|
||||
|
||||
OccaGeometry OccaIntegrator::GetGeometry(const int flags)
|
||||
{
|
||||
return OccaGeometry::Get(GetDevice(), *otrialFESpace, *ir, flags);
|
||||
}
|
||||
|
||||
::occa::kernel OccaIntegrator::GetAssembleKernel(const ::occa::properties
|
||||
&props)
|
||||
{
|
||||
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
|
||||
return GetKernel(stringWithDim("Assemble", fe.GetDim()),
|
||||
props);
|
||||
}
|
||||
|
||||
::occa::kernel OccaIntegrator::GetMultAddKernel(const ::occa::properties &props)
|
||||
{
|
||||
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
|
||||
return GetKernel(stringWithDim("MultAdd", fe.GetDim()),
|
||||
props);
|
||||
}
|
||||
|
||||
::occa::kernel OccaIntegrator::GetKernel(const std::string &kernelName,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
const std::string filename = GetName() + ".okl";
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
return GetDevice().buildKernel(okl_path + filename,
|
||||
kernelName,
|
||||
props);
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Diffusion Integrator ]---------
|
||||
OccaDiffusionIntegrator::OccaDiffusionIntegrator(const OccaCoefficient &coeff_)
|
||||
:
|
||||
OccaIntegrator(coeff_.OccaEngine()),
|
||||
coeff(coeff_),
|
||||
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
|
||||
{
|
||||
coeff.SetName("COEFF");
|
||||
}
|
||||
|
||||
OccaDiffusionIntegrator::~OccaDiffusionIntegrator() {}
|
||||
|
||||
|
||||
std::string OccaDiffusionIntegrator::GetName()
|
||||
{
|
||||
return "DiffusionIntegrator";
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::SetupIntegrationRule()
|
||||
{
|
||||
const FiniteElement &trialFE = *(trialFESpace->GetFE(0));
|
||||
const FiniteElement &testFE = *(testFESpace->GetFE(0));
|
||||
ir = &mfem::DiffusionIntegrator::GetRule(trialFE, testFE);
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::Setup()
|
||||
{
|
||||
::occa::properties kernelProps = props;
|
||||
|
||||
coeff.Setup(*this, kernelProps);
|
||||
|
||||
// Setup assemble and mult kernels
|
||||
assembleKernel = GetAssembleKernel(kernelProps);
|
||||
multKernel = GetMultAddKernel(kernelProps);
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::Assemble()
|
||||
{
|
||||
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
|
||||
|
||||
const int dims = fe.GetDim();
|
||||
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
|
||||
|
||||
const int elements = trialFESpace->GetNE();
|
||||
const int quadraturePoints = ir->GetNPoints();
|
||||
|
||||
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
|
||||
|
||||
assembledOperator.OccaResize(symmDims * quadraturePoints * elements,
|
||||
sizeof(double));
|
||||
|
||||
assembleKernel((int) mesh->GetNE(),
|
||||
maps.quadWeights,
|
||||
geom.J,
|
||||
coeff,
|
||||
assembledOperator.OccaMem());
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::MultAdd(Vector &x, Vector &y)
|
||||
{
|
||||
// Note: x and y are E-vectors
|
||||
|
||||
multKernel((int) mesh->GetNE(),
|
||||
maps.dofToQuad,
|
||||
maps.dofToQuadD,
|
||||
maps.quadToDof,
|
||||
maps.quadToDofD,
|
||||
assembledOperator.OccaMem(),
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Mass Integrator ]--------------
|
||||
OccaMassIntegrator::OccaMassIntegrator(const OccaCoefficient &coeff_) :
|
||||
OccaIntegrator(coeff_.OccaEngine()),
|
||||
coeff(coeff_),
|
||||
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
|
||||
{
|
||||
coeff.SetName("COEFF");
|
||||
}
|
||||
|
||||
OccaMassIntegrator::~OccaMassIntegrator() {}
|
||||
|
||||
std::string OccaMassIntegrator::GetName()
|
||||
{
|
||||
return "MassIntegrator";
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::SetupIntegrationRule()
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace->GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace->GetFE(0));
|
||||
mfem::ElementTransformation &T = *trialFESpace->GetElementTransformation(0);
|
||||
ir = &mfem::MassIntegrator::GetRule(trialFE, testFE, T);
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::Setup()
|
||||
{
|
||||
::occa::properties kernelProps = props;
|
||||
|
||||
coeff.Setup(*this, kernelProps);
|
||||
|
||||
// Setup assemble and mult kernels
|
||||
assembleKernel = GetAssembleKernel(kernelProps);
|
||||
multKernel = GetMultAddKernel(kernelProps);
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::Assemble()
|
||||
{
|
||||
if (assembledOperator.Size())
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
const int elements = trialFESpace->GetNE();
|
||||
const int quadraturePoints = ir->GetNPoints();
|
||||
|
||||
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
|
||||
|
||||
assembledOperator.Resize<double>(quadraturePoints * elements, NULL);
|
||||
|
||||
assembleKernel((int) mesh->GetNE(),
|
||||
maps.quadWeights,
|
||||
geom.J,
|
||||
coeff,
|
||||
assembledOperator.OccaMem());
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::SetOperator(Vector &v)
|
||||
{
|
||||
assembledOperator = v;
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::MultAdd(Vector &x, Vector &y)
|
||||
{
|
||||
multKernel((int) mesh->GetNE(),
|
||||
maps.dofToQuad,
|
||||
maps.dofToQuadD,
|
||||
maps.quadToDof,
|
||||
maps.quadToDofD,
|
||||
assembledOperator.OccaMem(),
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Vector Mass Integrator ]--------------
|
||||
OccaVectorMassIntegrator::OccaVectorMassIntegrator(const OccaCoefficient &
|
||||
coeff_)
|
||||
:
|
||||
OccaIntegrator(coeff_.OccaEngine()),
|
||||
coeff(coeff_),
|
||||
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
|
||||
{
|
||||
coeff.SetName("COEFF");
|
||||
}
|
||||
|
||||
OccaVectorMassIntegrator::~OccaVectorMassIntegrator() {}
|
||||
|
||||
std::string OccaVectorMassIntegrator::GetName()
|
||||
{
|
||||
return "VectorMassIntegrator";
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::SetupIntegrationRule()
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace->GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace->GetFE(0));
|
||||
mfem::ElementTransformation &T = *trialFESpace->GetElementTransformation(0);
|
||||
ir = &mfem::MassIntegrator::GetRule(trialFE, testFE, T);
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::Setup()
|
||||
{
|
||||
::occa::properties kernelProps = props;
|
||||
|
||||
coeff.Setup(*this, kernelProps);
|
||||
|
||||
// Setup assemble and mult kernels
|
||||
assembleKernel = GetAssembleKernel(kernelProps);
|
||||
multKernel = GetMultAddKernel(kernelProps);
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::Assemble()
|
||||
{
|
||||
const int elements = trialFESpace->GetNE();
|
||||
const int quadraturePoints = ir->GetNPoints();
|
||||
|
||||
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
|
||||
|
||||
assembledOperator.Resize<double>(quadraturePoints * elements, NULL);
|
||||
|
||||
assembleKernel((int) mesh->GetNE(),
|
||||
maps.quadWeights,
|
||||
geom.J,
|
||||
coeff,
|
||||
assembledOperator.OccaMem());
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::MultAdd(Vector &x, Vector &y)
|
||||
{
|
||||
multKernel((int) mesh->GetNE(),
|
||||
maps.dofToQuad,
|
||||
maps.dofToQuadD,
|
||||
maps.quadToDof,
|
||||
maps.quadToDofD,
|
||||
assembledOperator.OccaMem(),
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,323 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
|
||||
#define MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "coefficient.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class OccaGeometry
|
||||
{
|
||||
public:
|
||||
::occa::array<double> meshNodes;
|
||||
::occa::array<double> J, invJ, detJ;
|
||||
|
||||
// byVDIM -> [x y z x y z x y z]
|
||||
// byNodes -> [x x x y y y z z z]
|
||||
static const int Jacobian = (1 << 0);
|
||||
static const int JacobianInv = (1 << 1);
|
||||
static const int JacobianDet = (1 << 2);
|
||||
|
||||
static OccaGeometry Get(::occa::device device,
|
||||
FiniteElementSpace &ofespace,
|
||||
const IntegrationRule &ir,
|
||||
const int flags = (Jacobian |
|
||||
JacobianInv |
|
||||
JacobianDet));
|
||||
};
|
||||
|
||||
class OccaDofQuadMaps
|
||||
{
|
||||
private:
|
||||
// Reuse dof-quad maps
|
||||
static std::map<std::string, OccaDofQuadMaps> AllDofQuadMaps;
|
||||
std::string hash;
|
||||
|
||||
public:
|
||||
// Local stiffness matrices (B and B^T operators)
|
||||
::occa::array<double, ::occa::dynamic> dofToQuad, dofToQuadD; // B
|
||||
::occa::array<double, ::occa::dynamic> quadToDof, quadToDofD; // B^T
|
||||
::occa::array<double> quadWeights;
|
||||
|
||||
OccaDofQuadMaps();
|
||||
OccaDofQuadMaps(const OccaDofQuadMaps &maps);
|
||||
OccaDofQuadMaps& operator = (const OccaDofQuadMaps &maps);
|
||||
|
||||
// [[x y] [x y] [x y]]
|
||||
// [[x y z] [x y z] [x y z]]
|
||||
// mfem::GridFunction* mfem::Mesh::GetNodes() { return Nodes; }
|
||||
|
||||
// mfem::FiniteElementSpace *Nodes->FESpace()
|
||||
// 25
|
||||
// 1D [x x x x x x]
|
||||
// 2D [x y x y x y]
|
||||
// GetVdim()
|
||||
// 3D ordering == byVDIM -> [x y z x y z x y z x y z x y z x y z]
|
||||
// ordering == byNODES -> [x x x x x x y y y y y y z z z z z z]
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const FiniteElementSpace &trialFESpace,
|
||||
const FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps GetD2QTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps GetD2QSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
};
|
||||
|
||||
//---[ Define Methods ]---------------
|
||||
std::string stringWithDim(const std::string &s, const int dim);
|
||||
int closestWarpBatch(const int multiple, const int maxSize);
|
||||
|
||||
void SetProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &fespace,
|
||||
const IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
//---[ Base Integrator ]--------------
|
||||
class OccaIntegrator
|
||||
{
|
||||
protected:
|
||||
SharedPtr<const Engine> engine;
|
||||
|
||||
OccaBilinearForm *bform;
|
||||
mfem::Mesh *mesh;
|
||||
|
||||
FiniteElementSpace *otrialFESpace;
|
||||
FiniteElementSpace *otestFESpace;
|
||||
|
||||
mfem::FiniteElementSpace *trialFESpace;
|
||||
mfem::FiniteElementSpace *testFESpace;
|
||||
|
||||
::occa::properties props;
|
||||
OccaIntegratorType itype;
|
||||
|
||||
const IntegrationRule *ir;
|
||||
bool hasTensorBasis;
|
||||
OccaDofQuadMaps maps;
|
||||
OccaDofQuadMaps mapsTranspose;
|
||||
|
||||
public:
|
||||
OccaIntegrator(const Engine &e);
|
||||
virtual ~OccaIntegrator();
|
||||
|
||||
const Engine &OccaEngine() const { return *engine; }
|
||||
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return engine->GetDevice(idx); }
|
||||
|
||||
virtual std::string GetName() = 0;
|
||||
|
||||
FiniteElementSpace& GetTrialOccaFESpace() const;
|
||||
FiniteElementSpace& GetTestOccaFESpace() const;
|
||||
|
||||
mfem::FiniteElementSpace& GetTrialFESpace() const;
|
||||
mfem::FiniteElementSpace& GetTestFESpace() const;
|
||||
|
||||
void SetIntegrationRule(const mfem::IntegrationRule &ir_);
|
||||
const mfem::IntegrationRule& GetIntegrationRule() const;
|
||||
|
||||
OccaDofQuadMaps& GetDofQuadMaps();
|
||||
|
||||
void SetupMaps();
|
||||
|
||||
virtual void SetupIntegrationRule() = 0;
|
||||
|
||||
virtual void SetupIntegrator(OccaBilinearForm &bform_,
|
||||
const ::occa::properties &props_,
|
||||
const OccaIntegratorType itype_);
|
||||
|
||||
virtual void Setup() = 0;
|
||||
|
||||
virtual void Assemble() = 0;
|
||||
/// This method works on E-vectors!
|
||||
virtual void MultAdd(Vector &x, Vector &y) = 0;
|
||||
|
||||
virtual void MultTransposeAdd(Vector &x, Vector &y)
|
||||
{
|
||||
mfem_error("OccaIntegrator::MultTransposeAdd() is not overloaded!");
|
||||
}
|
||||
|
||||
OccaGeometry GetGeometry(const int flags = (OccaGeometry::Jacobian |
|
||||
OccaGeometry::JacobianInv |
|
||||
OccaGeometry::JacobianDet));
|
||||
|
||||
::occa::kernel GetAssembleKernel(const ::occa::properties &props);
|
||||
::occa::kernel GetMultAddKernel(const ::occa::properties &props);
|
||||
|
||||
::occa::kernel GetKernel(const std::string &kernelName,
|
||||
const ::occa::properties &props);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Diffusion Integrator ]---------
|
||||
class OccaDiffusionIntegrator : public OccaIntegrator
|
||||
{
|
||||
private:
|
||||
OccaCoefficient coeff;
|
||||
|
||||
::occa::kernel assembleKernel, multKernel;
|
||||
|
||||
Vector assembledOperator;
|
||||
|
||||
public:
|
||||
OccaDiffusionIntegrator(const OccaCoefficient &coeff_);
|
||||
virtual ~OccaDiffusionIntegrator();
|
||||
|
||||
virtual std::string GetName();
|
||||
|
||||
virtual void SetupIntegrationRule();
|
||||
|
||||
virtual void Setup();
|
||||
|
||||
virtual void Assemble();
|
||||
virtual void MultAdd(Vector &x, Vector &y);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Mass Integrator ]--------------
|
||||
class OccaMassIntegrator : public OccaIntegrator
|
||||
{
|
||||
private:
|
||||
OccaCoefficient coeff;
|
||||
|
||||
::occa::kernel assembleKernel, multKernel;
|
||||
|
||||
Vector assembledOperator;
|
||||
|
||||
public:
|
||||
OccaMassIntegrator(const OccaCoefficient &coeff_);
|
||||
virtual ~OccaMassIntegrator();
|
||||
|
||||
virtual std::string GetName();
|
||||
|
||||
virtual void SetupIntegrationRule();
|
||||
|
||||
virtual void Setup();
|
||||
|
||||
virtual void Assemble();
|
||||
void SetOperator(Vector &v);
|
||||
|
||||
virtual void MultAdd(Vector &x, Vector &y);
|
||||
};
|
||||
//====================================
|
||||
|
||||
//---[ Vector Mass Integrator ]--------------
|
||||
class OccaVectorMassIntegrator : public OccaIntegrator
|
||||
{
|
||||
private:
|
||||
OccaCoefficient coeff;
|
||||
|
||||
::occa::kernel assembleKernel, multKernel;
|
||||
|
||||
Vector assembledOperator;
|
||||
|
||||
public:
|
||||
OccaVectorMassIntegrator(const OccaCoefficient &coeff_);
|
||||
virtual ~OccaVectorMassIntegrator();
|
||||
|
||||
virtual std::string GetName();
|
||||
|
||||
virtual void SetupIntegrationRule();
|
||||
|
||||
virtual void Setup();
|
||||
|
||||
virtual void Assemble();
|
||||
|
||||
virtual void MultAdd(Vector &x, Vector &y);
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
|
||||
@@ -1,357 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "coefficient.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
//---[ Parameter ]------------
|
||||
OccaParameter::~OccaParameter() {}
|
||||
|
||||
void OccaParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props) {}
|
||||
|
||||
::occa::kernelArg OccaParameter::KernelArgs()
|
||||
{
|
||||
return ::occa::kernelArg();
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Include Parameter ]------------
|
||||
OccaIncludeParameter::OccaIncludeParameter(const std::string &filename_) :
|
||||
filename(filename_) {}
|
||||
|
||||
OccaParameter* OccaIncludeParameter::Clone()
|
||||
{
|
||||
return new OccaIncludeParameter(filename);
|
||||
}
|
||||
|
||||
void OccaIncludeParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["headers"].asArray() += "#include " + filename;
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Source Parameter ]------------
|
||||
OccaSourceParameter::OccaSourceParameter(const std::string &source_) :
|
||||
source(source_) {}
|
||||
|
||||
OccaParameter* OccaSourceParameter::Clone()
|
||||
{
|
||||
return new OccaSourceParameter(source);
|
||||
}
|
||||
|
||||
void OccaSourceParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["headers"].asArray() += source;
|
||||
}
|
||||
//====================================
|
||||
|
||||
//---[ Vector Parameter ]-------
|
||||
OccaVectorParameter::OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const bool useRestrict_) :
|
||||
name(name_),
|
||||
v(v_),
|
||||
useRestrict(useRestrict_),
|
||||
attr("") {}
|
||||
|
||||
OccaVectorParameter::OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const std::string &attr_,
|
||||
const bool useRestrict_) :
|
||||
name(name_),
|
||||
v(v_),
|
||||
useRestrict(useRestrict_),
|
||||
attr(attr_) {}
|
||||
|
||||
OccaParameter* OccaVectorParameter::Clone()
|
||||
{
|
||||
return new OccaVectorParameter(name, v, attr, useRestrict);
|
||||
}
|
||||
|
||||
void OccaVectorParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
std::string &args = (props["defines/COEFF_ARGS"]
|
||||
.asString()
|
||||
.string());
|
||||
args += "const double *";
|
||||
if (useRestrict)
|
||||
{
|
||||
args += " restrict ";
|
||||
}
|
||||
args += name;
|
||||
if (attr.size())
|
||||
{
|
||||
args += ' ';
|
||||
args += attr;
|
||||
}
|
||||
args += ",\n";
|
||||
}
|
||||
|
||||
::occa::kernelArg OccaVectorParameter::KernelArgs()
|
||||
{
|
||||
return ::occa::kernelArg(v.OccaMem());
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ GridFunction Parameter ]-------
|
||||
OccaGridFunctionParameter::OccaGridFunctionParameter(const std::string &name_,
|
||||
const Engine &e,
|
||||
mfem::GridFunction &gf_,
|
||||
const bool useRestrict_)
|
||||
: name(name_),
|
||||
gf(gf_),
|
||||
gfQuad(e),
|
||||
useRestrict(useRestrict_) {}
|
||||
|
||||
OccaParameter* OccaGridFunctionParameter::Clone()
|
||||
{
|
||||
OccaGridFunctionParameter *param =
|
||||
new OccaGridFunctionParameter(name, gfQuad.OccaEngine(), gf, useRestrict);
|
||||
param->gfQuad.MakeRef(gfQuad);
|
||||
return param;
|
||||
}
|
||||
|
||||
void OccaGridFunctionParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
|
||||
std::string &args = (props["defines/COEFF_ARGS"]
|
||||
.asString()
|
||||
.string());
|
||||
if (useRestrict)
|
||||
{
|
||||
args += "@restrict ";
|
||||
}
|
||||
args += "const double *";
|
||||
args += name;
|
||||
args += " @dim(NUM_QUAD, numElements),\n";
|
||||
|
||||
FiniteElementSpace &f = gf.FESpace()->Get_PFESpace()->As<FiniteElementSpace>();
|
||||
ToQuad(integ.GetIntegrationRule(), f, gf.Get_PVector()->As<Vector>(), gfQuad);
|
||||
}
|
||||
|
||||
::occa::kernelArg OccaGridFunctionParameter::KernelArgs()
|
||||
{
|
||||
return gfQuad.OccaMem();
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Coefficient ]------------------
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, const double value) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = value;
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, mfem::GridFunction &gf,
|
||||
const bool useRestrict) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = "(u(q, e))";
|
||||
AddGridFunction("u", gf, useRestrict);
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, const std::string &source) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = source;
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, const char *source) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = source;
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const OccaCoefficient &coeff) :
|
||||
engine(coeff.engine),
|
||||
integ(NULL),
|
||||
name(coeff.name),
|
||||
coeffValue(coeff.coeffValue)
|
||||
{
|
||||
|
||||
const int paramCount = (int) coeff.params.size();
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
params.push_back(coeff.params[i]->Clone());
|
||||
}
|
||||
}
|
||||
|
||||
OccaCoefficient::~OccaCoefficient()
|
||||
{
|
||||
const int paramCount = (int) params.size();
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
delete params[i];
|
||||
}
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::SetName(const std::string &name_)
|
||||
{
|
||||
name = name_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
void OccaCoefficient::Setup(OccaIntegrator &integ_,
|
||||
::occa::properties &props_)
|
||||
{
|
||||
integ = &integ_;
|
||||
|
||||
const int paramCount = (int) params.size();
|
||||
props_["defines"][name + "_ARGS"] = "";
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
params[i]->Setup(integ_, props_);
|
||||
}
|
||||
props_["defines"][name] = coeffValue;
|
||||
|
||||
props = props_;
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::Add(OccaParameter *param)
|
||||
{
|
||||
params.push_back(param);
|
||||
return *this;
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::IncludeHeader(const std::string &filename)
|
||||
{
|
||||
return Add(new OccaIncludeParameter(filename));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::IncludeSource(const std::string &source)
|
||||
{
|
||||
return Add(new OccaSourceParameter(source));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const bool useRestrict)
|
||||
{
|
||||
return Add(new OccaVectorParameter(name_, v, useRestrict));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const std::string &attr,
|
||||
const bool useRestrict)
|
||||
{
|
||||
return Add(new OccaVectorParameter(name_, v, attr, useRestrict));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::AddGridFunction(const std::string &name_,
|
||||
mfem::GridFunction &gf,
|
||||
const bool useRestrict)
|
||||
{
|
||||
MFEM_ASSERT(engine->CheckVector(gf.Get_PVector()) &&
|
||||
engine->CheckFESpace(gf.FESpace()->Get_PFESpace()),
|
||||
"invalid device GridFunction");
|
||||
return Add(new OccaGridFunctionParameter(name_, *engine, gf, useRestrict));
|
||||
}
|
||||
|
||||
bool OccaCoefficient::IsConstant()
|
||||
{
|
||||
return coeffValue.isNumber();
|
||||
}
|
||||
|
||||
double OccaCoefficient::GetConstantValue()
|
||||
{
|
||||
if (!IsConstant())
|
||||
{
|
||||
mfem_error("OccaCoefficient is not constant");
|
||||
}
|
||||
return coeffValue.number();
|
||||
}
|
||||
|
||||
Vector OccaCoefficient::Eval()
|
||||
{
|
||||
if (integ == NULL)
|
||||
{
|
||||
mfem_error("OccaCoefficient requires a Setup() call before Eval()");
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace &fespace = integ->GetTrialFESpace();
|
||||
const mfem::IntegrationRule &ir = integ->GetIntegrationRule();
|
||||
|
||||
const int elements = fespace.GetNE();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
Vector quadCoeff(*(new Layout(OccaEngine(), numQuad * elements)));
|
||||
Eval(quadCoeff);
|
||||
return quadCoeff;
|
||||
}
|
||||
|
||||
void OccaCoefficient::Eval(Vector &quadCoeff)
|
||||
{
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
static ::occa::kernelBuilder builder =
|
||||
::occa::kernelBuilder::fromFile(okl_path + "coefficient.okl",
|
||||
"CoefficientEval");
|
||||
|
||||
if (integ == NULL)
|
||||
{
|
||||
mfem_error("OccaCoefficient requires a Setup() call before Eval()");
|
||||
}
|
||||
|
||||
const int elements = integ->GetTrialFESpace().GetNE();
|
||||
|
||||
::occa::properties kernelProps = props;
|
||||
if (name != "COEFF")
|
||||
{
|
||||
kernelProps["defines/COEFF"] = name;
|
||||
kernelProps["defines/COEFF_ARGS"] = name + "_ARGS";
|
||||
}
|
||||
|
||||
::occa::kernel evalKernel = builder.build(GetDevice(), kernelProps);
|
||||
evalKernel(elements, *this, quadCoeff.OccaMem());
|
||||
}
|
||||
|
||||
OccaCoefficient::operator ::occa::kernelArg ()
|
||||
{
|
||||
::occa::kernelArg kArg;
|
||||
const int paramCount = (int) params.size();
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
kArg.add(params[i]->KernelArgs());
|
||||
}
|
||||
return kArg;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,287 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
|
||||
#define MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class OccaIntegrator;
|
||||
|
||||
|
||||
class OccaParameter
|
||||
{
|
||||
public:
|
||||
virtual ~OccaParameter();
|
||||
|
||||
virtual OccaParameter* Clone() = 0;
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs();
|
||||
};
|
||||
|
||||
|
||||
//---[ Include Parameter ]------------
|
||||
class OccaIncludeParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
std::string filename;
|
||||
|
||||
public:
|
||||
OccaIncludeParameter(const std::string &filename_);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Source Parameter ]------------
|
||||
class OccaSourceParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
std::string source;
|
||||
|
||||
public:
|
||||
OccaSourceParameter(const std::string &filename_);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Define Parameter ]------------
|
||||
template <class TM>
|
||||
class OccaDefineParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
TM value;
|
||||
|
||||
public:
|
||||
OccaDefineParameter(const std::string &name_,
|
||||
const TM &value_) :
|
||||
name(name_),
|
||||
value(value_) {}
|
||||
|
||||
virtual OccaParameter* Clone()
|
||||
{
|
||||
return new OccaDefineParameter(name, value);
|
||||
}
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["defines"][name] = value;
|
||||
}
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Variable Parameter ]-----------
|
||||
template <class TM>
|
||||
class OccaVariableParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
const TM &value;
|
||||
|
||||
public:
|
||||
OccaVariableParameter(const std::string &name_,
|
||||
const TM &value_) :
|
||||
name(name_),
|
||||
value(value_) {}
|
||||
|
||||
virtual OccaParameter* Clone()
|
||||
{
|
||||
return new OccaVariableParameter(name, value);
|
||||
}
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
std::string &args = (props["defines/COEFF_ARGS"]
|
||||
.asString()
|
||||
.string());
|
||||
// const TM name,\n"
|
||||
args += "const ";
|
||||
args += ::occa::primitiveinfo<TM>::name;
|
||||
args += ' ';
|
||||
args += name;
|
||||
args += ",\n";
|
||||
}
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs()
|
||||
{
|
||||
return ::occa::kernelArg(value);
|
||||
}
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Vector Parameter ]-------
|
||||
class OccaVectorParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
Vector v;
|
||||
bool useRestrict;
|
||||
std::string attr;
|
||||
|
||||
public:
|
||||
OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const bool useRestrict_ = false);
|
||||
|
||||
OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const std::string &attr_,
|
||||
const bool useRestrict_ = false);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs();
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ GridFunction Parameter ]-------
|
||||
class OccaGridFunctionParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
mfem::GridFunction &gf;
|
||||
Vector gfQuad;
|
||||
bool useRestrict;
|
||||
|
||||
public:
|
||||
OccaGridFunctionParameter(const std::string &name_,
|
||||
const Engine &e,
|
||||
mfem::GridFunction &gf_,
|
||||
const bool useRestrict_ = false);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs();
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Coefficient ]------------------
|
||||
// [MISSING]
|
||||
// Needs to know about the integrator's
|
||||
// - fespace
|
||||
// - ir
|
||||
// Step where parameters that need the ir get called for setup
|
||||
// For example, GridFunction (d, e) -> (q, e)
|
||||
class OccaCoefficient
|
||||
{
|
||||
private:
|
||||
SharedPtr<const Engine> engine;
|
||||
|
||||
OccaIntegrator *integ;
|
||||
|
||||
std::string name;
|
||||
::occa::json coeffValue;
|
||||
|
||||
::occa::properties props;
|
||||
std::vector<OccaParameter*> params;
|
||||
|
||||
public:
|
||||
OccaCoefficient(const Engine &e, const double value = 1.0);
|
||||
OccaCoefficient(const Engine &e, mfem::GridFunction &gf,
|
||||
const bool useRestrict = false);
|
||||
OccaCoefficient(const Engine &e, const std::string &source);
|
||||
OccaCoefficient(const Engine &e, const char *source);
|
||||
~OccaCoefficient();
|
||||
|
||||
OccaCoefficient(const OccaCoefficient &coeff);
|
||||
|
||||
const Engine &OccaEngine() const { return *engine; }
|
||||
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return engine->GetDevice(idx); }
|
||||
|
||||
OccaCoefficient& SetName(const std::string &name_);
|
||||
|
||||
void Setup(OccaIntegrator &integ_,
|
||||
::occa::properties &props_);
|
||||
|
||||
OccaCoefficient& Add(OccaParameter *param);
|
||||
|
||||
OccaCoefficient& IncludeHeader(const std::string &filename);
|
||||
OccaCoefficient& IncludeSource(const std::string &source);
|
||||
|
||||
template <class TM>
|
||||
OccaCoefficient& AddDefine(const std::string &name_, const TM &value)
|
||||
{
|
||||
return Add(new OccaDefineParameter<TM>(name_, value));
|
||||
}
|
||||
|
||||
template <class TM>
|
||||
OccaCoefficient& AddVariable(const std::string &name_, const TM &value)
|
||||
{
|
||||
return Add(new OccaVariableParameter<TM>(name_, value));
|
||||
}
|
||||
|
||||
OccaCoefficient& AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const bool useRestrict = false);
|
||||
|
||||
|
||||
OccaCoefficient& AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const std::string &attr,
|
||||
const bool useRestrict = false);
|
||||
|
||||
OccaCoefficient& AddGridFunction(const std::string &name_,
|
||||
mfem::GridFunction &gf,
|
||||
const bool useRestrict = false);
|
||||
|
||||
bool IsConstant();
|
||||
double GetConstantValue();
|
||||
|
||||
Vector Eval();
|
||||
void Eval(Vector &quadCoeff);
|
||||
|
||||
operator ::occa::kernelArg ();
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
|
||||
@@ -1,40 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_OCCA_DEFINES
|
||||
#define MFEM_OCCA_DEFINES
|
||||
|
||||
#ifndef USING_TENSOR_OPS
|
||||
# define USING_TENSOR_OPS 0
|
||||
#endif
|
||||
|
||||
#ifdef OCCA_USING_GPU
|
||||
# define GPU_ORDER_2(I0, I1) @dimOrder(I0, I1)
|
||||
# define GPU_ORDER_3(I0, I1, I2) @dimOrder(I0, I1, I2)
|
||||
# define GPU_ORDER_4(I0, I1, I2, I3) @dimOrder(I0, I1, I2, I3)
|
||||
#else
|
||||
# define GPU_ORDER_2(I0, I1) @dimOrder(0, 1)
|
||||
# define GPU_ORDER_3(I0, I1, I2) @dimOrder(0, 1, 2)
|
||||
# define GPU_ORDER_4(I0, I1, I2, I3) @dimOrder(0, 1, 2, 3)
|
||||
#endif
|
||||
|
||||
#ifndef COEFF
|
||||
# define COEFF 1.0
|
||||
# define COEFF_ARGS
|
||||
#endif
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# include "mfem-occa://defines/tensor.okl"
|
||||
#else
|
||||
# include "mfem-occa://defines/simplex.okl"
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -1,40 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#define USING_LOW_ORDER 1
|
||||
#define USING_HI_ORDER 0
|
||||
|
||||
typedef double* DofToQuad_t @dim(NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD2D_t @dim(2, NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD3D_t @dim(3, NUM_QUAD, NUM_DOFS);
|
||||
|
||||
typedef double* QuadToDof_t @dim(NUM_DOFS, NUM_QUAD);
|
||||
typedef double* QuadToDofD2D_t @dim(2, NUM_DOFS, NUM_QUAD);
|
||||
typedef double* QuadToDofD3D_t @dim(3, NUM_DOFS, NUM_QUAD);
|
||||
|
||||
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD, numElements);
|
||||
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD, numElements);
|
||||
|
||||
typedef double* SymmOperator2D_t @dim(3, NUM_QUAD, numElements);
|
||||
typedef double* SymmOperator3D_t @dim(6, NUM_QUAD, numElements);
|
||||
|
||||
typedef double* DLocal_t @dim(NUM_DOFS, numElements);
|
||||
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
|
||||
|
||||
#if VDIM_ORDERING == ORDERING_BY_VDIM
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements);
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements);
|
||||
#else
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements) @dimOrder(2,0,1);
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements) @dimOrder(2,0,1);
|
||||
#endif
|
||||
|
||||
typedef int* DLocalMap_t @dim(NUM_DOFS, numElements);
|
||||
@@ -1,85 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#if NUM_QUAD_1D < NUM_DOFS_1D
|
||||
# define NUM_MAX_1D NUM_DOFS_1D
|
||||
#else
|
||||
# define NUM_MAX_1D NUM_QUAD_1D
|
||||
#endif
|
||||
|
||||
#define NUM_MAX_2D (NUM_MAX_1D * NUM_MAX_1D)
|
||||
|
||||
#define NUM_QUAD_DOFS_1D (NUM_QUAD_1D * NUM_DOFS_1D)
|
||||
|
||||
#define QUAD_2D_ID(X, Y) (X + ((Y) * NUM_QUAD_1D))
|
||||
#define DOFS_2D_ID(X, Y) (X + ((Y) * NUM_DOFS_1D))
|
||||
|
||||
#define QUAD_3D_ID(X, Y, Z) (X + ((Y) * NUM_QUAD_1D) + ((Z) * NUM_QUAD_2D))
|
||||
#define DOFS_3D_ID(X, Y, Z) (X + ((Y) * NUM_DOFS_1D) + ((Z) * NUM_DOFS_2D))
|
||||
|
||||
#if NUM_MAX_1D < 8
|
||||
# define USING_LOW_ORDER 1
|
||||
# define USING_HI_ORDER 0
|
||||
#else
|
||||
# define USING_LOW_ORDER 0
|
||||
# define USING_HI_ORDER 1
|
||||
#endif
|
||||
|
||||
#define M1_ELEMENT_BATCHES (M1_OUTER_ELEMENT_BATCH * M1_INNER_ELEMENT_BATCH)
|
||||
|
||||
typedef double* DofToQuad_t @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
typedef double* QuadToDof_t @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
typedef double* Jacobian_t @dim(NUM_DIM, NUM_DIM, numElements);
|
||||
typedef double* Jacobian1D_t @dim(NUM_QUAD_1D, numElements);
|
||||
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD_2D, numElements);
|
||||
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD_3D, numElements);
|
||||
|
||||
typedef double* SymmOperator1D_t @dim(NUM_QUAD_1D, numElements);
|
||||
typedef double* SymmOperator2D_t @dim(3, NUM_QUAD_2D, numElements);
|
||||
typedef double* SymmOperator3D_t @dim(6, NUM_QUAD_3D, numElements);
|
||||
|
||||
typedef double* DLocal_t @dim(NUM_DOFS, numElements);
|
||||
typedef double* DLocal1D_t @dim(NUM_DOFS_1D, numElements);
|
||||
typedef double* DLocal2D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
typedef double* DLocal3D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
|
||||
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
|
||||
typedef double* QLocal1D_t @dim(NUM_QUAD_1D, numElements);
|
||||
typedef double* QLocal2D_t @dim(NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
typedef double* QLocal3D_t @dim(NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
|
||||
#if VDIM_ORDERING == ORDERING_BY_VDIM
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements);
|
||||
typedef double* DVLocal1D_t @dim(NUM_VDIM, NUM_DOFS_1D, numElements);
|
||||
typedef double* DVLocal2D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
typedef double* DVLocal3D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements);
|
||||
typedef double* QVLocal1D_t @dim(NUM_VDIM, NUM_QUAD_1D, numElements);
|
||||
typedef double* QVLocal2D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
typedef double* QVLocal3D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
#else
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements) @dimOrder(2,0,1);
|
||||
typedef double* DVLocal1D_t @dim(NUM_VDIM, NUM_DOFS_1D, numElements) @dimOrder(2,0,1);
|
||||
typedef double* DVLocal2D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, numElements) @dimOrder(3,0,1,2);
|
||||
typedef double* DVLocal3D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements) @dimOrder(4,0,1,2,3);
|
||||
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements) @dimOrder(2,0,1);
|
||||
typedef double* QVLocal1D_t @dim(NUM_VDIM, NUM_QUAD_1D, numElements) @dimOrder(2,0,1);
|
||||
typedef double* QVLocal2D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, numElements) @dimOrder(3,0,1,2);
|
||||
typedef double* QVLocal3D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements) @dimOrder(4,0,1,2,3);
|
||||
#endif
|
||||
|
||||
typedef int* DLocalMap_t @dim(NUM_DOFS, numElements);
|
||||
typedef int* DLocalMap1D_t @dim(NUM_DOFS_1D, numElements);
|
||||
typedef int* DLocalMap2D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
typedef int* DLocalMap3D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
@@ -1,168 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double *quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict SymmOperator2D_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J12*J12 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J12*J11 + J22*J21); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J21*J21); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD2D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD2D_t quadToDofD,
|
||||
@restrict const SymmOperator2D_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_sol[NUM_DOFS];
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] = 0;
|
||||
}
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
double gradX = 0, gradY = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
const double gradX2 = (O11 * gradX) + (O12 * gradY);
|
||||
const double gradY2 = (O12 * gradX) + (O22 * gradY);
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] += ((gradX2 * quadToDofD(0, d, q)) +
|
||||
(gradY2 * quadToDofD(1, d, q)));
|
||||
}
|
||||
}
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
solOut(d, e) += r_sol[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict SymmOperator3D_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3) + (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3) + (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD3D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD3D_t quadToDofD,
|
||||
@restrict const SymmOperator3D_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_sol[NUM_DOFS];
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] = 0;
|
||||
}
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
double gradX = 0, gradY = 0, gradZ = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
gradZ += s * quadToDofD(2, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
const double gradX2 = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
|
||||
const double gradY2 = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
|
||||
const double gradZ2 = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] += ((gradX2 * quadToDofD(0, d, q)) +
|
||||
(gradY2 * quadToDofD(1, d, q)) +
|
||||
(gradZ2 * quadToDofD(2, d, q)));
|
||||
}
|
||||
}
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
solOut(d, e) += r_sol[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,182 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator2D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += A2_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J12*J12 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J12*J11 + J22*J21); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J21*J21); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD2D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD2D_t quadToDofD,
|
||||
@restrict const SymmOperator2D_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gradX[NUM_QUAD];
|
||||
@shared double s_gradY[NUM_QUAD];
|
||||
|
||||
for (int qOff = 0; qOff < M2_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M2_INNER_BATCH) {
|
||||
double gradX = 0, gradY = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
s_gradX[q] = (O11 * gradX) + (O12 * gradY);
|
||||
s_gradY[q] = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
|
||||
for (int dOff = 0; dOff < M2_INNER_BATCH; ++dOff) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M2_INNER_BATCH) {
|
||||
double r_sol = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
// FIXME: s_gradX and s_gradY are @shared used outside of @inner
|
||||
r_sol += ((s_gradX[q] * quadToDofD(0, d, q)) +
|
||||
(s_gradY[q] * quadToDofD(1, d, q)));
|
||||
}
|
||||
solOut(d, e) += r_sol;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator3D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += A3_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3) + (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3) + (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD3D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD3D_t quadToDofD,
|
||||
@restrict const SymmOperator3D_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gradX[NUM_QUAD];
|
||||
@shared double s_gradY[NUM_QUAD];
|
||||
@shared double s_gradZ[NUM_QUAD];
|
||||
|
||||
for (int qOff = 0; qOff < M3_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M3_INNER_BATCH) {
|
||||
double gradX = 0, gradY = 0, gradZ = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
gradZ += s * quadToDofD(2, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
s_gradX[q] = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
|
||||
s_gradY[q] = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
|
||||
s_gradZ[q] = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
|
||||
}
|
||||
}
|
||||
|
||||
for (int dOff = 0; dOff < M3_INNER_BATCH; ++dOff) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M3_INNER_BATCH) {
|
||||
double r_sol = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_sol += ((s_gradX[q] * quadToDofD(0, d, q)) +
|
||||
(s_gradY[q] * quadToDofD(1, d, q)) +
|
||||
(s_gradZ[q] * quadToDofD(2, d, q)));
|
||||
}
|
||||
solOut(d, e) += r_sol;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,370 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict SymmOperator1D_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
oper(q, e) = quadWeights[q] * COEFF / J(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator1D_t oper,
|
||||
@restrict const DLocal1D_t solIn,
|
||||
@restrict DLocal1D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double grad[NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] += s * dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] *= oper(qx, e);
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double gradX = grad[qx];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, e) += gradX * quadToDofD(dx, qx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict SymmOperator2D_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_2D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J21*J21 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J21*J11 + J22*J12); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J12*J12); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator2D_t oper,
|
||||
@restrict const DLocal2D_t solIn,
|
||||
@restrict DLocal2D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double grad[NUM_QUAD_1D][NUM_QUAD_1D][2];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qy][qx][0] = 0;
|
||||
grad[qy][qx][1] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double gradX[NUM_QUAD_1D][2];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] = 0;
|
||||
gradX[qx][1] = 0;
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, dy, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] += s * dofToQuad(qx, dx);
|
||||
gradX[qx][1] += s * dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
const double wDy = dofToQuadD(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qy][qx][0] += gradX[qx][1] * wy;
|
||||
grad[qy][qx][1] += gradX[qx][0] * wDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate Dxy, xDy in plane
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const int q = QUAD_2D_ID(qx, qy);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
const double gradX = grad[qy][qx][0];
|
||||
const double gradY = grad[qy][qx][1];
|
||||
|
||||
grad[qy][qx][0] = (O11 * gradX) + (O12 * gradY);
|
||||
grad[qy][qx][1] = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double gradX[NUM_DOFS_1D][2];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradX[dx][0] = 0;
|
||||
gradX[dx][1] = 0;
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double gX = grad[qy][qx][0];
|
||||
const double gY = grad[qy][qx][1];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = quadToDof(dx, qx);
|
||||
const double wDx = quadToDofD(dx, qx);
|
||||
gradX[dx][0] += gX * wDx;
|
||||
gradX[dx][1] += gY * wx;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = quadToDof(dy, qy);
|
||||
const double wDy = quadToDofD(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, dy, e) += ((gradX[dx][0] * wy) +
|
||||
(gradX[dx][1] * wDy));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict SymmOperator3D_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_3D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3), (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3), (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator3D_t oper,
|
||||
@restrict const DLocal3D_t solIn,
|
||||
@restrict DLocal3D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double grad[NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D][4];
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qz][qy][qx][0] = 0;
|
||||
grad[qz][qy][qx][1] = 0;
|
||||
grad[qz][qy][qx][2] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
double gradXY[NUM_QUAD_1D][NUM_QUAD_1D][4];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradXY[qy][qx][0] = 0;
|
||||
gradXY[qy][qx][1] = 0;
|
||||
gradXY[qy][qx][2] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double gradX[NUM_QUAD_1D][2];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] = 0;
|
||||
gradX[qx][1] = 0;
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, dy, dz, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] += s * dofToQuad(qx, dx);
|
||||
gradX[qx][1] += s * dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
const double wDy = dofToQuadD(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double wx = gradX[qx][0];
|
||||
const double wDx = gradX[qx][1];
|
||||
gradXY[qy][qx][0] += wDx * wy;
|
||||
gradXY[qy][qx][1] += wx * wDy;
|
||||
gradXY[qy][qx][2] += wx * wy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
const double wz = dofToQuad(qz, dz);
|
||||
const double wDz = dofToQuadD(qz, dz);
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qz][qy][qx][0] += gradXY[qy][qx][0] * wz;
|
||||
grad[qz][qy][qx][1] += gradXY[qy][qx][1] * wz;
|
||||
grad[qz][qy][qx][2] += gradXY[qy][qx][2] * wDz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const int q = QUAD_3D_ID(qx, qy, qz);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
const double gradX = grad[qz][qy][qx][0];
|
||||
const double gradY = grad[qz][qy][qx][1];
|
||||
const double gradZ = grad[qz][qy][qx][2];
|
||||
|
||||
grad[qz][qy][qx][0] = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
|
||||
grad[qz][qy][qx][1] = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
|
||||
grad[qz][qy][qx][2] = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
double gradXY[NUM_DOFS_1D][NUM_DOFS_1D][4];
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradXY[dy][dx][0] = 0;
|
||||
gradXY[dy][dx][1] = 0;
|
||||
gradXY[dy][dx][2] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double gradX[NUM_DOFS_1D][4];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradX[dx][0] = 0;
|
||||
gradX[dx][1] = 0;
|
||||
gradX[dx][2] = 0;
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double gX = grad[qz][qy][qx][0];
|
||||
const double gY = grad[qz][qy][qx][1];
|
||||
const double gZ = grad[qz][qy][qx][2];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = quadToDof(dx, qx);
|
||||
const double wDx = quadToDofD(dx, qx);
|
||||
gradX[dx][0] += gX * wDx;
|
||||
gradX[dx][1] += gY * wx;
|
||||
gradX[dx][2] += gZ * wx;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = quadToDof(dy, qy);
|
||||
const double wDy = quadToDofD(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradXY[dy][dx][0] += gradX[dx][0] * wy;
|
||||
gradXY[dy][dx][1] += gradX[dx][1] * wDy;
|
||||
gradXY[dy][dx][2] += gradX[dx][2] * wy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = quadToDof(dz, qz);
|
||||
const double wDz = quadToDofD(dz, qz);
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, dy, dz, e) += ((gradXY[dy][dx][0] * wz) +
|
||||
(gradXY[dy][dx][1] * wz) +
|
||||
(gradXY[dy][dx][2] * wDz));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,435 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator1D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A1_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A1_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
oper(q, e) = quadWeights[q] * COEFF / J(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator1D_t oper,
|
||||
@restrict const DLocal1D_t solIn,
|
||||
@restrict DLocal1D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M1_ELEMENT_BATCHES; @outer) {
|
||||
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
@exclusive double grad[NUM_QUAD_1D];
|
||||
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
for (int i = el; i < NUM_QUAD_DOFS_1D; i += M1_INNER_ELEMENT_BATCH) {
|
||||
s_dofToQuadD[i] = dofToQuadD[i];
|
||||
s_quadToDofD[i] = quadToDofD[i];
|
||||
}
|
||||
}
|
||||
|
||||
for (int b = 0; b < M1_OUTER_ELEMENT_BATCH; ++b) {
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
const int e = eOff + b*M1_INNER_ELEMENT_BATCH + el;
|
||||
if (e < numElements) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] += s * s_dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] *= oper(qx, e);
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
double s = 0;
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
s += grad[qx] * s_quadToDofD(dx, qx);
|
||||
}
|
||||
solOut(dx, e) += s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator2D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD_2D; q += A2_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J21*J21 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J21*J11 + J22*J12); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J12*J12); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator2D_t oper,
|
||||
@restrict const DLocal2D_t solIn,
|
||||
@restrict DLocal2D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M2_ELEMENT_BATCH; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_xy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_xDy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_grad[2 * NUM_QUAD_2D] @dim(2, NUM_QUAD_1D, NUM_QUAD_1D);
|
||||
|
||||
@exclusive double r_x[NUM_MAX_1D];
|
||||
@exclusive double r_y[NUM_QUAD_1D];
|
||||
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
for (int id = x; id < NUM_QUAD_DOFS_1D; id += NUM_MAX_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
s_dofToQuadD[id] = dofToQuadD[id];
|
||||
s_quadToDof[id] = quadToDof[id];
|
||||
s_quadToDofD[id] = quadToDofD[id];
|
||||
}
|
||||
}
|
||||
|
||||
for (int e = eOff; e < (eOff + M2_ELEMENT_BATCH); ++e) {
|
||||
if (e < numElements) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
s_xy(dx, qy) = 0;
|
||||
s_xDy(dx, qy) = 0;
|
||||
}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
r_x[dy] = solIn(dx, dy, e);
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double xy = 0;
|
||||
double xDy = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
xy += r_x[dy] * s_dofToQuad(qy, dy);
|
||||
xDy += r_x[dy] * s_dofToQuadD(qy, dy);
|
||||
}
|
||||
s_xy(dx, qy) = xy;
|
||||
s_xDy(dx, qy) = xDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
if (qy < NUM_QUAD_1D) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
double gradX = 0, gradY = 0;
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradX += s_xy(dx, qy) * s_dofToQuadD(qx, dx);
|
||||
gradY += s_xDy(dx, qy) * s_dofToQuad(qx, dx);
|
||||
}
|
||||
|
||||
const int q = QUAD_2D_ID(qx, qy);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
s_grad(0, qx, qy) = (O11 * gradX) + (O12 * gradY);
|
||||
s_grad(1, qx, qy) = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx; @inner) {
|
||||
if (qx < NUM_QUAD_1D) {
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
s_xy(dy, qx) = 0;
|
||||
s_xDy(dy, qx) = 0;
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
r_x[qy] = s_grad(0, qx, qy);
|
||||
r_y[qy] = s_grad(1, qx, qy);
|
||||
}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double xy = 0;
|
||||
double xDy = 0;
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
xy += r_x[qy] * s_quadToDof(dy, qy);
|
||||
xDy += r_y[qy] * s_quadToDofD(dy, qy);
|
||||
}
|
||||
s_xy(dy, qx) = xy;
|
||||
s_xDy(dy, qx) = xDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double s = 0;
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
s += ((s_xy(dy, qx) * s_quadToDofD(dx, qx)) +
|
||||
(s_xDy(dy, qx) * s_quadToDof(dx, qx)));
|
||||
}
|
||||
solOut(dx, dy, e) += s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator3D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD_3D; q += A3_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3), (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3), (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator3D_t oper,
|
||||
@restrict const DLocal3D_t solIn,
|
||||
@restrict DLocal3D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_z[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
@shared double s_Dz[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
@shared double s_xyDz[NUM_QUAD_2D] @dim(NUM_QUAD_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store z axis as registers
|
||||
@exclusive double r_qz[NUM_QUAD_1D];
|
||||
@exclusive double r_qDz[NUM_QUAD_1D];
|
||||
@exclusive double r_dDxyz[NUM_DOFS_1D];
|
||||
@exclusive double r_dxDyz[NUM_DOFS_1D];
|
||||
@exclusive double r_dxyDz[NUM_DOFS_1D];
|
||||
|
||||
for (int y = 0; y < NUM_MAX_1D; ++y; @inner) {
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
const int id = (y * NUM_MAX_1D) + x;
|
||||
// Fetch Q <--> D maps
|
||||
if (id < NUM_QUAD_DOFS_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
s_dofToQuadD[id] = dofToQuadD[id];
|
||||
s_quadToDof[id] = quadToDof[id];
|
||||
s_quadToDofD[id] = quadToDofD[id];
|
||||
}
|
||||
// Initialize our Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] = 0;
|
||||
r_qDz[qz] = 0;
|
||||
}
|
||||
// Initialize our solution updates in the Z axis
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
r_dDxyz[dz] = 0;
|
||||
r_dxDyz[dz] = 0;
|
||||
r_dxyDz[dz] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double s = solIn(dx, dy, dz, e);
|
||||
// Calculate D -> Q in the Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] += s * s_dofToQuad(qz, dz);
|
||||
r_qDz[qz] += s * s_dofToQuadD(qz, dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// For each xy plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
// Fill xy plane at given z position
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
s_z(dx, dy) = r_qz[qz];
|
||||
s_Dz(dx, dy) = r_qDz[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
double Dxyz = 0;
|
||||
double xDyz = 0;
|
||||
double xyDz = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = s_dofToQuad(qy, dy);
|
||||
const double wDy = s_dofToQuadD(qy, dy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = s_dofToQuad(qx, dx);
|
||||
const double wDx = s_dofToQuadD(qx, dx);
|
||||
const double z = s_z(dx, dy);
|
||||
const double Dz = s_Dz(dx, dy);
|
||||
Dxyz += wDx * wy * z;
|
||||
xDyz += wx * wDy * z;
|
||||
xyDz += wx * wy * Dz;
|
||||
}
|
||||
}
|
||||
|
||||
const int q = QUAD_3D_ID(qx, qy, qz);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
const double qDxyz = (O11 * Dxyz) + (O12 * xDyz) + (O13 * xyDz);
|
||||
const double qxDyz = (O12 * Dxyz) + (O22 * xDyz) + (O23 * xyDz);
|
||||
const double qxyDz = (O13 * Dxyz) + (O23 * xDyz) + (O33 * xyDz);
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = s_quadToDof(dz, qz);
|
||||
const double wDz = s_quadToDofD(dz, qz);
|
||||
r_dDxyz[dz] += wz * qDxyz;
|
||||
r_dxDyz[dz] += wz * qxDyz;
|
||||
r_dxyDz[dz] += wDz * qxyDz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@barrier("s_z_s_Dz_sync_1");
|
||||
}
|
||||
// Iterate over xy planes to compute solution
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
// Place xy plane in shared memory
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
s_z(qx, qy) = r_dDxyz[dz];
|
||||
s_Dz(qx, qy) = r_dxDyz[dz];
|
||||
s_xyDz(qx, qy) = r_dxyDz[dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Finalize solution in xy plane
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
double solZ = 0;
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = s_quadToDof(dy, qy);
|
||||
const double wDy = s_quadToDofD(dy, qy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double wx = s_quadToDof(dx, qx);
|
||||
const double wDx = s_quadToDofD(dx, qx);
|
||||
const double Dxyz = s_z(qx, qy);
|
||||
const double xDyz = s_Dz(qx, qy);
|
||||
const double xyDz = s_xyDz(qx, qy);
|
||||
solZ += ((wDx * wy * Dxyz) +
|
||||
(wx * wDy * xDyz) +
|
||||
(wx * wy * xyDz));
|
||||
}
|
||||
}
|
||||
solOut(dx, dy, dz, e) += solZ;
|
||||
}
|
||||
}
|
||||
}
|
||||
@barrier("s_z_s_Dz_s_xyDz_sync_1");
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,167 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "url_handler.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
bool Engine::fileOpenerRegistered = false;
|
||||
|
||||
void Engine::Init(const std::string &engine_spec)
|
||||
{
|
||||
//
|
||||
// Initialize inherited fields
|
||||
//
|
||||
memory_resources[0] = NULL;
|
||||
workers_weights[0]= 1.0;
|
||||
workers_mem_res[0] = 0;
|
||||
|
||||
//
|
||||
// Initialize the OCCA engine
|
||||
//
|
||||
::occa::properties props(engine_spec);
|
||||
device = new ::occa::device[1];
|
||||
device[0].setup(props);
|
||||
|
||||
okl_path = "mfem-occa://";
|
||||
if (!fileOpenerRegistered)
|
||||
{
|
||||
// The directories from "MFEM_OCCA_OKL_PATH", if any, have the highest
|
||||
// priority.
|
||||
FileOpener *fo = new FileOpener("mfem-occa://", "MFEM_OCCA_OKL_PATH");
|
||||
// Next in priority is the source path, if it exists.
|
||||
std::string mfem_src_prefix = mfem::GetSourcePath();
|
||||
fo->AddDir(mfem_src_prefix + "/backends/occa");
|
||||
// And last in priority is the install path, if it exists.
|
||||
std::string mfem_install_prefix = mfem::GetInstallPath();
|
||||
fo->AddDir(mfem_install_prefix + "/lib/mfem/occa");
|
||||
::occa::io::fileOpener::add(fo);
|
||||
fileOpenerRegistered = true;
|
||||
}
|
||||
// std::cout << "OCCA device properties:\n" << device[0].properties();
|
||||
|
||||
force_cuda_aware_mpi = false;
|
||||
}
|
||||
|
||||
Engine::Engine(const std::string &engine_spec)
|
||||
: mfem::Engine(NULL, 1, 1)
|
||||
{
|
||||
Init(engine_spec);
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
Engine::Engine(MPI_Comm _comm, const std::string &engine_spec)
|
||||
: mfem::Engine(NULL, 1, 1)
|
||||
{
|
||||
comm = _comm;
|
||||
Init(engine_spec);
|
||||
}
|
||||
#endif
|
||||
|
||||
bool Engine::CheckEngine(const mfem::Engine *engine) const
|
||||
{
|
||||
return (engine != NULL && util::Is<const Engine>(engine) != NULL &&
|
||||
*util::As<const Engine>(engine) == *this);
|
||||
}
|
||||
|
||||
bool Engine::CheckLayout(const PLayout *layout) const
|
||||
{
|
||||
return (layout != NULL && util::Is<const Layout>(layout) != NULL &&
|
||||
layout->As<Layout>().OccaEngine() == *this);
|
||||
}
|
||||
|
||||
bool Engine::CheckArray(const PArray *array) const
|
||||
{
|
||||
return (array != NULL && util::Is<const Array>(array) != NULL &&
|
||||
array->As<Array>().OccaEngine() == *this);
|
||||
}
|
||||
|
||||
bool Engine::CheckVector(const PVector *vector) const
|
||||
{
|
||||
return (vector != NULL && util::Is<const Vector>(vector) != NULL &&
|
||||
vector->As<Vector>().OccaEngine() == *this);
|
||||
}
|
||||
|
||||
bool Engine::CheckFESpace(const PFiniteElementSpace *fes) const
|
||||
{
|
||||
return (fes != NULL && util::Is<const FiniteElementSpace>(fes) != NULL &&
|
||||
fes->As<FiniteElementSpace>().OccaEngine() == *this);
|
||||
}
|
||||
|
||||
DLayout Engine::MakeLayout(std::size_t size) const
|
||||
{
|
||||
return DLayout(new Layout(*this, size));
|
||||
}
|
||||
|
||||
DLayout Engine::MakeLayout(const mfem::Array<std::size_t> &offsets) const
|
||||
{
|
||||
MFEM_ASSERT(offsets.Size() == 2,
|
||||
"multiple workers are not supported yet");
|
||||
return DLayout(new Layout(*this, offsets.Last()));
|
||||
}
|
||||
|
||||
DArray Engine::MakeArray(PLayout &layout, std::size_t item_size) const
|
||||
{
|
||||
return DArray(new Array(layout.As<Layout>(), item_size));
|
||||
}
|
||||
|
||||
DVector Engine::MakeVector(PLayout &layout, int type_id) const
|
||||
{
|
||||
MFEM_ASSERT(type_id == ScalarId<double>::value, "type_id " << type_id
|
||||
<< " is not supported");
|
||||
return DVector(new Vector(layout.As<Layout>()));
|
||||
}
|
||||
|
||||
DFiniteElementSpace Engine::MakeFESpace(mfem::FiniteElementSpace &fespace) const
|
||||
{
|
||||
return DFiniteElementSpace(new FiniteElementSpace(*this, fespace));
|
||||
}
|
||||
|
||||
DBilinearForm Engine::MakeBilinearForm(mfem::BilinearForm &bf) const
|
||||
{
|
||||
return DBilinearForm(new BilinearForm(*this, bf));
|
||||
}
|
||||
|
||||
void Engine::AssembleLinearForm(LinearForm &l_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
}
|
||||
|
||||
mfem::Operator *Engine::MakeOperator(const MixedBilinearForm &mbl_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
mfem::Operator *Engine::MakeOperator(const NonlinearForm &nl_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,144 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_ENGINE_HPP
|
||||
#define MFEM_BACKENDS_OCCA_ENGINE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "../base/backend.hpp"
|
||||
#include <occa.hpp>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Engine : public mfem::Engine
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// mfem::Backend *backend;
|
||||
#ifdef MFEM_USE_MPI
|
||||
// MPI_Comm comm;
|
||||
#endif
|
||||
// int num_mem_res;
|
||||
// int num_workers;
|
||||
// MemoryResource **memory_resources;
|
||||
// double *workers_weights;
|
||||
// int *workers_mem_res;
|
||||
|
||||
static bool fileOpenerRegistered;
|
||||
/// An array of OCCA devices. Currently only a single device is supported.
|
||||
::occa::device *device;
|
||||
std::string okl_path;
|
||||
bool force_cuda_aware_mpi;
|
||||
|
||||
void Init(const std::string &engine_spec);
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
Engine(const std::string &engine_spec);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// TODO: doxygen
|
||||
Engine(MPI_Comm comm, const std::string &engine_spec);
|
||||
#endif
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual ~Engine() { delete [] device; }
|
||||
|
||||
/**
|
||||
@name OCCA specific interface, used by other objects in the OCCA backend
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// Get the associated OCCA device.
|
||||
::occa::device GetDevice(int idx = 0) const { return device[idx]; }
|
||||
|
||||
/// TODO: doxygen
|
||||
const std::string &GetOklPath() const { return okl_path; }
|
||||
|
||||
/// OCCA device memory allocation.
|
||||
::occa::memory Alloc(std::size_t bytes) const
|
||||
{ return GetDevice().malloc(bytes); }
|
||||
|
||||
/// Two mfem::occa::Engine%s are equal if they use the same OCCA device.
|
||||
bool operator==(const Engine &other) const
|
||||
{ return GetDevice() == other.GetDevice(); }
|
||||
|
||||
/// TODO: doxygen
|
||||
bool CheckEngine(const mfem::Engine *e) const;
|
||||
|
||||
/// TODO: doxygen
|
||||
bool CheckLayout(const PLayout *layout) const;
|
||||
|
||||
/// TODO: doxygen
|
||||
bool CheckArray(const PArray *array) const;
|
||||
|
||||
/// TODO: doxygen
|
||||
bool CheckVector(const PVector *vector) const;
|
||||
|
||||
/// TODO: doxygen
|
||||
bool CheckFESpace(const PFiniteElementSpace *fes) const;
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
void SetForceCudaAwareMPI(bool force = true)
|
||||
{ force_cuda_aware_mpi = force; }
|
||||
|
||||
bool GetForceCudaAwareMPI() const { return force_cuda_aware_mpi; }
|
||||
#endif
|
||||
|
||||
///@}
|
||||
// End: OCCA specific interface
|
||||
|
||||
/**
|
||||
@name Virtual interface: finite element data structures and algorithms
|
||||
*/
|
||||
///@{
|
||||
|
||||
virtual DLayout MakeLayout(std::size_t size) const;
|
||||
virtual DLayout MakeLayout(const mfem::Array<std::size_t> &offsets) const;
|
||||
|
||||
virtual DArray MakeArray(PLayout &layout, std::size_t item_size) const;
|
||||
|
||||
virtual DVector MakeVector(PLayout &layout,
|
||||
int type_id = ScalarId<double>::value) const;
|
||||
|
||||
virtual DFiniteElementSpace MakeFESpace(mfem::FiniteElementSpace &
|
||||
fespace) const;
|
||||
|
||||
virtual DBilinearForm MakeBilinearForm(mfem::BilinearForm &bf) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual void AssembleLinearForm(LinearForm &l_form) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual mfem::Operator *MakeOperator(const MixedBilinearForm &mbl_form) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual mfem::Operator *MakeOperator(const NonlinearForm &nl_form) const;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_ENGINE_HPP
|
||||
@@ -1,532 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "interpolation.hpp"
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#ifdef OMPI_RELEASE_VERSION
|
||||
#include <mpi-ext.h> // Check for cuda support
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
FiniteElementSpace::FiniteElementSpace(const Engine &e,
|
||||
mfem::FiniteElementSpace &fespace)
|
||||
: PFiniteElementSpace(e, fespace),
|
||||
e_layout(new Layout(e, 0)) // resized in SetupLocalGlobalMaps()
|
||||
{
|
||||
vdim = fespace.GetVDim();
|
||||
ordering = fespace.GetOrdering();
|
||||
|
||||
SetupLocalGlobalMaps();
|
||||
SetupOperators(); // calls virtual methods of 'fes'
|
||||
SetupKernels();
|
||||
}
|
||||
|
||||
FiniteElementSpace::~FiniteElementSpace()
|
||||
{
|
||||
delete restrictionOp;
|
||||
delete prolongationOp;
|
||||
}
|
||||
|
||||
void FiniteElementSpace::SetupLocalGlobalMaps()
|
||||
{
|
||||
const int elements = fes->GetNE();
|
||||
|
||||
if (elements == 0) { return; }
|
||||
|
||||
// Assuming of finite elements are the same.
|
||||
const mfem::FiniteElement &fe = *fes->GetFE(0);
|
||||
const mfem::TensorBasisElement *el =
|
||||
dynamic_cast<const mfem::TensorBasisElement*>(&fe);
|
||||
|
||||
const mfem::Table &e2dTable = fes->GetElementToDofTable();
|
||||
const int *elementMap = e2dTable.GetJ();
|
||||
|
||||
globalDofs = fes->GetNDofs();
|
||||
localDofs = fe.GetDof();
|
||||
|
||||
e_layout->OccaResize(e2dTable.Size_of_connections());
|
||||
|
||||
int *elementDofMap = new int[localDofs];
|
||||
if (el)
|
||||
{
|
||||
::memcpy(elementDofMap,
|
||||
el->GetDofMap().GetData(),
|
||||
localDofs * sizeof(int));
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i < localDofs; ++i)
|
||||
{
|
||||
elementDofMap[i] = i;
|
||||
}
|
||||
}
|
||||
|
||||
// Allocate device offsets and indices
|
||||
globalToLocalOffsets.allocate(GetDevice(),
|
||||
globalDofs + 1);
|
||||
globalToLocalIndices.allocate(GetDevice(),
|
||||
localDofs, elements);
|
||||
localToGlobalMap.allocate(GetDevice(),
|
||||
localDofs, elements);
|
||||
|
||||
int *offsets = globalToLocalOffsets.ptr();
|
||||
int *indices = globalToLocalIndices.ptr();
|
||||
int *l2gMap = localToGlobalMap.ptr();
|
||||
|
||||
// We'll be keeping a count of how many local nodes point
|
||||
// to its global dof
|
||||
for (int i = 0; i <= globalDofs; ++i)
|
||||
{
|
||||
offsets[i] = 0;
|
||||
}
|
||||
|
||||
for (int e = 0; e < elements; ++e)
|
||||
{
|
||||
MFEM_ASSERT(e2dTable.RowSize(e) == localDofs, "");
|
||||
for (int d = 0; d < localDofs; ++d)
|
||||
{
|
||||
const int gid = elementMap[localDofs*e + d];
|
||||
++offsets[gid + 1];
|
||||
}
|
||||
}
|
||||
// Aggregate to find offsets for each global dof
|
||||
for (int i = 1; i <= globalDofs; ++i)
|
||||
{
|
||||
offsets[i] += offsets[i - 1];
|
||||
}
|
||||
// For each global dof, fill in all local nodes that point
|
||||
// to it
|
||||
for (int e = 0; e < elements; ++e)
|
||||
{
|
||||
for (int d = 0; d < localDofs; ++d)
|
||||
{
|
||||
const int gid = elementMap[localDofs*e + elementDofMap[d]];
|
||||
const int lid = localDofs*e + d;
|
||||
indices[offsets[gid]++] = lid;
|
||||
l2gMap[lid] = gid;
|
||||
}
|
||||
}
|
||||
// We shifted the offsets vector by 1 by using it
|
||||
// as a counter. Now we shift it back.
|
||||
for (int i = globalDofs; i > 0; --i)
|
||||
{
|
||||
offsets[i] = offsets[i - 1];
|
||||
}
|
||||
offsets[0] = 0;
|
||||
|
||||
delete [] elementDofMap;
|
||||
|
||||
globalToLocalOffsets.keepInDevice();
|
||||
globalToLocalIndices.keepInDevice();
|
||||
localToGlobalMap.keepInDevice();
|
||||
}
|
||||
|
||||
void FiniteElementSpace::SetupOperators() const
|
||||
{
|
||||
// Construct 'restrictionOp' and 'prolongationOp'.
|
||||
|
||||
prolongationOp = restrictionOp = NULL;
|
||||
|
||||
const mfem::SparseMatrix *R = fes->GetRestrictionMatrix();
|
||||
const mfem::Operator *P = fes->GetProlongationMatrix();
|
||||
|
||||
if (!P) { return; }
|
||||
|
||||
Layout &v_layout = OccaVLayout();
|
||||
Layout &t_layout = OccaTrueVLayout();
|
||||
|
||||
// Assuming R has one entry per row equal to 1.
|
||||
MFEM_ASSERT(R->Finalized(), "");
|
||||
const int tdofs = R->Height();
|
||||
MFEM_ASSERT(tdofs == (int)t_layout.Size(), "");
|
||||
MFEM_ASSERT(tdofs == R->GetI()[tdofs], "");
|
||||
::occa::array<int> ltdof_ldof(GetDevice(), tdofs, R->GetJ());
|
||||
ltdof_ldof.keepInDevice();
|
||||
|
||||
restrictionOp = new RestrictionOperator(v_layout, t_layout, ltdof_ldof);
|
||||
|
||||
const mfem::SparseMatrix *pmat = dynamic_cast<const mfem::SparseMatrix*>(P);
|
||||
if (pmat)
|
||||
{
|
||||
const mfem::SparseMatrix *pmatT = Transpose(*pmat);
|
||||
|
||||
OccaSparseMatrix *occaP =
|
||||
CreateMappedSparseMatrix(t_layout, v_layout, *pmat);
|
||||
OccaSparseMatrix *occaPT =
|
||||
CreateMappedSparseMatrix(v_layout, t_layout, *pmatT);
|
||||
|
||||
prolongationOp = new ProlongationOperator(*occaP, *occaPT);
|
||||
|
||||
delete occaPT;
|
||||
delete occaP;
|
||||
}
|
||||
#ifdef MFEM_USE_MPI
|
||||
else if (fes->Conforming() && dynamic_cast<ParFiniteElementSpace*>(fes))
|
||||
{
|
||||
ParFiniteElementSpace *pfes = static_cast<ParFiniteElementSpace*>(fes);
|
||||
prolongationOp = new OccaConformingProlongation(*this, *pfes,
|
||||
ltdof_ldof.memory());
|
||||
}
|
||||
#endif
|
||||
else
|
||||
{
|
||||
prolongationOp = new ProlongationOperator(t_layout, v_layout, P);
|
||||
}
|
||||
}
|
||||
|
||||
void FiniteElementSpace::SetupKernels()
|
||||
{
|
||||
::occa::properties props("defines: {"
|
||||
" TILESIZE: 256,"
|
||||
"}");
|
||||
props["defines/NUM_VDIM"] = vdim;
|
||||
|
||||
props["defines/ORDERING_BY_NODES"] = 0;
|
||||
props["defines/ORDERING_BY_VDIM"] = 1;
|
||||
props["defines/VDIM_ORDERING"] = (int) (ordering == Ordering::byVDIM);
|
||||
|
||||
::occa::device device = GetDevice();
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
globalToLocalKernel = device.buildKernel(okl_path + "fespace.okl",
|
||||
"GlobalToLocal",
|
||||
props);
|
||||
localToGlobalKernel = device.buildKernel(okl_path + "fespace.okl",
|
||||
"LocalToGlobal",
|
||||
props);
|
||||
}
|
||||
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
OccaConformingProlongation::OccaConformingProlongation(
|
||||
const FiniteElementSpace &ofes, const mfem::ParFiniteElementSpace &pfes,
|
||||
::occa::memory ltdof_ldof_)
|
||||
|
||||
: Operator(ofes.OccaTrueVLayout(), ofes.OccaVLayout()),
|
||||
shr_ltdof(ofes.OccaEngine()),
|
||||
ext_ldof(ofes.OccaEngine()),
|
||||
shr_buf(shr_ltdof.OccaLayout(), sizeof(double)),
|
||||
ext_buf(ext_ldof.OccaLayout(), sizeof(double)),
|
||||
shr_buf_offsets(NULL), ext_buf_offsets(NULL),
|
||||
ltdof_ldof(ltdof_ldof_),
|
||||
gc(pfes.GroupComm())
|
||||
{
|
||||
MFEM_ASSERT(pfes.Conforming(), "internal error");
|
||||
|
||||
const Engine &engine = ofes.OccaEngine();
|
||||
const std::string &okl_path = engine.GetOklPath();
|
||||
::occa::device device = engine.GetDevice();
|
||||
|
||||
{
|
||||
Table nbr_ltdof;
|
||||
gc.GetNeighborLTDofTable(nbr_ltdof);
|
||||
shr_ltdof.OccaResize(nbr_ltdof.Size_of_connections(), sizeof(int));
|
||||
shr_ltdof.OccaPush(nbr_ltdof.GetJ());
|
||||
shr_buf.OccaResize(&shr_ltdof.OccaLayout(), sizeof(double));
|
||||
shr_buf_offsets = nbr_ltdof.GetI();
|
||||
{
|
||||
mfem::Array<int> shr_ltdof(nbr_ltdof.GetJ(),
|
||||
nbr_ltdof.Size_of_connections());
|
||||
mfem::Array<int> unique_ltdof(shr_ltdof);
|
||||
unique_ltdof.Sort();
|
||||
unique_ltdof.Unique();
|
||||
// Note: the next loop modifies the J array of nbr_ltdof
|
||||
for (int i = 0; i < shr_ltdof.Size(); i++)
|
||||
{
|
||||
shr_ltdof[i] = unique_ltdof.FindSorted(shr_ltdof[i]);
|
||||
MFEM_ASSERT(shr_ltdof[i] != -1, "internal error");
|
||||
}
|
||||
Table unique_shr;
|
||||
Transpose(shr_ltdof, unique_shr, unique_ltdof.Size());
|
||||
|
||||
unq_ltdof = device.malloc(unique_ltdof.Size()*sizeof(int),
|
||||
unique_ltdof.GetData());
|
||||
unq_shr_i = device.malloc((unique_shr.Size()+1)*sizeof(int),
|
||||
unique_shr.GetI());
|
||||
unq_shr_j = device.malloc(unique_shr.Size_of_connections()*sizeof(int),
|
||||
unique_shr.GetJ());
|
||||
}
|
||||
delete [] nbr_ltdof.GetJ();
|
||||
nbr_ltdof.LoseData();
|
||||
}
|
||||
{
|
||||
Table nbr_ldof;
|
||||
gc.GetNeighborLDofTable(nbr_ldof);
|
||||
ext_ldof.OccaResize(nbr_ldof.Size_of_connections(), sizeof(int));
|
||||
ext_ldof.OccaPush(nbr_ldof.GetJ());
|
||||
ext_buf.OccaResize(&ext_ldof.OccaLayout(), sizeof(double));
|
||||
ext_buf_offsets = nbr_ldof.GetI();
|
||||
delete [] nbr_ldof.GetJ();
|
||||
nbr_ldof.LoseData();
|
||||
}
|
||||
host_shr_buf = NULL;
|
||||
host_ext_buf = NULL;
|
||||
// If the device has a separate memory space (e.g. CUDA device) and the MPI
|
||||
// library does not support buffers in that separate memory space, we
|
||||
// allocate separate host buffers to use for MPI communication.
|
||||
if (device.hasSeparateMemorySpace())
|
||||
{
|
||||
bool need_host_buf = true;
|
||||
if (device.mode() == "CUDA")
|
||||
{
|
||||
#ifdef MPIX_CUDA_AWARE_SUPPORT
|
||||
need_host_buf = !MPIX_Query_cuda_support();
|
||||
#endif
|
||||
if (engine.GetForceCudaAwareMPI()) { need_host_buf = false; }
|
||||
if (gc.GetGroupTopology().MyRank() == 0)
|
||||
{
|
||||
mfem::out << "\nOccaConformingProlongation: CUDA-aware MPI: "
|
||||
<< (need_host_buf ? "NO" : "YES") << "\n\n";
|
||||
}
|
||||
}
|
||||
if (need_host_buf)
|
||||
{
|
||||
host_shr_buf = new char[shr_buf.OccaMem().size()];
|
||||
host_ext_buf = new char[ext_buf.OccaMem().size()];
|
||||
}
|
||||
}
|
||||
|
||||
ExtractSubVector = device.buildKernel(okl_path + "mappings.okl",
|
||||
"ExtractSubVector",
|
||||
"defines: { TILESIZE: 256 }");
|
||||
SetSubVector = device.buildKernel(okl_path + "mappings.okl",
|
||||
"SetSubVector",
|
||||
"defines: { TILESIZE: 256 }");
|
||||
AddSubVector = device.buildKernel(okl_path + "mappings.okl",
|
||||
"AddSubVector",
|
||||
"defines: { TILESIZE: 256 }");
|
||||
|
||||
const GroupTopology >opo = gc.GetGroupTopology();
|
||||
int req_counter = 0;
|
||||
for (int nbr = 1; nbr < gtopo.GetNumNeighbors(); nbr++)
|
||||
{
|
||||
const int send_offset = shr_buf_offsets[nbr];
|
||||
const int send_size = shr_buf_offsets[nbr+1] - send_offset;
|
||||
if (send_size > 0) { req_counter++; }
|
||||
|
||||
const int recv_offset = ext_buf_offsets[nbr];
|
||||
const int recv_size = ext_buf_offsets[nbr+1] - recv_offset;
|
||||
if (recv_size > 0) { req_counter++; }
|
||||
}
|
||||
requests = new MPI_Request[req_counter];
|
||||
}
|
||||
|
||||
OccaConformingProlongation::~OccaConformingProlongation()
|
||||
{
|
||||
delete [] requests;
|
||||
delete [] host_ext_buf;
|
||||
delete [] host_shr_buf;
|
||||
delete [] ext_buf_offsets;
|
||||
delete [] shr_buf_offsets;
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::BcastBeginCopy(const ::occa::memory &src,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// shr_buf[i] = src[shr_ltdof[i]]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (shr_ltdof.Size() == 0) { return; }
|
||||
ExtractSubVector((int)shr_ltdof.Size(), shr_ltdof.OccaMem(), src,
|
||||
shr_buf.OccaMem());
|
||||
// If the above kernel is executed asynchronously, wait for it to complete:
|
||||
shr_buf.OccaMem().getDevice().finish();
|
||||
if (host_shr_buf)
|
||||
{
|
||||
shr_buf.OccaMem().copyTo(host_shr_buf);
|
||||
}
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::BcastLocalCopy(const ::occa::memory &src,
|
||||
::occa::memory &dst,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// dst[ltdof_ldof[i]] = src[i]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (ltdof_ldof.size<int>() == 0) { return; }
|
||||
SetSubVector((int)ltdof_ldof.size<int>(), ltdof_ldof, src, dst);
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::BcastEndCopy(::occa::memory &dst,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// dst[ext_ldof[i]] = ext_buf[i]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (ext_ldof.Size() == 0) { return; }
|
||||
if (host_ext_buf)
|
||||
{
|
||||
ext_buf.OccaMem().copyFrom(host_ext_buf);
|
||||
}
|
||||
SetSubVector((int)ext_ldof.Size(), ext_ldof.OccaMem(),
|
||||
ext_buf.OccaMem(), dst);
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::ReduceBeginCopy(const ::occa::memory &src,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// ext_buf[i] = src[ext_ldof[i]]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (ext_ldof.Size() == 0) { return; }
|
||||
ExtractSubVector((int)ext_ldof.Size(), ext_ldof.OccaMem(), src,
|
||||
ext_buf.OccaMem());
|
||||
// If the above kernel is executed asynchronously, wait for it to complete:
|
||||
ext_buf.OccaMem().getDevice().finish();
|
||||
if (host_ext_buf)
|
||||
{
|
||||
ext_buf.OccaMem().copyTo(host_ext_buf);
|
||||
}
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::ReduceLocalCopy(const ::occa::memory &src,
|
||||
::occa::memory &dst,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// dst[i] = src[ltdof_ldof[i]]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (ltdof_ldof.size<int>() == 0) { return; }
|
||||
ExtractSubVector((int)ltdof_ldof.size<int>(), ltdof_ldof, src, dst);
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::ReduceEndAssemble(::occa::memory &dst,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// dst[shr_ltdof[i]] += shr_buf[i]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (unq_ltdof.size<int>() == 0) { return; }
|
||||
if (host_shr_buf)
|
||||
{
|
||||
shr_buf.OccaMem().copyFrom(host_shr_buf);
|
||||
}
|
||||
AddSubVector((int)unq_ltdof.size<int>(), unq_ltdof, unq_shr_i, unq_shr_j,
|
||||
shr_buf.OccaMem(), dst);
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
const GroupTopology >opo = gc.GetGroupTopology();
|
||||
|
||||
BcastBeginCopy(x.OccaMem(), sizeof(double)); // copy to 'shr_buf'
|
||||
|
||||
int req_counter = 0;
|
||||
for (int nbr = 1; nbr < gtopo.GetNumNeighbors(); nbr++)
|
||||
{
|
||||
const int send_offset = shr_buf_offsets[nbr];
|
||||
const int send_size = shr_buf_offsets[nbr+1] - send_offset;
|
||||
if (send_size > 0)
|
||||
{
|
||||
void *send_buf;
|
||||
if (host_shr_buf)
|
||||
{
|
||||
send_buf = host_shr_buf + send_offset*sizeof(double);
|
||||
}
|
||||
else
|
||||
{
|
||||
send_buf = (shr_buf.OccaMem() + send_offset*sizeof(double)).ptr();
|
||||
}
|
||||
MPI_Isend(send_buf, send_size, MPI_DOUBLE, gtopo.GetNeighborRank(nbr),
|
||||
41822, gtopo.GetComm(), &requests[req_counter++]);
|
||||
}
|
||||
|
||||
const int recv_offset = ext_buf_offsets[nbr];
|
||||
const int recv_size = ext_buf_offsets[nbr+1] - recv_offset;
|
||||
if (recv_size > 0)
|
||||
{
|
||||
void *recv_buf;
|
||||
if (host_ext_buf)
|
||||
{
|
||||
recv_buf = host_ext_buf + recv_offset*sizeof(double);
|
||||
}
|
||||
else
|
||||
{
|
||||
recv_buf = (ext_buf.OccaMem() + recv_offset*sizeof(double)).ptr();
|
||||
}
|
||||
MPI_Irecv(recv_buf, recv_size, MPI_DOUBLE, gtopo.GetNeighborRank(nbr),
|
||||
41822, gtopo.GetComm(), &requests[req_counter++]);
|
||||
}
|
||||
}
|
||||
|
||||
BcastLocalCopy(x.OccaMem(), y.OccaMem(), sizeof(double));
|
||||
|
||||
MPI_Waitall(req_counter, requests, MPI_STATUSES_IGNORE);
|
||||
|
||||
BcastEndCopy(y.OccaMem(), sizeof(double)); // copy from 'ext_buf'
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::MultTranspose_(const Vector &x,
|
||||
Vector &y) const
|
||||
{
|
||||
const GroupTopology >opo = gc.GetGroupTopology();
|
||||
|
||||
ReduceBeginCopy(x.OccaMem(), sizeof(double)); // copy to 'ext_buf'
|
||||
|
||||
int req_counter = 0;
|
||||
for (int nbr = 1; nbr < gtopo.GetNumNeighbors(); nbr++)
|
||||
{
|
||||
const int send_offset = ext_buf_offsets[nbr];
|
||||
const int send_size = ext_buf_offsets[nbr+1] - send_offset;
|
||||
if (send_size > 0)
|
||||
{
|
||||
void *send_buf;
|
||||
if (host_ext_buf)
|
||||
{
|
||||
send_buf = host_ext_buf + send_offset*sizeof(double);
|
||||
}
|
||||
else
|
||||
{
|
||||
send_buf = (ext_buf.OccaMem() + send_offset*sizeof(double)).ptr();
|
||||
}
|
||||
MPI_Isend(send_buf, send_size, MPI_DOUBLE, gtopo.GetNeighborRank(nbr),
|
||||
41823, gtopo.GetComm(), &requests[req_counter++]);
|
||||
}
|
||||
|
||||
const int recv_offset = shr_buf_offsets[nbr];
|
||||
const int recv_size = shr_buf_offsets[nbr+1] - recv_offset;
|
||||
if (recv_size > 0)
|
||||
{
|
||||
void *recv_buf;
|
||||
if (host_shr_buf)
|
||||
{
|
||||
recv_buf = host_shr_buf + recv_offset*sizeof(double);
|
||||
}
|
||||
else
|
||||
{
|
||||
recv_buf = (shr_buf.OccaMem() + recv_offset*sizeof(double)).ptr();
|
||||
}
|
||||
MPI_Irecv(recv_buf, recv_size, MPI_DOUBLE, gtopo.GetNeighborRank(nbr),
|
||||
41823, gtopo.GetComm(), &requests[req_counter++]);
|
||||
}
|
||||
}
|
||||
|
||||
ReduceLocalCopy(x.OccaMem(), y.OccaMem(), sizeof(double));
|
||||
|
||||
MPI_Waitall(req_counter, requests, MPI_STATUSES_IGNORE);
|
||||
|
||||
ReduceEndAssemble(y.OccaMem(), sizeof(double)); // assemble from 'shr_buf'
|
||||
}
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,210 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_FE_SPACE_HPP
|
||||
#define MFEM_BACKENDS_OCCA_FE_SPACE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "operator.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
/// TODO: doxygen
|
||||
class FiniteElementSpace : public mfem::PFiniteElementSpace
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// mfem::FiniteElementSpace *fes;
|
||||
|
||||
SharedPtr<Layout> e_layout;
|
||||
|
||||
::occa::array<int> globalToLocalOffsets;
|
||||
::occa::array<int> globalToLocalIndices;
|
||||
::occa::array<int> localToGlobalMap;
|
||||
::occa::kernel globalToLocalKernel, localToGlobalKernel;
|
||||
|
||||
mfem::Ordering::Type ordering;
|
||||
|
||||
int globalDofs, localDofs;
|
||||
int vdim;
|
||||
|
||||
mutable Operator *prolongationOp, *restrictionOp;
|
||||
|
||||
void SetupLocalGlobalMaps();
|
||||
void SetupOperators() const; // calls virtual methods of 'fes' !!!
|
||||
void SetupKernels();
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
FiniteElementSpace(const Engine &e, mfem::FiniteElementSpace &fespace);
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~FiniteElementSpace();
|
||||
|
||||
/// TODO: doxygen
|
||||
const Engine &OccaEngine() const { return engine->As<Engine>(); }
|
||||
|
||||
/// TODO: doxygen
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return OccaEngine().GetDevice(idx); }
|
||||
|
||||
mfem::Mesh* GetMesh() const { return fes->GetMesh(); }
|
||||
|
||||
Layout &OccaVLayout() const
|
||||
{ return *fes->GetVLayout().As<Layout>(); }
|
||||
|
||||
Layout &OccaTrueVLayout() const
|
||||
{ return *fes->GetTrueVLayout().As<Layout>(); }
|
||||
|
||||
Layout &OccaEVLayout() { return *e_layout; }
|
||||
|
||||
bool hasTensorBasis() const
|
||||
{ return dynamic_cast<const mfem::TensorBasisElement*>(fes->GetFE(0)); }
|
||||
|
||||
mfem::Ordering::Type GetOrdering() const { return ordering; }
|
||||
|
||||
int GetGlobalDofs() const { return globalDofs; }
|
||||
int GetLocalDofs() const { return localDofs; }
|
||||
|
||||
int GetDim() const { return fes->GetMesh()->Dimension(); }
|
||||
int GetVDim() const { return vdim; }
|
||||
|
||||
int GetVSize() const { return globalDofs * vdim; }
|
||||
int GetTrueVSize() const { return fes->GetTrueVSize(); }
|
||||
int GetGlobalVSize() const { return globalDofs*vdim; /* FIXME: MPI */ }
|
||||
int GetGlobalTrueVSize() const { return fes->GetTrueVSize(); }
|
||||
|
||||
int GetNE() const { return fes->GetNE(); }
|
||||
|
||||
const mfem::FiniteElementCollection *FEColl() const
|
||||
{ return fes->FEColl(); }
|
||||
const mfem::FiniteElement *GetFE(const int idx) const
|
||||
{ return fes->GetFE(idx); }
|
||||
|
||||
virtual const mfem::Operator *GetProlongationOperator() const
|
||||
{ return prolongationOp; }
|
||||
|
||||
virtual const mfem::Operator *GetRestrictionOperator() const
|
||||
{ return restrictionOp; }
|
||||
|
||||
virtual const mfem::Operator *GetInterpolationOperator(
|
||||
const mfem::QuadratureSpace &qspace) const
|
||||
{ return NULL; /* FIXME */ }
|
||||
|
||||
virtual const mfem::Operator *GetGradientOperator(
|
||||
const mfem::QuadratureSpace &qspace) const
|
||||
{ return NULL; /* FIXME */ }
|
||||
|
||||
const ::occa::array<int> GetLocalToGlobalMap() const
|
||||
{ return localToGlobalMap; }
|
||||
|
||||
/// L-vector to E-vector
|
||||
void GlobalToLocal(const Vector &globalVec, Vector &localVec) const
|
||||
{
|
||||
globalToLocalKernel(globalDofs,
|
||||
localDofs * fes->GetNE(),
|
||||
globalToLocalOffsets,
|
||||
globalToLocalIndices,
|
||||
globalVec.OccaMem(), localVec.OccaMem());
|
||||
}
|
||||
|
||||
/// E-vector to L-vector, transpose of GlobalToLocal
|
||||
void LocalToGlobal(const Vector &localVec, Vector &globalVec) const
|
||||
{
|
||||
localToGlobalKernel(globalDofs,
|
||||
localDofs * fes->GetNE(),
|
||||
globalToLocalOffsets,
|
||||
globalToLocalIndices,
|
||||
localVec.OccaMem(), globalVec.OccaMem());
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
/// OCCA version of mfem::ConformingProlongationOperator
|
||||
class OccaConformingProlongation : public Operator
|
||||
{
|
||||
protected:
|
||||
// size(shr_buf)=size(shr_ltdof)
|
||||
// size(ext_buf)=size(ext_ldof)
|
||||
Array shr_ltdof, ext_ldof;
|
||||
mutable Array shr_buf, ext_buf;
|
||||
mutable char *host_shr_buf, *host_ext_buf;
|
||||
// Offsets into {shr,ext}_buf; size is num. neighbors, i.e.
|
||||
// gc.GetGroupTopology().GetNumNeighbors():
|
||||
int *shr_buf_offsets, *ext_buf_offsets;
|
||||
|
||||
::occa::memory ltdof_ldof; // shared with the restriction operator
|
||||
|
||||
::occa::memory unq_ltdof; // enumeration of the unique ltdofs in shr_ltdof
|
||||
::occa::memory unq_shr_i, unq_shr_j;
|
||||
|
||||
::occa::kernel ExtractSubVector, SetSubVector, AddSubVector;
|
||||
|
||||
MPI_Request *requests;
|
||||
|
||||
const GroupCommunicator &gc;
|
||||
|
||||
// Kernel: copy ltdofs from 'src' to 'shr_buf' - prepare for send.
|
||||
// shr_buf[i] = src[shr_ltdof[i]]
|
||||
void BcastBeginCopy(const ::occa::memory &src, std::size_t item_size) const;
|
||||
// Kernel: copy ltdofs from 'src' to ldofs in 'dst'.
|
||||
// dst[ltdof_ldof[i]] = src[i]
|
||||
void BcastLocalCopy(const ::occa::memory &src, ::occa::memory &dst,
|
||||
std::size_t item_size) const;
|
||||
// Kernel: copy ext. dofs from 'ext_buf' to 'dst' - after recv.
|
||||
// dst[ext_ldof[i]] = ext_buf[i]
|
||||
void BcastEndCopy(::occa::memory &dst, std::size_t item_size) const;
|
||||
|
||||
// Kernel: copy ext. dofs from 'src' to 'ext_buf' - prepare for send.
|
||||
// ext_buf[i] = src[ext_ldof[i]]
|
||||
void ReduceBeginCopy(const ::occa::memory &src, std::size_t item_size) const;
|
||||
// Kernel: copy owned ldofs from 'src' to ltdofs in 'dst'.
|
||||
// dst[i] = src[ltdof_ldof[i]]
|
||||
void ReduceLocalCopy(const ::occa::memory &src, ::occa::memory &dst,
|
||||
std::size_t item_size) const;
|
||||
// Kernel: assemble dofs from 'shr_buf' into to 'dst' - after recv.
|
||||
// dst[shr_ltdof[i]] += shr_buf[i]
|
||||
void ReduceEndAssemble(::occa::memory &dst, std::size_t item_size) const;
|
||||
|
||||
public:
|
||||
OccaConformingProlongation(const FiniteElementSpace &ofes,
|
||||
const mfem::ParFiniteElementSpace &pfes,
|
||||
::occa::memory ltdof_ldof_);
|
||||
|
||||
virtual ~OccaConformingProlongation();
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_FE_SPACE_HPP
|
||||
@@ -1,67 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
TILESIZE : Tilesize for iterating over entries
|
||||
================================================
|
||||
*/
|
||||
|
||||
#if VDIM_ORDERING == ORDERING_BY_VDIM
|
||||
typedef double *Global_t @dim(NUM_VDIM, globalEntries);
|
||||
typedef double *Local_t @dim(NUM_VDIM, localEntries);
|
||||
#else
|
||||
typedef double *Global_t @dim(NUM_VDIM, globalEntries) @dimOrder(1, 0);
|
||||
typedef double *Local_t @dim(NUM_VDIM, localEntries) @dimOrder(1, 0);
|
||||
#endif
|
||||
|
||||
@kernel void GlobalToLocal(const int globalEntries,
|
||||
const int localEntries,
|
||||
@restrict const int * offsets,
|
||||
@restrict const int * indices,
|
||||
@restrict const Global_t globalX,
|
||||
@restrict Local_t localX) {
|
||||
|
||||
for (int i = 0; i < globalEntries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < globalEntries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double dofValue = globalX(v, i);
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
localX(v, indices[j]) = dofValue;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void LocalToGlobal(const int globalEntries,
|
||||
const int localEntries,
|
||||
@restrict const int * offsets,
|
||||
@restrict const int * indices,
|
||||
@restrict const Local_t localX,
|
||||
@restrict Global_t globalX) {
|
||||
|
||||
for (int i = 0; i < globalEntries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < globalEntries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
double dofValue = 0;
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
dofValue += localX(v, indices[j]);
|
||||
}
|
||||
globalX(v, i) = dofValue;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,181 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef STORE_JACOBIAN
|
||||
# define STORE_JACOBIAN 1
|
||||
#endif
|
||||
|
||||
#ifndef STORE_JACOBIAN_INV
|
||||
# define STORE_JACOBIAN_INV 1
|
||||
#endif
|
||||
|
||||
#ifndef STORE_JACOBIAN_DET
|
||||
# define STORE_JACOBIAN_DET 1
|
||||
#endif
|
||||
|
||||
typedef double* Local1D_t @dim(1, NUM_DOFS, numElements);
|
||||
typedef double* Local2D_t @dim(2, NUM_DOFS, numElements);
|
||||
typedef double* Local3D_t @dim(3, NUM_DOFS, numElements);
|
||||
|
||||
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
|
||||
|
||||
typedef double* DofToQuadD1D_t @dim(NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD2D_t @dim(2, NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD3D_t @dim(3, NUM_QUAD, NUM_DOFS);
|
||||
|
||||
typedef double* Jacobian1D_t @dim(NUM_QUAD, numElements);
|
||||
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD, numElements);
|
||||
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD, numElements);
|
||||
|
||||
@kernel void InitGeometryInfo1D(const int numElements,
|
||||
@restrict const DofToQuadD1D_t dofToQuadD,
|
||||
@restrict const Local1D_t nodes,
|
||||
@restrict Jacobian1D_t J,
|
||||
@restrict Jacobian1D_t invJ,
|
||||
@restrict QLocal_t detJ) {
|
||||
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_nodes[NUM_DOFS];
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
|
||||
s_nodes[d] = nodes(0, d, e);
|
||||
}
|
||||
}
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
double J11 = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double wx = dofToQuadD(q, d);
|
||||
J11 += wx * s_nodes[d];
|
||||
}
|
||||
#if STORE_JACOBIAN
|
||||
J(q, e) = J11;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_INV
|
||||
invJ(q, e) = 1.0 / J11;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_DET
|
||||
detJ(q, e) = J11;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void InitGeometryInfo2D(const int numElements,
|
||||
@restrict const DofToQuadD2D_t dofToQuadD,
|
||||
@restrict const Local2D_t nodes,
|
||||
@restrict Jacobian2D_t J,
|
||||
@restrict Jacobian2D_t invJ,
|
||||
@restrict QLocal_t detJ) {
|
||||
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_nodes[2 * NUM_DOFS] @dim(2, NUM_DOFS);
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
|
||||
s_nodes(0, d) = nodes(0, d, e);
|
||||
s_nodes(1, d) = nodes(1, d, e);
|
||||
}
|
||||
}
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
double J11 = 0, J12 = 0;
|
||||
double J21 = 0, J22 = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double wx = dofToQuadD(0, q, d);
|
||||
const double wy = dofToQuadD(1, q, d);
|
||||
const double x = s_nodes(0, d);
|
||||
const double y = s_nodes(1, d);
|
||||
J11 += (wx * x); J12 += (wx * y);
|
||||
J21 += (wy * x); J22 += (wy * y);
|
||||
}
|
||||
#if STORE_JACOBIAN_INV || STORE_JACOBIAN_DET
|
||||
const double r_detJ = (J11 * J22) - (J12 * J21);
|
||||
#endif
|
||||
#if STORE_JACOBIAN
|
||||
J(0, 0, q, e) = J11; J(1, 0, q, e) = J12;
|
||||
J(0, 1, q, e) = J21; J(1, 1, q, e) = J22;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_INV
|
||||
const double r_idetJ = 1.0 / r_detJ;
|
||||
invJ(0, 0, q, e) = J22 * r_idetJ;
|
||||
invJ(1, 0, q, e) = -J12 * r_idetJ;
|
||||
|
||||
invJ(0, 1, q, e) = -J21 * r_idetJ;
|
||||
invJ(1, 1, q, e) = J11 * r_idetJ;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_DET
|
||||
detJ(q, e) = r_detJ;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void InitGeometryInfo3D(const int numElements,
|
||||
@restrict const DofToQuadD3D_t dofToQuadD,
|
||||
@restrict const Local3D_t nodes,
|
||||
@restrict Jacobian3D_t J,
|
||||
@restrict Jacobian3D_t invJ,
|
||||
@restrict QLocal_t detJ) {
|
||||
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_nodes[3 * NUM_DOFS] @dim(3, NUM_DOFS);
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
|
||||
s_nodes(0, d) = nodes(0, d, e);
|
||||
s_nodes(1, d) = nodes(1, d, e);
|
||||
s_nodes(2, d) = nodes(2, d, e);
|
||||
}
|
||||
}
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
double J11 = 0, J12 = 0, J13 = 0;
|
||||
double J21 = 0, J22 = 0, J23 = 0;
|
||||
double J31 = 0, J32 = 0, J33 = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double wx = dofToQuadD(0, q, d);
|
||||
const double wy = dofToQuadD(1, q, d);
|
||||
const double wz = dofToQuadD(2, q, d);
|
||||
const double x = s_nodes(0, d);
|
||||
const double y = s_nodes(1, d);
|
||||
const double z = s_nodes(2, d);
|
||||
J11 += (wx * x); J12 += (wx * y); J13 += (wx * z);
|
||||
J21 += (wy * x); J22 += (wy * y); J23 += (wy * z);
|
||||
J31 += (wz * x); J32 += (wz * y); J33 += (wz * z);
|
||||
}
|
||||
#if STORE_JACOBIAN_INV || STORE_JACOBIAN_DET
|
||||
const double r_detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
#endif
|
||||
#if STORE_JACOBIAN
|
||||
J(0, 0, q, e) = J11; J(1, 0, q, e) = J12; J(2, 0, q, e) = J13;
|
||||
J(0, 1, q, e) = J21; J(1, 1, q, e) = J22; J(2, 1, q, e) = J23;
|
||||
J(0, 2, q, e) = J31; J(1, 2, q, e) = J32; J(2, 2, q, e) = J33;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_INV
|
||||
const double r_idetJ = 1.0 / r_detJ;
|
||||
invJ(0, 0, q, e) = r_idetJ * ((J22 * J33) - (J23 * J32));
|
||||
invJ(1, 0, q, e) = r_idetJ * ((J32 * J13) - (J33 * J12));
|
||||
invJ(2, 0, q, e) = r_idetJ * ((J12 * J23) - (J13 * J22));
|
||||
|
||||
invJ(0, 1, q, e) = r_idetJ * ((J23 * J31) - (J21 * J33));
|
||||
invJ(1, 1, q, e) = r_idetJ * ((J33 * J11) - (J31 * J13));
|
||||
invJ(2, 1, q, e) = r_idetJ * ((J13 * J21) - (J11 * J23));
|
||||
|
||||
invJ(0, 2, q, e) = r_idetJ * ((J21 * J32) - (J22 * J31));
|
||||
invJ(1, 2, q, e) = r_idetJ * ((J31 * J12) - (J32 * J11));
|
||||
invJ(2, 2, q, e) = r_idetJ * ((J11 * J22) - (J12 * J21));
|
||||
#endif
|
||||
#if STORE_JACOBIAN_DET
|
||||
detJ(q, e) = r_detJ;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,86 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "gridfunc.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "../../fem/gridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
std::map<std::string, ::occa::kernel> gridFunctionKernels;
|
||||
|
||||
::occa::kernel GetGridFunctionKernel(::occa::device device,
|
||||
FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir)
|
||||
{
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
const FiniteElement &fe = *(fespace.GetFE(0));
|
||||
const int dim = fe.GetDim();
|
||||
const int vdim = fespace.GetVDim();
|
||||
|
||||
std::stringstream ss;
|
||||
ss << ::occa::hash(device)
|
||||
<< "FEColl : " << fespace.FEColl()->Name()
|
||||
<< "Quad: " << numQuad
|
||||
<< "Dim: " << dim
|
||||
<< "VDim: " << vdim;
|
||||
std::string hash = ss.str();
|
||||
|
||||
// Kernel defines
|
||||
::occa::properties props;
|
||||
props["defines/NUM_VDIM"] = vdim;
|
||||
|
||||
SetProperties(fespace, ir, props);
|
||||
|
||||
::occa::kernel kernel = gridFunctionKernels[hash];
|
||||
if (!kernel.isInitialized())
|
||||
{
|
||||
const std::string &okl_path = fespace.OccaEngine().GetOklPath();
|
||||
kernel = device.buildKernel(okl_path + "gridfunc.okl",
|
||||
stringWithDim("GridFuncToQuad", dim),
|
||||
props);
|
||||
}
|
||||
return kernel;
|
||||
}
|
||||
|
||||
void ToQuad(const IntegrationRule &ir, FiniteElementSpace &fespace, Vector &gf,
|
||||
Vector &quadValues)
|
||||
{
|
||||
const Engine &engine = fespace.OccaEngine();
|
||||
::occa::device device = engine.GetDevice();
|
||||
|
||||
OccaDofQuadMaps &maps = OccaDofQuadMaps::Get(device, fespace, ir);
|
||||
|
||||
const int elements = fespace.GetNE();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
quadValues.OccaResize(numQuad * elements, sizeof(double));
|
||||
|
||||
::occa::kernel g2qKernel = GetGridFunctionKernel(device, fespace, ir);
|
||||
g2qKernel(elements,
|
||||
maps.dofToQuad,
|
||||
fespace.GetLocalToGlobalMap(),
|
||||
gf.OccaMem(),
|
||||
quadValues.OccaMem());
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,55 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
|
||||
#define MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "fespace.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class IntegrationRule;
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
// TODO: make this object part of the backend or the engine.
|
||||
extern std::map<std::string, ::occa::kernel> gridFunctionKernels;
|
||||
|
||||
// TODO: make this a method of the backend or the engine.
|
||||
::occa::kernel GetGridFunctionKernel(::occa::device device,
|
||||
FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir);
|
||||
|
||||
// ToQuad version without the deprecated class.
|
||||
//
|
||||
// FIXME: This is the action of a global B matrix, mapping L-vector to Q-vector,
|
||||
// so it should be made into an operator that can be constructed by the
|
||||
// FE space class. A batched version, where only a subset of the elements
|
||||
// are processed should be defined as well.
|
||||
//
|
||||
// The abstract operator construction method in the FE space class is:
|
||||
// PFiniteElementSpace::GetInterpolationOperator(...)
|
||||
void ToQuad(const IntegrationRule &ir, FiniteElementSpace &ofespace, Vector &gf,
|
||||
Vector &quadValues);
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
|
||||
@@ -1,26 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
#ifdef USING_TENSOR_OPS
|
||||
# ifdef OCCA_USING_CPU
|
||||
# include "mfem-occa://gridfunc/tensor/cpu.okl"
|
||||
# else
|
||||
# include "mfem-occa://gridfunc/tensor/gpuHighOrder.okl"
|
||||
# endif
|
||||
#else
|
||||
# ifdef OCCA_USING_CPU
|
||||
# include "mfem-occa://gridfunc/simplex/cpu.okl"
|
||||
# else
|
||||
# include "mfem-occa://gridfunc/simplex/gpuHighOrder.okl"
|
||||
# endif
|
||||
#endif
|
||||
@@ -1,63 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal_t out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
double r_out = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_out += r_gf * dofToQuad(d, q);
|
||||
}
|
||||
out(v, d, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal_t out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
double r_out = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_out += r_gf * dofToQuad(d, q);
|
||||
}
|
||||
out(v, d, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,79 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal_t out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gf[NUM_VDIM][NUM_DOFS];
|
||||
|
||||
for (int dOff = 0; dOff < M2_INNER_BATCH; ++dOff; @inner) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M2_INNER_BATCH) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
s_gf[v][d] = gf[v + gid*NUM_VDIM]];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qOff = 0; qOff < M2_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M2_INNER_BATCH) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
double r_out = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_out += s_gf[v][d] * dofToQuad(d, q);
|
||||
}
|
||||
out(v, q, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal_t out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gf[NUM_VDIM][NUM_DOFS];
|
||||
|
||||
for (int dOff = 0; dOff < M3_INNER_BATCH; ++dOff; @inner) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M3_INNER_BATCH) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
s_gf[v][d] = gf[v + gid*NUM_VDIM]];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qOff = 0; qOff < M3_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M3_INNER_BATCH) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
double r_out = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_out += s_gf[v][d] * dofToQuad(d, q);
|
||||
}
|
||||
out(v, q, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,188 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void GridFuncToQuad1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap1D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal1D_t out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_out[NUM_VDIM][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[v][qx] = 0;
|
||||
}
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const int gid = l2gMap(dx, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[v][qx] += r_gf * dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
out(v, qx, e) = r_out[v][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap2D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal2D_t out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double out_xy[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double out_x[NUM_VDIM][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
out_x[v][qy] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const int gid = l2gMap(dx, dy, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
out_x[v][qy] += r_gf * dofToQuad(qy, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double d2q = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] += d2q * out_x[v][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
out(v, qx, qy, e) = out_xy[v][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap3D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal3D_t out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double out_xyz[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xyz[v][qz][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
double out_xy[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double out_x[NUM_VDIM][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_x[v][qx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const int gid = l2gMap(dx, dy, dz, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_x[v][qx] += r_gf * dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] += wy * out_x[v][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
const double wz = dofToQuad(qz, dz);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xyz[v][qz][qy][qx] += wz * out_xy[v][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
out(v, qx, qy, qz, e) = out_xyz[v][qz][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,183 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void GridFuncToQuad1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap1D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QLocal1D_t out) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M1_ELEMENT_BATCHES; @outer) {
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@exclusive double r_out[NUM_QUAD_1D];
|
||||
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
for (int i = el; i < NUM_QUAD_DOFS_1D; i += M1_INNER_ELEMENT_BATCH) {
|
||||
s_dofToQuad[i] = dofToQuad[i];
|
||||
}
|
||||
}
|
||||
|
||||
for (int b = 0; b < M1_OUTER_ELEMENT_BATCH; ++b) {
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
const int e = eOff + b*M1_INNER_ELEMENT_BATCH + el;
|
||||
if (e < numElements) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double r_gf = gf[l2gMap(dx, e)];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[qx] += r_gf * s_dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out(qx, e) = r_out[qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap2D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QLocal2D_t out) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M2_ELEMENT_BATCH; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_xy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
for (int id = x; id < NUM_QUAD_DOFS_1D; id += NUM_MAX_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
}
|
||||
}
|
||||
|
||||
for (int e = eOff; e < (eOff + M2_ELEMENT_BATCH); ++e) {
|
||||
if (e < numElements) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
double r_x[NUM_DOFS_1D];
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
r_x[dy] = gf[l2gMap(dx, dy, e)];
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double xy = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
xy += r_x[dy] * s_dofToQuad(qy, dy);
|
||||
}
|
||||
s_xy(dx, qy) = xy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
if (qy < NUM_QUAD_1D) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
double val = 0;
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
val += s_xy(dx, qy) * s_dofToQuad(qx, dx);
|
||||
}
|
||||
out(qx, qy, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap3D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QLocal3D_t out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_z[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
|
||||
// Store z axis as registers
|
||||
@exclusive double r_qz[NUM_QUAD_1D];
|
||||
|
||||
for (int y = 0; y < NUM_MAX_1D; ++y; @inner) {
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
const int id = (y * NUM_MAX_1D) + x;
|
||||
// Fetch Q <--> D maps
|
||||
if (id < NUM_QUAD_DOFS_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
}
|
||||
// Initialize our Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double val = gf[l2gMap(dx, dy, dz, e)];
|
||||
// Calculate D -> Q in the Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] += val * s_dofToQuad(qz, dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// For each xy plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
// Fill xy plane at given z position
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
s_z(dx, dy) = r_qz[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
double val = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = s_dofToQuad(qy, dy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = s_dofToQuad(qx, dx);
|
||||
val += wx * wy * s_z(dx, dy);
|
||||
}
|
||||
}
|
||||
out(qx, qy, qz, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,119 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "interpolation.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
RestrictionOperator::RestrictionOperator(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> indices)
|
||||
: Operator(in_layout, out_layout)
|
||||
{
|
||||
trueIndices = indices;
|
||||
|
||||
::occa::device device = in_layout.OccaEngine().GetDevice();
|
||||
const std::string &okl_path = in_layout.OccaEngine().GetOklPath();
|
||||
multOp = device.buildKernel(okl_path + "mappings.okl",
|
||||
"ExtractSubVector",
|
||||
"defines: { TILESIZE: 256 }");
|
||||
|
||||
multTransposeOp = device.buildKernel(okl_path + "mappings.okl",
|
||||
"SetSubVector",
|
||||
"defines: { TILESIZE: 256 }");
|
||||
}
|
||||
|
||||
void RestrictionOperator::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
// y[i] = x[trueIndices[i]]
|
||||
multOp(height, trueIndices, x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
|
||||
void RestrictionOperator::MultTranspose_(const Vector &x, Vector &y) const
|
||||
{
|
||||
y.OccaFill<double>(0.0);
|
||||
// y[trueIndices[i]] = x[i]
|
||||
multTransposeOp(height, trueIndices, x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
|
||||
ProlongationOperator::ProlongationOperator(OccaSparseMatrix &multOp_,
|
||||
OccaSparseMatrix &multTransposeOp_)
|
||||
: Operator(multOp_),
|
||||
pmat(NULL),
|
||||
multOp(multOp_),
|
||||
multTransposeOp(multTransposeOp_)
|
||||
{ }
|
||||
|
||||
ProlongationOperator::ProlongationOperator(Layout &in_layout,
|
||||
Layout &out_layout,
|
||||
const mfem::Operator *pmat_)
|
||||
: Operator(in_layout, out_layout),
|
||||
pmat(pmat_),
|
||||
multOp(*this),
|
||||
multTransposeOp(*this)
|
||||
{ }
|
||||
|
||||
void ProlongationOperator::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
MFEM_VERIFY(pmat == NULL, "");
|
||||
multOp.Mult_(x, y);
|
||||
}
|
||||
|
||||
void ProlongationOperator::MultTranspose_(const Vector &x, Vector &y) const
|
||||
{
|
||||
MFEM_VERIFY(pmat == NULL, "");
|
||||
multTransposeOp.Mult_(x, y);
|
||||
}
|
||||
|
||||
void ProlongationOperator::Mult(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{
|
||||
if (pmat)
|
||||
{
|
||||
// FIXME: create an OCCA version of 'pmat'
|
||||
x.Pull();
|
||||
y.Pull(false);
|
||||
pmat->Mult(x, y);
|
||||
y.Push();
|
||||
}
|
||||
else
|
||||
{
|
||||
multOp.Mult(x, y);
|
||||
}
|
||||
}
|
||||
|
||||
void ProlongationOperator::MultTranspose(const mfem::Vector &x,
|
||||
mfem::Vector &y) const
|
||||
{
|
||||
if (pmat)
|
||||
{
|
||||
// FIXME: create an OCCA version of 'pmat'
|
||||
x.Pull();
|
||||
y.Pull(false);
|
||||
pmat->MultTranspose(x, y);
|
||||
y.Push();
|
||||
}
|
||||
else
|
||||
{
|
||||
multTransposeOp.Mult(x, y);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,73 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
|
||||
#define MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "vector.hpp"
|
||||
#include "engine.hpp"
|
||||
#include "sparsemat.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class RestrictionOperator : public Operator
|
||||
{
|
||||
protected:
|
||||
::occa::array<int> trueIndices; // ldof = trueIndices[ltdof]
|
||||
::occa::kernel multOp, multTransposeOp;
|
||||
|
||||
public:
|
||||
RestrictionOperator(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> indices);
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
class ProlongationOperator : public Operator
|
||||
{
|
||||
protected:
|
||||
const mfem::Operator *pmat;
|
||||
OccaSparseMatrix multOp, multTransposeOp;
|
||||
|
||||
public:
|
||||
ProlongationOperator(OccaSparseMatrix &multOp_,
|
||||
OccaSparseMatrix &multTransposeOp_);
|
||||
|
||||
ProlongationOperator(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::Operator *pmat_);
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
|
||||
// overrides
|
||||
virtual void Mult(const mfem::Vector &x, mfem::Vector &y) const;
|
||||
virtual void MultTranspose(const mfem::Vector &x, mfem::Vector &y) const;
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
|
||||
@@ -1,40 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "layout.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
void Layout::Resize(std::size_t new_size)
|
||||
{
|
||||
size = new_size;
|
||||
}
|
||||
|
||||
void Layout::Resize(const Array<std::size_t> &offsets)
|
||||
{
|
||||
MFEM_ASSERT(offsets.Size() == 2,
|
||||
"multiple workers are not supported yet");
|
||||
size = offsets.Last();
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,67 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_LAYOUT_HPP
|
||||
#define MFEM_BACKENDS_OCCA_LAYOUT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "../base/layout.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Layout : public PLayout
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// std::size_t size;
|
||||
|
||||
public:
|
||||
Layout(const Engine &e, std::size_t s = 0) : PLayout(e, s) { }
|
||||
|
||||
const Engine &OccaEngine() const
|
||||
{ return *static_cast<const Engine *>(engine.Get()); }
|
||||
|
||||
void OccaResize(std::size_t new_size) { size = new_size; }
|
||||
|
||||
virtual ~Layout() { }
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// Resize the layout
|
||||
virtual void Resize(std::size_t new_size);
|
||||
|
||||
/// Resize the layout based on the given worker offsets
|
||||
virtual void Resize(const Array<std::size_t> &offsets);
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_LAYOUT_HPP
|
||||
@@ -1,76 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
TILESIZE : Tilesize for iterating over entries
|
||||
================================================
|
||||
*/
|
||||
|
||||
@kernel void ExtractSubVector(const int entries,
|
||||
@restrict const int *indices,
|
||||
@restrict const double *in,
|
||||
@restrict double *out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
out[i] = in[indices[i]]; // indices can be repeated
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void SetSubVector(const int entries,
|
||||
@restrict const int *indices,
|
||||
@restrict const double *in,
|
||||
@restrict double *out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
out[indices[i]] = in[i]; // indices CANNOT be repeated
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void AddSubVector(const int num_unique_dst_indices,
|
||||
@restrict const int *unique_dst_indices,
|
||||
@restrict const int *unique_to_src_offsets,
|
||||
@restrict const int *unique_to_src_indices,
|
||||
@restrict const double *src,
|
||||
@restrict double *dst) {
|
||||
|
||||
for (int i = 0; i < num_unique_dst_indices; ++i;
|
||||
@tile(TILESIZE, @outer, @inner)) {
|
||||
|
||||
if (i < num_unique_dst_indices) {
|
||||
const int dst_idx = unique_dst_indices[i];
|
||||
double sum = dst[dst_idx];
|
||||
const int end = unique_to_src_offsets[i+1];
|
||||
for (int j = unique_to_src_offsets[i]; j != end; ++j) {
|
||||
sum += src[unique_to_src_indices[j]];
|
||||
}
|
||||
dst[dst_idx] = sum;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MapSubVector(const int entries,
|
||||
@restrict const int *indices,
|
||||
@restrict const double *in,
|
||||
@restrict double *out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
const int fromIdx = indices[2*i + 0]; // fromIdx indices can be repeated
|
||||
const int toIdx = indices[2*i + 1]; // toIdx indices CANNOT be repeated
|
||||
out[toIdx] = in[fromIdx];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,122 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * ((J11 * J22) - (J21 * J12));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD2D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD2D_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_sol[NUM_DOFS];
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] = 0;
|
||||
}
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
double s = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
s += solIn(d, e) * quadToDof(d, q);
|
||||
}
|
||||
s *= oper(q, e);
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] += (s * quadToDof(d, q));
|
||||
}
|
||||
}
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
solOut(d, e) += r_sol[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD3D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD3D_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_sol[NUM_DOFS];
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] = 0;
|
||||
}
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
double s = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
s += solIn(d, e) * quadToDof(d, q);
|
||||
}
|
||||
s *= oper(q, e);
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] += (s * quadToDof(d, q));
|
||||
}
|
||||
}
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
solOut(d, e) += r_sol[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,129 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += A2_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * ((J11 * J22) - (J21 * J12));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD2D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD2D_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_sol[NUM_QUAD];
|
||||
|
||||
for (int qOff = 0; qOff < M2_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M2_INNER_BATCH) {
|
||||
double s = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
s += solIn(d, e) * dofToQuad(d, q);
|
||||
}
|
||||
s_sol[q] = s * oper(q, e);
|
||||
}
|
||||
}
|
||||
|
||||
for (int dOff = 0; dOff < M2_INNER_BATCH; ++dOff) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M2_INNER_BATCH) {
|
||||
double r_sol = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_sol += (s_sol[q] * quadToDof(d, q));
|
||||
}
|
||||
solOut(d, e) += r_sol;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += A3_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD3D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD3D_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_sol[NUM_QUAD];
|
||||
|
||||
for (int qOff = 0; qOff < M3_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M3_INNER_BATCH) {
|
||||
double s = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
s += solIn(d, e) * dofToQuad(d, q);
|
||||
}
|
||||
s_sol[q] = s * oper(q, e);
|
||||
}
|
||||
}
|
||||
|
||||
for (int dOff = 0; dOff < M3_INNER_BATCH; ++dOff) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M3_INNER_BATCH) {
|
||||
double r_sol = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_sol += (s_sol[q] * quadToDof(d, q));
|
||||
}
|
||||
solOut(d, e) += r_sol;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,281 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
const double detJ = J(q, e);
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal1D_t oper,
|
||||
@restrict const DLocal1D_t solIn,
|
||||
@restrict DLocal1D_t solOut) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double sol_x[NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] = 0;
|
||||
}
|
||||
// sol_x{qx} = dofToQuad{qx,dx} * sol{dx}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] += s * dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
// sol_x{q} *= oper{q}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] *= oper(qx, e);
|
||||
}
|
||||
// sol{dx} = quadToDof{dx,qx} * sol_x{qx}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, e) += sol_x[qx] * quadToDof(dx, qx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_2D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
const double detJ = ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal2D_t oper,
|
||||
@restrict const DLocal2D_t solIn,
|
||||
@restrict DLocal2D_t solOut) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double sol_xy[NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double sol_x[NUM_QUAD_1D];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
sol_x[qy] = 0;
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, dy, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] += dofToQuad(qx, dx) * s;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double d2q = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[qy][qx] += d2q * sol_x[qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[qy][qx] *= oper(qx, qy, e);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double sol_x[NUM_DOFS_1D];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[dx] = 0;
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double s = sol_xy[qy][qx];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[dx] += quadToDof(dx, qx) * s;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double q2d = quadToDof(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, dy, e) += q2d * sol_x[dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_3D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal3D_t oper,
|
||||
@restrict const DLocal3D_t solIn,
|
||||
@restrict DLocal3D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double sol_xyz[NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xyz[qz][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
double sol_xy[NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double sol_x[NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] = 0;
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, dy, dz, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] += dofToQuad(qx, dx) * s;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[qy][qx] += wy * sol_x[qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
const double wz = dofToQuad(qz, dz);
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xyz[qz][qy][qx] += wz * sol_xy[qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xyz[qz][qy][qx] *= oper(qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
double sol_xy[NUM_DOFS_1D][NUM_DOFS_1D];
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_xy[dy][dx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double sol_x[NUM_DOFS_1D];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[dx] = 0;
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double s = sol_xyz[qz][qy][qx];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[dx] += quadToDof(dx, qx) * s;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = quadToDof(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_xy[dy][dx] += wy * sol_x[dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = quadToDof(dz, qz);
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, dy, dz, e) += wz * sol_xy[dy][dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,341 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
QLocal_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A1_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A1_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
oper(q, e) = quadWeights[q] * COEFF * J(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal1D_t oper,
|
||||
@restrict const DLocal1D_t solIn,
|
||||
@restrict DLocal1D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M1_ELEMENT_BATCHES; @outer) {
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
@exclusive double r_sol[NUM_QUAD_1D];
|
||||
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
for (int i = el; i < NUM_QUAD_DOFS_1D; i += M1_INNER_ELEMENT_BATCH) {
|
||||
s_dofToQuad[i] = dofToQuad[i];
|
||||
s_quadToDof[i] = quadToDof[i];
|
||||
}
|
||||
}
|
||||
|
||||
for (int b = 0; b < M1_OUTER_ELEMENT_BATCH; ++b) {
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
const int e = eOff + b*M1_INNER_ELEMENT_BATCH + el;
|
||||
if (e < numElements) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_sol[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_sol[qx] += s * s_dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_sol[qx] *= oper(qx, e);
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
double s = 0;
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
s += r_sol[qx] * s_quadToDof(dx, qx);
|
||||
}
|
||||
solOut(dx, e) += s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
QLocal_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD_2D; q += A2_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * ((J11 * J22) - (J21 * J12));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal2D_t oper,
|
||||
@restrict const DLocal2D_t solIn,
|
||||
@restrict DLocal2D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M2_ELEMENT_BATCH; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store xy planes in @shared memory
|
||||
@shared double s_xy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_xy2[NUM_QUAD_2D] @dim(NUM_QUAD_1D, NUM_QUAD_1D);
|
||||
|
||||
@exclusive double r_x[NUM_MAX_1D];
|
||||
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
for (int id = x; id < NUM_QUAD_DOFS_1D; id += NUM_MAX_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
s_quadToDof[id] = quadToDof[id];
|
||||
}
|
||||
}
|
||||
|
||||
for (int e = eOff; e < (eOff + M2_ELEMENT_BATCH); ++e) {
|
||||
if (e < numElements) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
s_xy(dx, qy) = 0;
|
||||
}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
r_x[dy] = solIn(dx, dy, e);
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double xy = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
xy += r_x[dy] * s_dofToQuad(qy, dy);
|
||||
}
|
||||
s_xy(dx, qy) = xy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
if (qy < NUM_QUAD_1D) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
double s = 0;
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
s += s_xy(dx, qy) * s_dofToQuad(qx, dx);
|
||||
}
|
||||
s_xy2(qx, qy) = s * oper(qx, qy, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if (qx < NUM_QUAD_1D) {
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
s_xy(dy, qx) = 0;
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
r_x[qy] = s_xy2(qx, qy);
|
||||
}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double s = 0;
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
s += r_x[qy] * s_quadToDof(dy, qy);
|
||||
}
|
||||
s_xy(dy, qx) = s;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double s = 0;
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
s += (s_xy(dy, qx) * s_quadToDof(dx, qx));
|
||||
}
|
||||
solOut(dx, dy, e) += s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
QLocal_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD_3D; q += A3_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal3D_t oper,
|
||||
@restrict const DLocal3D_t solIn,
|
||||
@restrict DLocal3D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store xy planes in @shared memory
|
||||
@shared double s_xy[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
|
||||
// Store z axis as registers
|
||||
@exclusive double r_z[NUM_QUAD_1D];
|
||||
@exclusive double r_z2[NUM_DOFS_1D];
|
||||
|
||||
for (int y = 0; y < NUM_MAX_1D; ++y; @inner) {
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
const int id = (y * NUM_MAX_1D) + x;
|
||||
// Fetch Q <--> D maps
|
||||
if (id < NUM_QUAD_DOFS_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
s_quadToDof[id] = quadToDof[id];
|
||||
}
|
||||
// Initialize our Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_z[qz] = 0;
|
||||
}
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
r_z2[dz] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double s = solIn(dx, dy, dz, e);
|
||||
// Calculate D -> Q in the Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_z[qz] += s * s_dofToQuad(qz, dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// For each xy plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
// Fill xy plane at given z position
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
s_xy(dx, dy) = r_z[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
double s = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = s_dofToQuad(qy, dy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = s_dofToQuad(qx, dx);
|
||||
s += wx * wy * s_xy(dx, dy);
|
||||
}
|
||||
}
|
||||
|
||||
s *= oper(qx, qy, qz, e);
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = s_quadToDof(dz, qz);
|
||||
r_z2[dz] += wz * s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@barrier("s_xy_sync_1");
|
||||
}
|
||||
// Iterate over xy planes to compute solution
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
// Place xy plane in @shared memory
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
s_xy(qx, qy) = r_z2[dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Finalize solution in xy plane
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
double solZ = 0;
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = s_quadToDof(dy, qy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double wx = s_quadToDof(dx, qx);
|
||||
solZ += wx * wy * s_xy(qx, qy);
|
||||
}
|
||||
}
|
||||
solOut(dx, dy, dz, e) += solZ;
|
||||
}
|
||||
}
|
||||
}
|
||||
@barrier("s_xy_sync_2");
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -1,142 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "operator.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
// FIXME: move this object to the Backend?
|
||||
::occa::kernelBuilder OccaConstrainedOperator::mapDofBuilder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_map_dofs",
|
||||
|
||||
"const int idx = v2[i];"
|
||||
"v0[idx] = v1[idx];",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}");
|
||||
|
||||
// FIXME: move this object to the Backend?
|
||||
::occa::kernelBuilder OccaConstrainedOperator::clearDofBuilder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_clear_dofs",
|
||||
|
||||
"v0[v1[i]] = 0.0;",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}");
|
||||
|
||||
OccaConstrainedOperator::OccaConstrainedOperator(
|
||||
mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_)
|
||||
|
||||
: Operator(A_->InLayout()->As<Layout>()),
|
||||
z(OutLayout_()),
|
||||
w(OutLayout_()),
|
||||
mfem_z((z.DontDelete(), z)),
|
||||
mfem_w((w.DontDelete(), w))
|
||||
{
|
||||
Setup(OutLayout_().OccaEngine().GetDevice(), A_, constraintList_, own_A_);
|
||||
}
|
||||
|
||||
void OccaConstrainedOperator::Setup(::occa::device device_,
|
||||
mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_)
|
||||
{
|
||||
device = device_;
|
||||
|
||||
A = A_;
|
||||
own_A = own_A_;
|
||||
|
||||
constraintIndices = constraintList_.Size();
|
||||
if (constraintList_.Size() > 0)
|
||||
{
|
||||
constraintList = constraintList_.Get_PArray()->As<Array>().OccaMem();
|
||||
}
|
||||
else
|
||||
{
|
||||
// constraintList is not used
|
||||
}
|
||||
}
|
||||
|
||||
void OccaConstrainedOperator::EliminateRHS(const Vector &x, Vector &b) const
|
||||
{
|
||||
::occa::kernel mapDofs = mapDofBuilder.build(device);
|
||||
|
||||
w.OccaFill(0.0);
|
||||
|
||||
if (constraintIndices)
|
||||
{
|
||||
mapDofs(constraintIndices, w.OccaMem(), x.OccaMem(), constraintList);
|
||||
}
|
||||
|
||||
A->Mult(mfem_w, mfem_z);
|
||||
|
||||
b.Axpby<double>(1.0, b, -1.0, z);
|
||||
|
||||
if (constraintIndices)
|
||||
{
|
||||
mapDofs(constraintIndices, b.OccaMem(), x.OccaMem(), constraintList);
|
||||
}
|
||||
}
|
||||
|
||||
void OccaConstrainedOperator::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
mfem::Vector mfem_y(y);
|
||||
if (constraintIndices == 0)
|
||||
{
|
||||
A->Mult(x.Wrap(), mfem_y);
|
||||
return;
|
||||
}
|
||||
|
||||
::occa::kernel mapDofs = mapDofBuilder.build(device);
|
||||
::occa::kernel clearDofs = clearDofBuilder.build(device);
|
||||
|
||||
// z.OccaAssign(x); // z = x
|
||||
// Is Axpy faster than DtoD copy on Volta?
|
||||
z.Axpby(1.0, x, 0.0, x);
|
||||
|
||||
clearDofs(constraintIndices, z.OccaMem(), constraintList);
|
||||
|
||||
A->Mult(mfem_z, mfem_y);
|
||||
|
||||
mapDofs(constraintIndices, y.OccaMem(), x.OccaMem(), constraintList);
|
||||
}
|
||||
|
||||
OccaConstrainedOperator::~OccaConstrainedOperator()
|
||||
{
|
||||
if (own_A)
|
||||
{
|
||||
delete A;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,127 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_OPERATOR_HPP
|
||||
#define MFEM_BACKENDS_OCCA_OPERATOR_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "../../linalg/operator.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Operator : public mfem::Operator
|
||||
{
|
||||
public:
|
||||
/// Creare an operator with the same dimensions as @a orig.
|
||||
Operator(const Operator &orig)
|
||||
: mfem::Operator(orig) { }
|
||||
|
||||
Operator(Layout &layout)
|
||||
: mfem::Operator(layout) { }
|
||||
|
||||
Operator(Layout &in_layout, Layout &out_layout)
|
||||
: mfem::Operator(in_layout, out_layout) { }
|
||||
|
||||
Layout &InLayout_() const { return in_layout->As<Layout>(); }
|
||||
|
||||
Layout &OutLayout_() const { return out_layout->As<Layout>(); }
|
||||
|
||||
virtual void Mult_(const Vector &x, Vector &y) const = 0;
|
||||
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const
|
||||
{ MFEM_ABORT("method is not supported"); }
|
||||
|
||||
// override
|
||||
virtual void Mult(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{
|
||||
Mult_(x.Get_PVector()->As<Vector>(),
|
||||
y.Get_PVector()->As<Vector>());
|
||||
}
|
||||
|
||||
// override
|
||||
virtual void MultTranspose(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{
|
||||
MultTranspose_(x.Get_PVector()->As<Vector>(),
|
||||
y.Get_PVector()->As<Vector>());
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
class OccaConstrainedOperator : public Operator
|
||||
{
|
||||
protected:
|
||||
::occa::device device;
|
||||
|
||||
mfem::Operator *A; //< The unconstrained Operator.
|
||||
bool own_A; //< Ownership flag for A.
|
||||
::occa::memory constraintList; //< List of constrained indices/dofs.
|
||||
int constraintIndices;
|
||||
mutable Vector z, w; //< Auxiliary vectors.
|
||||
mutable mfem::Vector mfem_z, mfem_w; // Wrap z, w
|
||||
|
||||
static ::occa::kernelBuilder mapDofBuilder, clearDofBuilder;
|
||||
|
||||
public:
|
||||
/** @brief Constructor from a general Operator and a list of essential
|
||||
indices/dofs.
|
||||
|
||||
Specify the unconstrained operator @a *A and a @a list of indices to
|
||||
constrain, i.e. each entry @a list[i] represents an essential-dof. If the
|
||||
ownership flag @a own_A is true, the operator @a *A will be destroyed
|
||||
when this object is destroyed. */
|
||||
OccaConstrainedOperator(mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_ = false);
|
||||
|
||||
void Setup(::occa::device device_,
|
||||
mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_ = false);
|
||||
|
||||
/** @brief Eliminate "essential boundary condition" values specified in @a x
|
||||
from the given right-hand side @a b.
|
||||
|
||||
Performs the following steps:
|
||||
|
||||
z = A((0,x_b)); b_i -= z_i; b_b = x_b;
|
||||
|
||||
where the "_b" subscripts denote the essential (boundary) indices/dofs of
|
||||
the vectors, and "_i" -- the rest of the entries. */
|
||||
void EliminateRHS(const Vector &x, Vector &b) const;
|
||||
|
||||
/** @brief Constrained operator action.
|
||||
|
||||
Performs the following steps:
|
||||
|
||||
z = A((x_i,0)); y_i = z_i; y_b = x_b;
|
||||
|
||||
where the "_b" subscripts denote the essential (boundary) indices/dofs of
|
||||
the vectors, and "_i" -- the rest of the entries. */
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
|
||||
// Destructor: destroys the unconstrained Operator @a A if @a own_A is true.
|
||||
virtual ~OccaConstrainedOperator();
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_OPERATOR_HPP
|
||||
@@ -1,57 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
TILESIZE : Tilesize for iterating over dofs
|
||||
================================================
|
||||
*/
|
||||
|
||||
@kernel void Mult(const int entries,
|
||||
@restrict const int *offsets,
|
||||
@restrict const int *indices,
|
||||
@restrict const double *weights,
|
||||
@restrict const double *in,
|
||||
@restrict double *out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
double value = 0;
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
value += weights[j] * in[indices[j]];
|
||||
}
|
||||
out[i] = value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MappedMult(const int entries,
|
||||
@restrict const int *offsets,
|
||||
@restrict const int *indices,
|
||||
@restrict const double *weights,
|
||||
@restrict const int *outIndices,
|
||||
@restrict const double *in,
|
||||
@restrict double *out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
double value = 0;
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
value += weights[j] * in[indices[j]];
|
||||
}
|
||||
out[outIndices[i]] = value;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,221 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "sparsemat.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props)
|
||||
: Operator(in_layout, out_layout)
|
||||
{
|
||||
Setup(in_layout.OccaEngine().GetDevice(), m, props);
|
||||
}
|
||||
|
||||
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> offsets_,
|
||||
::occa::array<int> indices_,
|
||||
::occa::array<double> weights_,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props)
|
||||
: Operator(in_layout, out_layout),
|
||||
offsets(offsets_),
|
||||
indices(indices_),
|
||||
weights(weights_),
|
||||
reorderIndices(reorderIndices_),
|
||||
mappedIndices(mappedIndices_)
|
||||
{
|
||||
SetupKernel(in_layout.OccaEngine().GetDevice(), props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::Setup(::occa::device device, const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
Setup(device, m, ::occa::array<int>(), ::occa::array<int>(), props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::Setup(::occa::device device, const SparseMatrix &m,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
MFEM_ASSERT(m.Finalized(), "");
|
||||
MFEM_ASSERT(m.Height() == height, "");
|
||||
MFEM_ASSERT(m.Width() == width, "");
|
||||
|
||||
const int nnz = m.GetI()[height];
|
||||
offsets.allocate(device,
|
||||
height + 1, m.GetI());
|
||||
indices.allocate(device,
|
||||
nnz, m.GetJ());
|
||||
weights.allocate(device,
|
||||
nnz, m.GetData());
|
||||
|
||||
offsets.keepInDevice();
|
||||
indices.keepInDevice();
|
||||
weights.keepInDevice();
|
||||
|
||||
reorderIndices = reorderIndices_;
|
||||
mappedIndices = mappedIndices_;
|
||||
|
||||
SetupKernel(device, props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::SetupKernel(::occa::device device,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
const bool hasOutIndices = mappedIndices.isInitialized();
|
||||
|
||||
const ::occa::properties defaultProps("defines: {"
|
||||
" TILESIZE: 256,"
|
||||
"}");
|
||||
|
||||
const std::string &okl_path = InLayout_().OccaEngine().GetOklPath();
|
||||
mapKernel = device.buildKernel(okl_path + "mappings.okl",
|
||||
"MapSubVector",
|
||||
defaultProps + props);
|
||||
|
||||
multKernel = device.buildKernel(okl_path + "sparse.okl",
|
||||
hasOutIndices ? "MappedMult" : "Mult",
|
||||
defaultProps + props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (reorderIndices.isInitialized() ||
|
||||
mappedIndices.isInitialized())
|
||||
{
|
||||
if (reorderIndices.isInitialized())
|
||||
{
|
||||
mapKernel((int) (reorderIndices.size() / 2),
|
||||
reorderIndices,
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
if (mappedIndices.isInitialized())
|
||||
{
|
||||
multKernel((int) (mappedIndices.size()),
|
||||
offsets, indices, weights,
|
||||
mappedIndices,
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
multKernel((int) height,
|
||||
offsets, indices, weights,
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
OccaSparseMatrix* CreateMappedSparseMatrix(Layout &in_layout,
|
||||
Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
const int mHeight = m.Height();
|
||||
// const int mWidth = m.Width();
|
||||
|
||||
// Count indices that are only reordered (true dofs)
|
||||
const int *I = m.GetI();
|
||||
const int *J = m.GetJ();
|
||||
const double *D = m.GetData();
|
||||
|
||||
int trueCount = 0;
|
||||
for (int i = 0; i < mHeight; ++i)
|
||||
{
|
||||
trueCount += ((I[i + 1] - I[i]) == 1);
|
||||
}
|
||||
const int dupCount = (mHeight - trueCount);
|
||||
|
||||
// Create the reordering map for entries that aren't modified (true dofs)
|
||||
::occa::device device(in_layout.OccaEngine().GetDevice());
|
||||
::occa::array<int> reorderIndices(device,
|
||||
2 * trueCount);
|
||||
::occa::array<int> mappedIndices, offsets, indices;
|
||||
::occa::array<double> weights;
|
||||
|
||||
if (dupCount)
|
||||
{
|
||||
mappedIndices.allocate(device,
|
||||
dupCount);
|
||||
}
|
||||
int trueIdx = 0, dupIdx = 0;
|
||||
for (int i = 0; i < mHeight; ++i)
|
||||
{
|
||||
const int i1 = I[i];
|
||||
if ((I[i + 1] - i1) == 1)
|
||||
{
|
||||
reorderIndices[trueIdx++] = J[i1];
|
||||
reorderIndices[trueIdx++] = i;
|
||||
}
|
||||
else
|
||||
{
|
||||
mappedIndices[dupIdx++] = i;
|
||||
}
|
||||
}
|
||||
reorderIndices.keepInDevice();
|
||||
|
||||
if (dupCount)
|
||||
{
|
||||
mappedIndices.keepInDevice();
|
||||
|
||||
// Extract sparse matrix without reordered identity
|
||||
const int dupNnz = I[mHeight] - trueCount;
|
||||
|
||||
offsets.allocate(device,
|
||||
dupCount + 1);
|
||||
indices.allocate(device,
|
||||
dupNnz);
|
||||
weights.allocate(device,
|
||||
dupNnz);
|
||||
|
||||
int nnz = 0;
|
||||
offsets[0] = 0;
|
||||
for (int i = 0; i < dupCount; ++i)
|
||||
{
|
||||
const int idx = mappedIndices[i];
|
||||
const int offStart = I[idx];
|
||||
const int offEnd = I[idx + 1];
|
||||
offsets[i + 1] = offsets[i] + (offEnd - offStart);
|
||||
for (int j = offStart; j < offEnd; ++j)
|
||||
{
|
||||
indices[nnz] = J[j];
|
||||
weights[nnz] = D[j];
|
||||
++nnz;
|
||||
}
|
||||
}
|
||||
|
||||
offsets.keepInDevice();
|
||||
indices.keepInDevice();
|
||||
weights.keepInDevice();
|
||||
}
|
||||
|
||||
return new OccaSparseMatrix(in_layout, out_layout,
|
||||
offsets, indices, weights,
|
||||
reorderIndices, mappedIndices,
|
||||
props);
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,89 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
|
||||
#define MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "vector.hpp"
|
||||
#include "engine.hpp"
|
||||
#include "operator.hpp"
|
||||
#include "../../linalg/sparsemat.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
/// TODO: doxygen
|
||||
class OccaSparseMatrix : public Operator
|
||||
{
|
||||
protected:
|
||||
::occa::array<int> offsets, indices;
|
||||
::occa::array<double> weights;
|
||||
::occa::array<int> reorderIndices, mappedIndices;
|
||||
::occa::kernel mapKernel, multKernel;
|
||||
|
||||
void Setup(::occa::device device, const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props);
|
||||
|
||||
void Setup(::occa::device device, const mfem::SparseMatrix &m,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props);
|
||||
|
||||
void SetupKernel(::occa::device device,
|
||||
const ::occa::properties &props);
|
||||
|
||||
public:
|
||||
/// Construct an empty OccaSparseMatrix.
|
||||
OccaSparseMatrix(const Operator &orig)
|
||||
: Operator(orig) { }
|
||||
|
||||
// Implicitly defined copy constructor.
|
||||
|
||||
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> offsets_,
|
||||
::occa::array<int> indices_,
|
||||
::occa::array<double> weights_,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
const ::occa::array<int> &GetReorderIndices() const
|
||||
{ return reorderIndices; }
|
||||
|
||||
// override
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
|
||||
/// TODO: doxygen
|
||||
OccaSparseMatrix* CreateMappedSparseMatrix(
|
||||
Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
|
||||
@@ -1,81 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "url_handler.hpp"
|
||||
#include "../../general/error.hpp"
|
||||
#include <cstdlib>
|
||||
#include <sys/stat.h>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
FileOpener::FileOpener(const std::string &prefix,
|
||||
const std::string &env_variable)
|
||||
: pfx(prefix)
|
||||
{
|
||||
const char *env_path = getenv(env_variable.c_str());
|
||||
if (!env_path) { return; }
|
||||
std::string path(env_path);
|
||||
for (std::size_t start = 0, end; start < path.size(); start = end + 1)
|
||||
{
|
||||
end = path.find(':', start);
|
||||
if (end == std::string::npos)
|
||||
{
|
||||
AddDir(path.substr(start, end));
|
||||
break;
|
||||
}
|
||||
AddDir(path.substr(start, end - start));
|
||||
}
|
||||
}
|
||||
|
||||
bool FileOpener::AddDir(const std::string &dir)
|
||||
{
|
||||
if (dir.size() == 0 || dir[0] != '/') { return false; }
|
||||
struct stat dir_stat;
|
||||
if (stat(dir.c_str(), &dir_stat)) { return false; }
|
||||
if (!S_ISDIR(dir_stat.st_mode)) { return false; }
|
||||
paths.push_back(dir + (*dir.rbegin() == '/' ? "" : "/"));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool FileOpener::handles(const std::string &filename)
|
||||
{
|
||||
return filename.size() >= pfx.size() &&
|
||||
filename.compare(0, pfx.size(), pfx) == 0;
|
||||
}
|
||||
|
||||
std::string FileOpener::expand(const std::string &filename)
|
||||
{
|
||||
std::string sfx(filename.substr(pfx.size()));
|
||||
for (std::size_t i = 0; i < paths.size(); i++)
|
||||
{
|
||||
std::string file = paths[i] + sfx;
|
||||
struct stat file_stat;
|
||||
if (stat(file.c_str(), &file_stat) == 0 && S_ISREG(file_stat.st_mode))
|
||||
{
|
||||
return file;
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("invalid url: " << filename);
|
||||
return sfx;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,47 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
|
||||
#define MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class FileOpener : public ::occa::io::fileOpener
|
||||
{
|
||||
protected:
|
||||
std::string pfx; // prefix, e.g. "mfem://"
|
||||
std::vector<std::string> paths; // paths to search for prefix replacement
|
||||
|
||||
public:
|
||||
FileOpener(const std::string &prefix, const std::string &env_variable);
|
||||
|
||||
bool AddDir(const std::string &dir);
|
||||
|
||||
virtual bool handles(const std::string &filename);
|
||||
virtual std::string expand(const std::string &filename);
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
|
||||
@@ -1,22 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
typedef double* Local_t @dim(numDofs, numElements);
|
||||
|
||||
@kernel void InitLocalVector(const int numElements,
|
||||
const int numDofs,
|
||||
@restrict Local_t sol) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int d = 0; d < numDofs; ++d; @inner) {
|
||||
sol(d, e) = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,196 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "../../linalg/vector.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
PVector *Vector::DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const
|
||||
{
|
||||
MFEM_ASSERT(buffer_type_id == ScalarId<double>::value, "");
|
||||
Vector *new_vector = new Vector(OccaLayout());
|
||||
if (copy_data)
|
||||
{
|
||||
new_vector->slice.copyFrom(slice);
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
*buffer = new_vector->GetBuffer();
|
||||
}
|
||||
return new_vector;
|
||||
}
|
||||
|
||||
void Vector::DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const
|
||||
{
|
||||
// Can be called when Size() == 0, e.g. when an MPI-parallel vector has a
|
||||
// local size of 0.
|
||||
|
||||
MFEM_ASSERT(result_type_id == ScalarId<double>::value, "");
|
||||
double *res = (double *)result;
|
||||
const Vector &xp = x.As<Vector>();
|
||||
MFEM_ASSERT(this->Size() == xp.Size(), "");
|
||||
*res = ::occa::linalg::dot<double, double, double>(this->slice, xp.slice);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
double local_dot = *res;
|
||||
if (IsParallel())
|
||||
{
|
||||
MPI_Allreduce(&local_dot, res, 1, MPI_DOUBLE, MPI_SUM,
|
||||
OccaEngine().GetComm());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void Vector::DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id)
|
||||
{
|
||||
//
|
||||
// TODO: move all kernel builders to class mfem::occa::Backend
|
||||
//
|
||||
static ::occa::kernelBuilder axpby1_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"mfem_occa_axpby1",
|
||||
"v0[i] = c0 * v1[i];",
|
||||
"defines: {"
|
||||
" CTYPE0: 'double',"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" TILESIZE: '128',"
|
||||
"}");
|
||||
|
||||
static ::occa::kernelBuilder axpby2_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"mfem_occa_axpby2",
|
||||
"v0[i] = c0 * v0[i] + c1 * v1[i];",
|
||||
"defines: {"
|
||||
" CTYPE0: 'double',"
|
||||
" CTYPE1: 'double',"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" TILESIZE: '128',"
|
||||
"}");
|
||||
|
||||
static ::occa::kernelBuilder axpby3_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"mfem_occa_axpby3",
|
||||
"v0[i] = c0 * v1[i] + c1 * v2[i];",
|
||||
"defines: {"
|
||||
" CTYPE0: 'double',"
|
||||
" CTYPE1: 'double',"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'double',"
|
||||
" TILESIZE: '128',"
|
||||
"}");
|
||||
|
||||
// called only when Size() != 0
|
||||
|
||||
MFEM_ASSERT(ab_type_id == ScalarId<double>::value, "");
|
||||
const double da = *static_cast<const double *>(a);
|
||||
const double db = *static_cast<const double *>(b);
|
||||
MFEM_ASSERT(da == 0.0 || dynamic_cast<const Vector *>(&x) != NULL,
|
||||
"invalid Vector x");
|
||||
MFEM_ASSERT(db == 0.0 || dynamic_cast<const Vector *>(&y) != NULL,
|
||||
"invalid Vector y");
|
||||
const Vector *xp = static_cast<const Vector *>(&x);
|
||||
const Vector *yp = static_cast<const Vector *>(&y);
|
||||
|
||||
MFEM_ASSERT(da == 0.0 || this->Size() == xp->Size(), "");
|
||||
MFEM_ASSERT(db == 0.0 || this->Size() == yp->Size(), "");
|
||||
|
||||
if (da == 0.0)
|
||||
{
|
||||
if (db == 0.0)
|
||||
{
|
||||
OccaFill(da);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (this->slice == yp->slice)
|
||||
{
|
||||
// *this *= db
|
||||
::occa::linalg::operator_mult_eq(slice, db);
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = db * y
|
||||
::occa::kernel kernel = axpby1_builder.build(slice.getDevice());
|
||||
kernel((int)Size(), db, slice, yp->slice);
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (db == 0.0)
|
||||
{
|
||||
if (this->slice == xp->slice)
|
||||
{
|
||||
// *this *= da
|
||||
::occa::linalg::operator_mult_eq(slice, da);
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = da * x
|
||||
::occa::kernel kernel = axpby1_builder.build(slice.getDevice());
|
||||
kernel((int)Size(), da, slice, xp->slice);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ASSERT(xp->slice != yp->slice, "invalid input");
|
||||
if (this->slice == xp->slice)
|
||||
{
|
||||
// *this = da * (*this) + db * y
|
||||
::occa::kernel kernel = axpby2_builder.build(slice.getDevice());
|
||||
kernel((int)Size(), da, db, slice, yp->slice);
|
||||
}
|
||||
else if (this->slice == yp->slice)
|
||||
{
|
||||
// *this = da * x + db * (*this)
|
||||
::occa::kernel kernel = axpby2_builder.build(slice.getDevice());
|
||||
kernel((int)Size(), db, da, slice, xp->slice);
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = da * x + db * y
|
||||
::occa::kernel kernel = axpby3_builder.build(slice.getDevice());
|
||||
kernel((int)Size(), da, db, slice, xp->slice, yp->slice);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mfem::Vector Vector::Wrap()
|
||||
{
|
||||
return mfem::Vector(*this);
|
||||
}
|
||||
|
||||
const mfem::Vector Vector::Wrap() const
|
||||
{
|
||||
return mfem::Vector(*const_cast<Vector*>(this));
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -1,83 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_VECTOR_HPP
|
||||
#define MFEM_BACKENDS_OCCA_VECTOR_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "../base/vector.hpp"
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
// FIXME: Once XL fixes this code quirk we can remove this #ifdef switch
|
||||
#ifdef __ibmxl__
|
||||
class Vector : public Array, public PVector
|
||||
#else
|
||||
class Vector : virtual public Array, public PVector
|
||||
#endif
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// DLayout layout;
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
virtual PVector *DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const;
|
||||
|
||||
virtual void DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const;
|
||||
|
||||
virtual void DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id);
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
public:
|
||||
Vector(const Engine &e)
|
||||
: PArray(*(new Layout(e, 0))), Array(e), PVector(*layout)
|
||||
{ }
|
||||
|
||||
Vector(Layout <)
|
||||
: PArray(lt), Array(lt, sizeof(double)), PVector(lt)
|
||||
{ }
|
||||
|
||||
mfem::Vector Wrap();
|
||||
|
||||
const mfem::Vector Wrap() const;
|
||||
|
||||
#if defined(MFEM_USE_MPI)
|
||||
bool IsParallel() const { return (OccaEngine().GetComm() != MPI_COMM_NULL); }
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_VECTOR_HPP
|
||||
@@ -1,321 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
oper(q, e) = quadWeights[q] * COEFF * J(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DVLocal1D_t solIn,
|
||||
@restrict DVLocal1D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double sol_x[1][NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[0][qx] = 0;
|
||||
}
|
||||
// sol_x{qx} = dofToQuad{qx,dx} * sol{dx}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[0][qx] += dofToQuad(qx, dx) * solIn(0, dx, e);
|
||||
}
|
||||
}
|
||||
// sol_x{q} *= oper{q}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[0][qx] *= oper(qx, e);
|
||||
}
|
||||
// sol{dx} = quadToDof{dx,qx} * sol_x{qx}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(0, dx, e) += sol_x[0][qx] * quadToDof(dx, qx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_2D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * ((J11 * J22) - (J21 * J12));
|
||||
}
|
||||
} // e
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DVLocal2D_t solIn,
|
||||
@restrict DVLocal2D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy=0; dummy<1; ++dummy; @inner) {
|
||||
double sol_xy[2][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[0][qx][qy] = 0;
|
||||
sol_xy[1][qx][qy] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double sol_x[2][NUM_QUAD_1D];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
sol_x[0][qy] = 0;
|
||||
sol_x[1][qy] = 0;
|
||||
}
|
||||
|
||||
// sol_x{vd, dx, qy} = dofToQuad{qy, dy} * sol{vd, dx, dy}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
sol_x[0][qy] += dofToQuad(qy, dx) * solIn(0, dx, dy, e);
|
||||
sol_x[1][qy] += dofToQuad(qy, dx) * solIn(1, dx, dy, e);
|
||||
}
|
||||
}
|
||||
|
||||
// sol_xy{qx, qy} = dofToQuad{qx, dx} * sol_x{dx, qy}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double d2q = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[0][qx][qy] += d2q * sol_x[0][qx];
|
||||
sol_xy[1][qx][qy] += d2q * sol_x[1][qx];
|
||||
}
|
||||
}
|
||||
} // dy
|
||||
|
||||
// sol_xy{qx, qy} = sol_xy{q} *= oper{q, e}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const int q = QUAD_2D_ID(qx, qy);
|
||||
sol_xy[0][qx][qy] *= oper(q, e);
|
||||
sol_xy[1][qx][qy] *= oper(q, e);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double sol_x[2][NUM_DOFS_1D];
|
||||
for (int dx = 0; dx < NUM_QUAD_1D; ++dx) {
|
||||
sol_x[0][dx] = 0;
|
||||
sol_x[1][dx] = 0;
|
||||
}
|
||||
|
||||
// sol_x{qx, dy} = quadToDof{dy, qy} * sol_xy{qx, qy}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[0][dx] += quadToDof(dx, qx) * sol_xy[0][qx][qy];
|
||||
sol_x[1][dx] += quadToDof(dx, qx) * sol_xy[1][qx][qy];
|
||||
}
|
||||
}
|
||||
|
||||
// sol{dx, dy, e} = quadToDof{dx, qx} * sol_x{qx, dy}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double q2d = quadToDof(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(0, dx, dy, e) += q2d * sol_x[0][dx];
|
||||
solOut(1, dx, dy, e) += q2d * sol_x[1][dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
} // dummy
|
||||
} // e
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_3D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DVLocal3D_t solIn,
|
||||
@restrict DVLocal3D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double sol_xyz[3][NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xyz[0][qz][qy][qx] = 0;
|
||||
sol_xyz[1][qz][qy][qx] = 0;
|
||||
sol_xyz[2][qz][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
double sol_xy[3][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[0][qy][qx] = 0;
|
||||
sol_xy[1][qy][qx] = 0;
|
||||
sol_xy[2][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double sol_x[3][NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[0][qx] = 0;
|
||||
sol_x[1][qx] = 0;
|
||||
sol_x[2][qx] = 0;
|
||||
}
|
||||
|
||||
// sol_x{qx} = dofToQuad{qx, dx} * sol{dx, dy, dz, e}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[0][qx] += dofToQuad(qx, dx) * solIn(0, dx, dy, dz, e);
|
||||
sol_x[1][qx] += dofToQuad(qx, dx) * solIn(1, dx, dy, dz, e);
|
||||
sol_x[2][qx] += dofToQuad(qx, dx) * solIn(2, dx, dy, dz, e);
|
||||
}
|
||||
}
|
||||
|
||||
// sol_xy{qx, qy} = dofToQuad{qy, dy} * sol_x{dx}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[0][qy][qx] += wy * sol_x[0][qx];
|
||||
sol_xy[1][qy][qx] += wy * sol_x[1][qx];
|
||||
sol_xy[2][qy][qx] += wy * sol_x[2][qx];
|
||||
}
|
||||
}
|
||||
} // dy
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
const double wz = dofToQuad(qz, dz);
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xyz[0][qz][qy][qx] += wz * sol_xy[0][qy][qx];
|
||||
sol_xyz[1][qz][qy][qx] += wz * sol_xy[1][qy][qx];
|
||||
sol_xyz[2][qz][qy][qx] += wz * sol_xy[2][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
} // dz
|
||||
|
||||
// sol_xyz{qz, qy, qx} *= oper{q, e}
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const int q = QUAD_3D_ID(qx, qy, qz);
|
||||
sol_xyz[0][qz][qy][qx] *= oper(q, e);
|
||||
sol_xyz[1][qz][qy][qx] *= oper(q, e);
|
||||
sol_xyz[2][qz][qy][qx] *= oper(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
double sol_xy[3][NUM_DOFS_1D][NUM_DOFS_1D];
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_xy[0][dy][dx] = 0;
|
||||
sol_xy[1][dy][dx] = 0;
|
||||
sol_xy[2][dy][dx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double sol_x[3][NUM_DOFS_1D];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[0][dx] = 0;
|
||||
sol_x[1][dx] = 0;
|
||||
sol_x[2][dx] = 0;
|
||||
}
|
||||
|
||||
// sol_x{dx} = quadToDof{dx, qx} * sol_xyz{qz, qy, qx}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[0][dx] += quadToDof(dx, qx) * sol_xyz[0][qz][qy][qx];
|
||||
sol_x[1][dx] += quadToDof(dx, qx) * sol_xyz[1][qz][qy][qx];
|
||||
sol_x[2][dx] += quadToDof(dx, qx) * sol_xyz[2][qz][qy][qx];
|
||||
}
|
||||
}
|
||||
|
||||
// sol_xy{dy, dx} = quadToDof{dy, qy} * sol_x{dx}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = quadToDof(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_xy[0][dy][dx] += wy * sol_x[0][dx];
|
||||
sol_xy[1][dy][dx] += wy * sol_x[1][dx];
|
||||
sol_xy[2][dy][dx] += wy * sol_x[2][dx];
|
||||
}
|
||||
}
|
||||
} // qy
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = quadToDof(dz, qz);
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(0, dx, dy, dz, e) += wz * sol_xy[0][dy][dx];
|
||||
solOut(1, dx, dy, dz, e) += wz * sol_xy[1][dy][dx];
|
||||
solOut(2, dx, dy, dz, e) += wz * sol_xy[2][dy][dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
} // qz
|
||||
} // dummy
|
||||
} // e
|
||||
|
||||
}
|
||||
//======================================
|
||||
@@ -1,182 +0,0 @@
|
||||
##################################################################################
|
||||
#
|
||||
# Set defaults for XSDK CMake projects
|
||||
#
|
||||
##################################################################################
|
||||
|
||||
#
|
||||
# This module implements standard behavior for XSDK CMake projects. The main
|
||||
# thing it does in XSDK mode (i.e. USE_XSDK_DEFAULTS=TRUE) is to print out
|
||||
# when the env vars CC, CXX, FC and compiler flags CFLAGS, CXXFLAGS, and
|
||||
# FFLAGS/FCFLAGS are used to select the compilers and compiler flags (raw
|
||||
# CMake does this silently) and to set BUILD_SHARED_LIBS=TRUE and
|
||||
# CMAKE_BUILD_TYPE=DEBUG by default. It does not implement *all* of the
|
||||
# standard XSDK configuration parameters. The parent CMake project must do
|
||||
# that.
|
||||
#
|
||||
# Note that when USE_XSDK_DEFAULTS=TRUE, then the Fortran flags will be read
|
||||
# from either of the env vars FFLAGS or FCFLAGS. If both are set, but are the
|
||||
# same, then FFLAGS it used (which is the same as FCFLAGS). However, if both
|
||||
# are set but are not equal, then a FATAL_ERROR is raised and CMake configure
|
||||
# processing is stopped.
|
||||
#
|
||||
# To be used in a parent project, this module must be included after
|
||||
#
|
||||
# PROJECT(${PROJECT_NAME} NONE)
|
||||
#
|
||||
# is called but before the compilers are defined and processed using:
|
||||
#
|
||||
# ENABLE_LANGUAGE(<LANG>)
|
||||
#
|
||||
# For example, one would do:
|
||||
#
|
||||
# PROJECT(${PROJECT_NAME} NONE)
|
||||
# ...
|
||||
# SET(USE_XSDK_DEFAULTS_DEFAULT TRUE) # Set to false if desired
|
||||
# INCLUDE("${CMAKE_CURRENT_SOURCE_DIR}/stdk/XSDKDefaults.cmake")
|
||||
# ...
|
||||
# ENABLE_LANGUAGE(C)
|
||||
# ENABLE_LANGUAGE(C++)
|
||||
# ENABLE_LANGUAGE(Fortran)
|
||||
#
|
||||
# The variable `USE_XSDK_DEFAULTS_DEFAULT` is used as the default for the
|
||||
# cache var `USE_XSDK_DEFAULTS`. That way, a project can decide if it wants
|
||||
# XSDK defaults turned on or off by default and users can independently decide
|
||||
# if they want the CMake project to use standard XSDK behavior or raw CMake
|
||||
# behavior.
|
||||
#
|
||||
# By default, the XSDKDefaults.cmake module assumes that the project will need
|
||||
# C, C++, and Fortran. If any language is not needed then, set
|
||||
# XSDK_ENABLE_C=OFF, XSDK_ENABLE_CXX=OFF, or XSDK_ENABLE_Fortran=OFF *before*
|
||||
# including this module. Note, these variables are *not* cache vars because a
|
||||
# project either does or does not have C, C++ or Fortran source files, the
|
||||
# user has nothing to do with this so there is no need for cache vars. The
|
||||
# parent CMake project just needs to tell XSDKDefault.cmake what languages is
|
||||
# needs or does not need.
|
||||
#
|
||||
# For example, if the parent CMake project only needs C, then it would do:
|
||||
#
|
||||
# PROJECT(${PROJECT_NAME} NONE)'
|
||||
# ...
|
||||
# SET(USE_XSDK_DEFAULTS_DEFAULT TRUE)
|
||||
# SET(XSDK_ENABLE_CXX OFF)
|
||||
# SET(XSDK_ENABLE_Fortran OFF)
|
||||
# INCLUDE("${CMAKE_CURRENT_SOURCE_DIR}/stdk/XSDKDefaults.cmake")
|
||||
# ...
|
||||
# ENABLE_LANGAUGE(C)
|
||||
#
|
||||
# This module code will announce when it sets any variables.
|
||||
#
|
||||
|
||||
#
|
||||
# Helper functions
|
||||
#
|
||||
|
||||
IF (NOT COMMAND PRINT_VAR)
|
||||
FUNCTION(PRINT_VAR VAR_NAME)
|
||||
MESSAGE("-- " "${VAR_NAME} = '${${VAR_NAME}}'")
|
||||
ENDFUNCTION()
|
||||
ENDIF()
|
||||
|
||||
IF (NOT COMMAND SET_DEFAULT)
|
||||
MACRO(SET_DEFAULT VAR)
|
||||
IF ("${${VAR}}" STREQUAL "")
|
||||
SET(${VAR} ${ARGN})
|
||||
ENDIF()
|
||||
ENDMACRO()
|
||||
ENDIF()
|
||||
|
||||
#
|
||||
# XSDKDefaults.cmake control variables
|
||||
#
|
||||
|
||||
# USE_XSDK_DEFAULTS
|
||||
IF ("${USE_XSDK_DEFAULTS_DEFAULT}" STREQUAL "")
|
||||
SET(USE_XSDK_DEFAULTS_DEFAULT FALSE)
|
||||
ENDIF()
|
||||
SET(USE_XSDK_DEFAULTS ${USE_XSDK_DEFAULTS_DEFAULT} CACHE BOOL
|
||||
"Use XSDK defaults and behavior.")
|
||||
PRINT_VAR(USE_XSDK_DEFAULTS)
|
||||
|
||||
SET_DEFAULT(XSDK_ENABLE_C TRUE)
|
||||
SET_DEFAULT(XSDK_ENABLE_CXX TRUE)
|
||||
SET_DEFAULT(XSDK_ENABLE_Fortran TRUE)
|
||||
|
||||
# Handle the compiler and flags for a language
|
||||
MACRO(XSDK_HANDLE_LANG_DEFAULTS CMAKE_LANG_NAME ENV_LANG_NAME
|
||||
ENV_LANG_FLAGS_NAMES
|
||||
)
|
||||
|
||||
# Announce using env var ${ENV_LANG_NAME}
|
||||
IF (NOT "$ENV{${ENV_LANG_NAME}}" STREQUAL "" AND
|
||||
"${CMAKE_${CMAKE_LANG_NAME}_COMPILER}" STREQUAL ""
|
||||
)
|
||||
MESSAGE("-- " "XSDK: Setting CMAKE_${CMAKE_LANG_NAME}_COMPILER from env var"
|
||||
" ${ENV_LANG_NAME}='$ENV{${ENV_LANG_NAME}}'!")
|
||||
SET(CMAKE_${CMAKE_LANG_NAME}_COMPILER "$ENV{${ENV_LANG_NAME}}" CACHE FILEPATH
|
||||
"XSDK: Set by default from env var ${ENV_LANG_NAME}")
|
||||
ENDIF()
|
||||
|
||||
# Announce using env var ${ENV_LANG_FLAGS_NAME}
|
||||
FOREACH(ENV_LANG_FLAGS_NAME ${ENV_LANG_FLAGS_NAMES})
|
||||
IF (NOT "$ENV{${ENV_LANG_FLAGS_NAME}}" STREQUAL "" AND
|
||||
"${CMAKE_${CMAKE_LANG_NAME}_FLAGS}" STREQUAL ""
|
||||
)
|
||||
MESSAGE("-- " "XSDK: Setting CMAKE_${CMAKE_LANG_NAME}_FLAGS from env var"
|
||||
" ${ENV_LANG_FLAGS_NAME}='$ENV{${ENV_LANG_FLAGS_NAME}}'!")
|
||||
SET(CMAKE_${CMAKE_LANG_NAME}_FLAGS "$ENV{${ENV_LANG_FLAGS_NAME}} " CACHE STRING
|
||||
"XSDK: Set by default from env var ${ENV_LANG_FLAGS_NAME}")
|
||||
# NOTE: CMake adds the space after $ENV{${ENV_LANG_FLAGS_NAME}} so we
|
||||
# duplicate that here!
|
||||
ENDIF()
|
||||
ENDFOREACH()
|
||||
|
||||
ENDMACRO()
|
||||
|
||||
|
||||
#
|
||||
# Set XSDK Defaults
|
||||
#
|
||||
|
||||
# Set default compilers and flags
|
||||
IF (USE_XSDK_DEFAULTS)
|
||||
|
||||
# Handle env vars for languages C, C++, and Fortran
|
||||
|
||||
IF (XSDK_ENABLE_C)
|
||||
XSDK_HANDLE_LANG_DEFAULTS(C CC CFLAGS)
|
||||
ENDIF()
|
||||
|
||||
IF (XSDK_ENABLE_CXX)
|
||||
XSDK_HANDLE_LANG_DEFAULTS(CXX CXX CXXFLAGS)
|
||||
ENDIF()
|
||||
|
||||
IF (XSDK_ENABLE_Fortran)
|
||||
SET(ENV_FFLAGS "$ENV{FFLAGS}")
|
||||
SET(ENV_FCFLAGS "$ENV{FCFLAGS}")
|
||||
IF (
|
||||
(NOT "${ENV_FFLAGS}" STREQUAL "") AND (NOT "${ENV_FCFLAGS}" STREQUAL "")
|
||||
AND
|
||||
("${CMAKE_Fortran_FLAGS}" STREQUAL "")
|
||||
)
|
||||
IF (NOT "${ENV_FFLAGS}" STREQUAL "${ENV_FCFLAGS}")
|
||||
MESSAGE(FATAL_ERROR "Error, env vars FFLAGS='${ENV_FFLAGS}' and"
|
||||
" FCFLAGS='${ENV_FCFLAGS}' are both set in the env but are not equal!")
|
||||
ENDIF()
|
||||
ENDIF()
|
||||
XSDK_HANDLE_LANG_DEFAULTS(Fortran FC "FFLAGS;FCFLAGS")
|
||||
ENDIF()
|
||||
|
||||
# Set XSDK defaults for other CMake variables
|
||||
|
||||
IF ("${BUILD_SHARED_LIBS}" STREQUAL "")
|
||||
MESSAGE("-- " "XSDK: Setting default BUILD_SHARED_LIBS=TRUE")
|
||||
SET(BUILD_SHARED_LIBS TRUE CACHE BOOL "Set by default in XSDK mode")
|
||||
ENDIF()
|
||||
|
||||
IF ("${CMAKE_BUILD_TYPE}" STREQUAL "")
|
||||
MESSAGE("-- " "XSDK: Setting default CMAKE_BUILD_TYPE=DEBUG")
|
||||
SET(CMAKE_BUILD_TYPE DEBUG CACHE STRING "Set by default in XSDK mode")
|
||||
ENDIF()
|
||||
|
||||
ENDIF()
|
||||
@@ -12,42 +12,31 @@
|
||||
include(${CMAKE_CURRENT_LIST_DIR}/MFEMConfigVersion.cmake)
|
||||
|
||||
set(MFEM_VERSION ${PACKAGE_VERSION})
|
||||
set(MFEM_VERSION_INT @MFEM_VERSION@)
|
||||
set(MFEM_GIT_STRING "@MFEM_GIT_STRING@")
|
||||
|
||||
set(MFEM_USE_MPI @MFEM_USE_MPI@)
|
||||
set(MFEM_USE_METIS @MFEM_USE_METIS@)
|
||||
set(MFEM_USE_METIS_5 @MFEM_USE_METIS_5@)
|
||||
set(MFEM_DEBUG @MFEM_DEBUG@)
|
||||
set(MFEM_USE_EXCEPTIONS @MFEM_USE_EXCEPTIONS@)
|
||||
set(MFEM_USE_GZSTREAM @MFEM_USE_GZSTREAM@)
|
||||
set(MFEM_USE_LIBUNWIND @MFEM_USE_LIBUNWIND@)
|
||||
set(MFEM_USE_LAPACK @MFEM_USE_LAPACK@)
|
||||
set(MFEM_THREAD_SAFE @MFEM_THREAD_SAFE@)
|
||||
set(MFEM_USE_OPENMP @MFEM_USE_OPENMP@)
|
||||
set(MFEM_USE_MEMALLOC @MFEM_USE_MEMALLOC@)
|
||||
set(MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@)
|
||||
set(MFEM_USE_SUNDIALS @MFEM_USE_SUNDIALS@)
|
||||
set(MFEM_USE_MESQUITE @MFEM_USE_MESQUITE@)
|
||||
set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
|
||||
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
|
||||
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
|
||||
set(MFEM_USE_GECKO @MFEM_USE_GECKO@)
|
||||
set(MFEM_USE_GNUTLS @MFEM_USE_GNUTLS@)
|
||||
set(MFEM_USE_NETCDF @MFEM_USE_NETCDF@)
|
||||
set(MFEM_USE_PETSC @MFEM_USE_PETSC@)
|
||||
set(MFEM_USE_MPFR @MFEM_USE_MPFR@)
|
||||
set(MFEM_USE_SIDRE @MFEM_USE_SIDRE@)
|
||||
set(MFEM_USE_CONDUIT @MFEM_USE_CONDUIT@)
|
||||
set(MFEM_USE_PUMI @MFEM_USE_PUMI@)
|
||||
|
||||
set(MFEM_CXX_COMPILER "@CMAKE_CXX_COMPILER@")
|
||||
set(MFEM_CXX_FLAGS "@CMAKE_CXX_FLAGS@")
|
||||
|
||||
@PACKAGE_INIT@
|
||||
|
||||
set(MFEM_INCLUDE_DIRS "@PACKAGE_INCLUDE_INSTALL_DIRS@")
|
||||
foreach (dir ${MFEM_INCLUDE_DIRS})
|
||||
message("DIR = ${dir}")
|
||||
|
||||
set_and_check(MFEM_INCLUDE_DIR "${dir}")
|
||||
endforeach (dir "${MFEM_INCLUDE_DIRS}")
|
||||
|
||||
|
||||
@@ -12,27 +12,6 @@
|
||||
#ifndef MFEM_CONFIG_HEADER
|
||||
#define MFEM_CONFIG_HEADER
|
||||
|
||||
// MFEM version: integer of the form: (major*100 + minor)*100 + patch.
|
||||
#cmakedefine MFEM_VERSION @MFEM_VERSION@
|
||||
|
||||
// MFEM version string of the form "3.3" or "3.3.1".
|
||||
#cmakedefine MFEM_VERSION_STRING "@MFEM_VERSION_STRING@"
|
||||
|
||||
// MFEM version type, see the MFEM_VERSION_TYPE_* constants below.
|
||||
#define MFEM_VERSION_TYPE ((MFEM_VERSION)%2)
|
||||
|
||||
// MFEM version type constants.
|
||||
#define MFEM_VERSION_TYPE_RELEASE 0
|
||||
#define MFEM_VERSION_TYPE_DEVELOPMENT 1
|
||||
|
||||
// Separate MFEM version numbers for major, minor, and patch.
|
||||
#define MFEM_VERSION_MAJOR ((MFEM_VERSION)/10000)
|
||||
#define MFEM_VERSION_MINOR (((MFEM_VERSION)/100)%100)
|
||||
#define MFEM_VERSION_PATCH ((MFEM_VERSION)%100)
|
||||
|
||||
// Description of the git commit used to build MFEM.
|
||||
#cmakedefine MFEM_GIT_STRING "@MFEM_GIT_STRING@"
|
||||
|
||||
// Build the parallel MFEM library.
|
||||
// Requires an MPI compiler, and the libraries HYPRE and METIS.
|
||||
#cmakedefine MFEM_USE_MPI
|
||||
@@ -40,18 +19,12 @@
|
||||
// Enable debug checks in MFEM.
|
||||
#cmakedefine MFEM_DEBUG
|
||||
|
||||
// Throw an exception on errors.
|
||||
#cmakedefine MFEM_USE_EXCEPTIONS
|
||||
|
||||
// Enable gzstream in MFEM.
|
||||
#cmakedefine MFEM_USE_GZSTREAM
|
||||
|
||||
// Enable backtraces for mfem_error through libunwind.
|
||||
#cmakedefine MFEM_USE_LIBUNWIND
|
||||
|
||||
// Enable MFEM features that use the METIS library (parallel MFEM).
|
||||
#cmakedefine MFEM_USE_METIS
|
||||
|
||||
// Enable this option if linking with METIS version 5 (parallel MFEM).
|
||||
#cmakedefine MFEM_USE_METIS_5
|
||||
|
||||
@@ -74,9 +47,6 @@
|
||||
// Enable MFEM functionality based on the SuperLU_DIST library.
|
||||
#cmakedefine MFEM_USE_SUPERLU
|
||||
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
#cmakedefine MFEM_USE_STRUMPACK
|
||||
|
||||
// Internal MFEM option: enable group/batch allocation for some small objects.
|
||||
#cmakedefine MFEM_USE_MEMALLOC
|
||||
|
||||
@@ -95,14 +65,12 @@
|
||||
// Enable MFEM functionality based on the Sidre library
|
||||
#cmakedefine MFEM_USE_SIDRE
|
||||
|
||||
// Enable MFEM functionality based on Conduit
|
||||
#cmakedefine MFEM_USE_CONDUIT
|
||||
|
||||
// Enable MFEM functionality based on the PUMI library
|
||||
#cmakedefine MFEM_USE_PUMI
|
||||
|
||||
// Which library functions to use in class StopWatch for measuring time.
|
||||
// For a list of the available options, see INSTALL.
|
||||
// The available options are:
|
||||
// 0 - use std::clock from <ctime>
|
||||
// 1 - use times from <sys/times.h>
|
||||
// 2 - use high-resolution POSIX clocks
|
||||
// 3 - use QueryPerformanceCounter from <windows.h>
|
||||
// If not defined, an option is selected automatically.
|
||||
#define MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@
|
||||
|
||||
@@ -113,11 +81,4 @@
|
||||
// Macro needed to get defines like M_PI from <cmath>. (Visual Studio C++ only?)
|
||||
#cmakedefine _USE_MATH_DEFINES
|
||||
|
||||
// Version of HYPRE used for building MFEM.
|
||||
#cmakedefine MFEM_HYPRE_VERSION @MFEM_HYPRE_VERSION@
|
||||
|
||||
// Macro defined when PUMI is built with support for the Simmetrix SimModSuite
|
||||
// library.
|
||||
#cmakedefine MFEM_USE_SIMMETRIX
|
||||
|
||||
#endif // MFEM_CONFIG_HEADER
|
||||
|
||||
@@ -10,14 +10,15 @@
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Defines the following variables:
|
||||
# - AXOM_FOUND
|
||||
# - AXOM_LIBRARIES
|
||||
# - AXOM_INCLUDE_DIRS
|
||||
# - ATK_FOUND
|
||||
# - ATK_LIBRARIES
|
||||
# - ATK_INCLUDE_DIRS
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
# Note: components are enabled based on the find_package() parameters.
|
||||
mfem_find_package(Axom AXOM AXOM_DIR "include" "" "lib" ""
|
||||
"Paths to headers required by Axom." "Libraries required by Axom."
|
||||
mfem_find_package(ATK ATK ATK_DIR "include" "" "lib" ""
|
||||
"Paths to headers required by ATK." "Libraries required by ATK."
|
||||
ADD_COMPONENT Sidre "include" sidre/sidre.hpp "lib" sidre
|
||||
ADD_COMPONENT SPIO "include" spio/IOManager.hpp "lib" spio
|
||||
ADD_COMPONENT SLIC "include" slic/slic.hpp "lib" slic
|
||||
ADD_COMPONENT axom_utils "include" axom_utils/Utilities.hpp "lib" axom_utils)
|
||||
ADD_COMPONENT common "include" common/ATKMacros.hpp "lib" common)
|
||||
@@ -14,22 +14,9 @@
|
||||
# - CONDUIT_LIBRARIES
|
||||
# - CONDUIT_INCLUDE_DIRS
|
||||
|
||||
# check to see if relay requires hdf5, if so make sure to set HDF5
|
||||
# as a required dep
|
||||
if(EXISTS ${CONDUIT_DIR}/include/conduit/conduit_relay_hdf5.hpp)
|
||||
message(STATUS "Conduit Relay HDF5 Support is ENABLED")
|
||||
# we only need HDF5 if Conduit was built with HDF5 support
|
||||
set(Conduit_REQUIRED_PACKAGES "HDF5" CACHE STRING
|
||||
"Additional packages required by Conduit.")
|
||||
else()
|
||||
message(STATUS "Conduit Relay HDF5 Support is DISABLED")
|
||||
endif()
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(Conduit CONDUIT CONDUIT_DIR
|
||||
"include;include/conduit" conduit.hpp "lib" conduit
|
||||
"Paths to headers required by Conduit." "Libraries required by Conduit."
|
||||
ADD_COMPONENT relay
|
||||
"include;include/conduit" conduit_relay.hpp "lib" conduit_relay
|
||||
ADD_COMPONENT blueprint
|
||||
"include;include/conduit" conduit_blueprint.hpp "lib" conduit_blueprint)
|
||||
"include;include/conduit" conduit_relay.hpp "lib" conduit_relay)
|
||||
|
||||
@@ -13,23 +13,7 @@
|
||||
# - HYPRE_FOUND
|
||||
# - HYPRE_LIBRARIES
|
||||
# - HYPRE_INCLUDE_DIRS
|
||||
# - HYPRE_VERSION
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(HYPRE HYPRE HYPRE_DIR "include" "HYPRE.h" "lib" "HYPRE"
|
||||
"Paths to headers required by HYPRE." "Libraries required by HYPRE.")
|
||||
|
||||
if (HYPRE_FOUND AND (NOT HYPRE_VERSION))
|
||||
try_run(HYPRE_VERSION_RUN_RESULT HYPRE_VERSION_COMPILE_RESULT
|
||||
${CMAKE_CURRENT_BINARY_DIR}/config
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/config/get_hypre_version.cpp
|
||||
CMAKE_FLAGS -DINCLUDE_DIRECTORIES:STRING=${HYPRE_INCLUDE_DIRS}
|
||||
RUN_OUTPUT_VARIABLE HYPRE_VERSION_OUTPUT)
|
||||
if ((HYPRE_VERSION_RUN_RESULT EQUAL 0) AND HYPRE_VERSION_OUTPUT)
|
||||
string(STRIP "${HYPRE_VERSION_OUTPUT}" HYPRE_VERSION)
|
||||
set(HYPRE_VERSION ${HYPRE_VERSION} CACHE STRING "HYPRE version." FORCE)
|
||||
message(STATUS "Found HYPRE version ${HYPRE_VERSION}")
|
||||
else()
|
||||
message(FATAL_ERROR "Unable to determine HYPRE version.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -1,36 +0,0 @@
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Sets the following variables:
|
||||
# - STRUMPACK_FOUND
|
||||
# - STRUMPACK_INCLUDE_DIRS
|
||||
# - STRUMPACK_LIBRARIES
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(STRUMPACK STRUMPACK STRUMPACK_DIR
|
||||
"include" "StrumpackSparseSolverMPIDist.hpp"
|
||||
"lib" "strumpack;strumpack_sparse" # add NAMES_PER_DIR?
|
||||
"Paths to headers required by STRUMPACK."
|
||||
"Libraries required by STRUMPACK."
|
||||
CHECK_BUILD STRUMPACK_VERSION_OK TRUE
|
||||
"
|
||||
#include <StrumpackSparseSolverMPIDist.hpp>
|
||||
using namespace strumpack;
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
MPI_Init(&argc, &argv);
|
||||
MPI_Comm comm = MPI_COMM_WORLD;
|
||||
StrumpackSparseSolverMPIDist<double,int> solver(comm, argc, argv, false);
|
||||
solver.options().set_from_command_line();
|
||||
return 0;
|
||||
}
|
||||
"
|
||||
)
|
||||
@@ -1,29 +0,0 @@
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Sets the following variables:
|
||||
# - Scotch_FOUND
|
||||
# - Scotch_INCLUDE_DIRS
|
||||
# - Scotch_LIBRARIES
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(Scotch Scotch Scotch_DIR "" "" "" ""
|
||||
"Paths to headers required by Scotch."
|
||||
"Libraries required by Scotch."
|
||||
ADD_COMPONENT "scotch" "include" scotch.h "lib" scotch
|
||||
ADD_COMPONENT "scotcherr" "" "" "lib" scotcherr
|
||||
ADD_COMPONENT "scotcherrexit" "" "" "lib" scotcherrexit
|
||||
ADD_COMPONENT "scotchmetis" "include" "metis.h" "lib" scotchmetis
|
||||
ADD_COMPONENT "ptscotch" "include" ptscotch.h "lib" ptscotch
|
||||
ADD_COMPONENT "ptscotcherr" "" "" "lib" ptscotcherr
|
||||
ADD_COMPONENT "ptscotcherrexit" "" "" "lib" ptscotcherrexit
|
||||
ADD_COMPONENT "ptscotchparmetis" "include" "parmetis.h" "lib" ptscotchparmetis
|
||||
)
|
||||
@@ -9,30 +9,6 @@
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Function that converts a version string of the form 'major[.minor[.patch]]' to
|
||||
# the integer ((major * 100) + minor) * 100 + patch.
|
||||
function(mfem_version_to_int VersionString VersionIntVar)
|
||||
if ("${VersionString}" MATCHES "^([0-9]+)(.*)$")
|
||||
set(Major "${CMAKE_MATCH_1}")
|
||||
set(MinorPatchString "${CMAKE_MATCH_2}")
|
||||
else()
|
||||
set(Major 0)
|
||||
endif()
|
||||
if ("${MinorPatchString}" MATCHES "^\\.([0-9]+)(.*)$")
|
||||
set(Minor "${CMAKE_MATCH_1}")
|
||||
set(PatchString "${CMAKE_MATCH_2}")
|
||||
else()
|
||||
set(Minor 0)
|
||||
endif()
|
||||
if ("${PatchString}" MATCHES "^\\.([0-9]+)(.*)$")
|
||||
set(Patch "${CMAKE_MATCH_1}")
|
||||
else()
|
||||
set(Patch 0)
|
||||
endif()
|
||||
math(EXPR VersionInt "(${Major}*100+${Minor})*100+${Patch}")
|
||||
set(${VersionIntVar} ${VersionInt} PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
# A handy function to add the current source directory to a local
|
||||
# filename. To be used for creating a list of sources.
|
||||
function(convert_filenames_to_full_paths NAMES)
|
||||
@@ -74,8 +50,10 @@ function(add_mfem_examples EXE_SRCS)
|
||||
|
||||
string(REPLACE ".cpp" "" EXE_NAME "${EXE_PREFIX}${SRC_FILENAME}")
|
||||
add_executable(${EXE_NAME} ${SRC_FILE})
|
||||
add_dependencies(${MFEM_ALL_EXAMPLES_TARGET_NAME} ${EXE_NAME})
|
||||
if (EXE_NEEDED_BY)
|
||||
# If given a prefix, don't add the example to the list of examples to build.
|
||||
if (NOT EXE_PREFIX)
|
||||
add_dependencies(${MFEM_ALL_EXAMPLES_TARGET_NAME} ${EXE_NAME})
|
||||
elseif (EXE_NEEDED_BY)
|
||||
add_dependencies(${EXE_NEEDED_BY} ${EXE_NAME})
|
||||
endif()
|
||||
add_dependencies(${EXE_NAME}
|
||||
@@ -83,8 +61,7 @@ function(add_mfem_examples EXE_SRCS)
|
||||
|
||||
target_link_libraries(${EXE_NAME} mfem)
|
||||
if (MFEM_USE_MPI)
|
||||
# Not needed: (mfem already links with MPI_CXX_LIBRARIES)
|
||||
# target_link_libraries(${EXE_NAME} ${MPI_CXX_LIBRARIES})
|
||||
target_link_libraries(${EXE_NAME} ${MPI_CXX_LIBRARIES})
|
||||
|
||||
# Language-specific include directories:
|
||||
if (MPI_CXX_INCLUDE_PATH)
|
||||
@@ -155,7 +132,6 @@ function(add_mfem_miniapp MFEM_EXE_NAME)
|
||||
|
||||
# Handle the MPI separately
|
||||
if (MFEM_USE_MPI)
|
||||
# Add MPI_CXX_LIBRARIES, in case this target does not link with mfem.
|
||||
if(CMAKE_VERSION VERSION_GREATER 2.8.11)
|
||||
target_link_libraries(${MFEM_EXE_NAME} PRIVATE ${MPI_CXX_LIBRARIES})
|
||||
else()
|
||||
@@ -183,26 +159,26 @@ function(mfem_find_component Prefix DirVar IncSuffixes Header LibSuffixes Lib
|
||||
|
||||
if (Lib)
|
||||
if (${DirVar} OR EnvDirVar)
|
||||
find_library(${Prefix}_LIBRARY ${Lib}
|
||||
find_library(${Prefix}_LIBRARIES ${Lib}
|
||||
HINTS ${${DirVar}} ENV ${DirVar}
|
||||
PATH_SUFFIXES ${LibSuffixes}
|
||||
NO_DEFAULT_PATH
|
||||
DOC "${LibDoc}")
|
||||
endif()
|
||||
find_library(${Prefix}_LIBRARY ${Lib}
|
||||
find_library(${Prefix}_LIBRARIES ${Lib}
|
||||
PATH_SUFFIXES ${LibSuffixes}
|
||||
DOC "${LibDoc}")
|
||||
endif()
|
||||
|
||||
if (Header)
|
||||
if (${DirVar} OR EnvDirVar)
|
||||
find_path(${Prefix}_INCLUDE_DIR ${Header}
|
||||
find_path(${Prefix}_INCLUDE_DIRS ${Header}
|
||||
HINTS ${${DirVar}} ENV ${DirVar}
|
||||
PATH_SUFFIXES ${IncSuffixes}
|
||||
NO_DEFAULT_PATH
|
||||
DOC "${IncDoc}")
|
||||
endif()
|
||||
find_path(${Prefix}_INCLUDE_DIR ${Header}
|
||||
find_path(${Prefix}_INCLUDE_DIRS ${Header}
|
||||
PATH_SUFFIXES ${IncSuffixes}
|
||||
DOC "${IncDoc}")
|
||||
endif()
|
||||
@@ -214,9 +190,8 @@ endfunction(mfem_find_component)
|
||||
# successful, optionally checks building (compile + link) one or more given
|
||||
# code snippets. Additionally, a list of required/optional/alternative
|
||||
# packages (given by ${Name}_REQUIRED_PACKAGES) are searched for and added to
|
||||
# the ${Prefix}_INCLUDE_DIRS and ${Prefix}_LIBRARIES lists. The variable
|
||||
# ${Name}_REQUIRED_LIBRARIES can be set to spcecify any additional libraries
|
||||
# that are needed. This function defines the following CACHE variables:
|
||||
# the ${Prefix}_INCLUDE_DIRS and ${Prefix}_LIBRARIES lists. The function
|
||||
# defines the following CACHE variables:
|
||||
#
|
||||
# ${Prefix}_FOUND
|
||||
# ${Prefix}_INCLUDE_DIRS
|
||||
@@ -255,14 +230,12 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
mfem_find_component("${Prefix}" "${DirVar}" "${IncSuffixes}" "${Header}"
|
||||
"${LibSuffixes}" "${Lib}" "${IncDoc}" "${LibDoc}")
|
||||
|
||||
if (((NOT Lib) OR ${Prefix}_LIBRARY) AND
|
||||
((NOT Header) OR ${Prefix}_INCLUDE_DIR))
|
||||
if (((NOT Lib) OR ${Prefix}_LIBRARIES) AND
|
||||
((NOT Header) OR ${Prefix}_INCLUDE_DIRS))
|
||||
set(Found TRUE)
|
||||
else()
|
||||
set(Found FALSE)
|
||||
endif()
|
||||
set(${Prefix}_LIBRARIES ${${Prefix}_LIBRARY})
|
||||
set(${Prefix}_INCLUDE_DIRS ${${Prefix}_INCLUDE_DIR})
|
||||
|
||||
set(ReqVars "")
|
||||
|
||||
@@ -301,22 +274,25 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
"${CompLibSuffixes}" "${CompLib}" "" "")
|
||||
if (CompRequired)
|
||||
if (CompLib)
|
||||
list(APPEND ReqVars ${FullPrefix}_LIBRARY)
|
||||
list(APPEND ReqVars ${FullPrefix}_LIBRARIES)
|
||||
endif()
|
||||
if (CompHeader)
|
||||
list(APPEND ReqVars ${FullPrefix}_INCLUDE_DIR)
|
||||
list(APPEND ReqVars ${FullPrefix}_INCLUDE_DIRS)
|
||||
endif()
|
||||
endif(CompRequired)
|
||||
if (((NOT CompLib) OR ${FullPrefix}_LIBRARY) AND
|
||||
((NOT CompHeader) OR ${FullPrefix}_INCLUDE_DIR))
|
||||
if (((NOT CompLib) OR ${FullPrefix}_LIBRARIES) AND
|
||||
((NOT CompHeader) OR ${FullPrefix}_INCLUDE_DIRS))
|
||||
# Component found
|
||||
list(APPEND ${Prefix}_LIBRARIES ${${FullPrefix}_LIBRARY})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${${FullPrefix}_INCLUDE_DIR})
|
||||
set(${FullPrefix}_FOUND TRUE CACHE BOOL
|
||||
"${Name}/${CompPrefix} was found." FORCE)
|
||||
list(APPEND ${Prefix}_LIBRARIES ${${FullPrefix}_LIBRARIES})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${${FullPrefix}_INCLUDE_DIRS})
|
||||
if (NOT ${Name}_FIND_QUIETLY)
|
||||
# message(STATUS "${Name}: ${CompPrefix}: found")
|
||||
message(STATUS
|
||||
"${Name}: ${CompPrefix}: ${${FullPrefix}_LIBRARY}")
|
||||
"${Name}: ${CompPrefix}: ${${FullPrefix}_LIBRARIES}")
|
||||
# message(STATUS
|
||||
# "${Name}: ${CompPrefix}: ${${FullPrefix}_INCLUDE_DIR}")
|
||||
# "${Name}: ${CompPrefix}: ${${FullPrefix}_INCLUDE_DIRS}")
|
||||
endif()
|
||||
else()
|
||||
# Let FindPackageHandleStandardArgs() handle errors
|
||||
@@ -369,8 +345,6 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
if (NOT ${Name}_FIND_QUIETLY)
|
||||
message(STATUS "${Name}: trying alternative package: ${ReqPackM}")
|
||||
endif()
|
||||
# Do not add ${Required} here, since that will prevent other potential
|
||||
# alternative packages from being found.
|
||||
find_package(${ReqPack} ${Quiet} COMPONENTS ${PackComps})
|
||||
string(TOUPPER ${ReqPack} ReqPACK)
|
||||
if (${ReqPack}_FOUND)
|
||||
@@ -384,7 +358,7 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
endif()
|
||||
elseif (Alternative)
|
||||
set(Alternative FALSE)
|
||||
elseif (Found)
|
||||
else()
|
||||
if (NOT ${Name}_FIND_QUIETLY)
|
||||
if (Required)
|
||||
message(STATUS "${Name}: looking for required package: ${ReqPackM}")
|
||||
@@ -394,139 +368,23 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
endif()
|
||||
string(TOUPPER ${ReqPack} ReqPACK)
|
||||
if (NOT (${ReqPack}_FOUND OR ${ReqPACK}_FOUND))
|
||||
if (NOT ${ReqPack}_TARGET_NAMES)
|
||||
find_package(${ReqPack} ${Required} ${Quiet} COMPONENTS ${PackComps})
|
||||
else()
|
||||
foreach(_target ${ReqPack} ${${ReqPack}_TARGET_NAMES})
|
||||
# Do not use ${Required} here:
|
||||
find_package(${_target} NAMES ${_target} ${ReqPack} ${Quiet}
|
||||
COMPONENTS ${PackComps})
|
||||
string(TOUPPER ${_target} _TARGET)
|
||||
if (${_target}_FOUND OR ${_TARGET}_FOUND)
|
||||
set(${ReqPack}_FOUND TRUE)
|
||||
break()
|
||||
endif()
|
||||
endforeach()
|
||||
if (${Required} AND NOT ${ReqPack}_FOUND)
|
||||
message(FATAL_ERROR " *** Required package ${ReqPack} not found."
|
||||
"Checked target names: ${ReqPack} ${${ReqPack}_TARGET_NAMES}")
|
||||
endif()
|
||||
endif()
|
||||
find_package(${ReqPack} ${Required} ${Quiet} COMPONENTS ${PackComps})
|
||||
endif()
|
||||
if (Required AND NOT (${ReqPack}_FOUND OR ${ReqPACK}_FOUND))
|
||||
message(FATAL_ERROR " --------- INTERNAL ERROR")
|
||||
endif()
|
||||
if ("${ReqPack}" STREQUAL "MPI" AND MPI_CXX_FOUND)
|
||||
if ("${ReqPack}" STREQUAL "MPI")
|
||||
list(APPEND ${Prefix}_LIBRARIES ${MPI_CXX_LIBRARIES})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${MPI_CXX_INCLUDE_PATH})
|
||||
elseif (${ReqPack}_FOUND OR ${ReqPACK}_FOUND)
|
||||
else()
|
||||
if (${ReqPack}_FOUND)
|
||||
set(_Pack ${ReqPack})
|
||||
else()
|
||||
set(_Pack ${ReqPACK})
|
||||
list(APPEND ${Prefix}_LIBRARIES ${${ReqPack}_LIBRARIES})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${${ReqPack}_INCLUDE_DIRS})
|
||||
elseif (${ReqPACK}_FOUND)
|
||||
list(APPEND ${Prefix}_LIBRARIES ${${ReqPACK}_LIBRARIES})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${${ReqPACK}_INCLUDE_DIRS})
|
||||
endif()
|
||||
set(_Pack_LIBS)
|
||||
set(_Pack_INCS)
|
||||
# - ${_Pack}_CONFIG is defined by find_package() when a config file was
|
||||
# loaded
|
||||
# - If ${ReqPack}_TARGET_NAMES is defined, use target mode
|
||||
if (NOT ((DEFINED ${_Pack}_CONFIG) OR
|
||||
(DEFINED ${ReqPack}_TARGET_NAMES)))
|
||||
# Defined variables expected:
|
||||
# - ${ReqPack}_LIB_VARS, optional, default: ${_Pack}_LIBRARIES
|
||||
# - ${ReqPack}_INCLUDE_VARS, optional, default: ${_Pack}_INCLUDE_DIRS
|
||||
set(_lib_vars ${${ReqPack}_LIB_VARS})
|
||||
if (NOT _lib_vars)
|
||||
set(_lib_vars ${_Pack}_LIBRARIES)
|
||||
endif()
|
||||
foreach (_var ${_lib_vars})
|
||||
if (${_var})
|
||||
list(APPEND _Pack_LIBS ${${_var}})
|
||||
endif()
|
||||
endforeach()
|
||||
# Includes
|
||||
set(_inc_vars ${${ReqPack}_INCLUDE_VARS})
|
||||
if (NOT _inc_vars)
|
||||
set(_inc_vars ${_Pack}_INCLUDE_DIRS)
|
||||
endif()
|
||||
foreach (_include ${_inc_vars})
|
||||
# message(STATUS "${Name}: ${ReqPack}: ${_include}")
|
||||
if (${_include})
|
||||
list(APPEND _Pack_INCS ${${_include}})
|
||||
endif()
|
||||
endforeach()
|
||||
else()
|
||||
# Target mode: check for a valid target:
|
||||
# - an entry in the variable ${ReqPack}_TARGET_NAMES (optional)
|
||||
# - ${_Pack}
|
||||
# Other optional variables:
|
||||
# - ${ReqPack}_IMPORT_CONFIG, default value: "RELEASE"
|
||||
# - ${ReqPack}_TARGET_FORCE, default value: "FALSE"
|
||||
set(TargetName)
|
||||
foreach (_target ${${ReqPack}_TARGET_NAMES} ${_Pack})
|
||||
if (TARGET ${_target})
|
||||
set(TargetName ${_target})
|
||||
break()
|
||||
endif()
|
||||
endforeach()
|
||||
if ("${TargetName}" STREQUAL "")
|
||||
message(FATAL_ERROR " *** ${ReqPack}: unknown target. "
|
||||
"Please set ${ReqPack}_TARGET_NAMES.")
|
||||
endif()
|
||||
get_target_property(IsImported ${TargetName} IMPORTED)
|
||||
if (IsImported)
|
||||
set(ImportConfig ${${ReqPack}_IMPORT_CONFIG})
|
||||
if (NOT ImportConfig)
|
||||
set(ImportConfig RELEASE)
|
||||
endif()
|
||||
get_target_property(ImpConfigs ${TargetName} IMPORTED_CONFIGURATIONS)
|
||||
list(FIND ImpConfigs ${ImportConfig} _Index)
|
||||
if (_Index EQUAL -1)
|
||||
message(FATAL_ERROR " *** ${ReqPack}: configuration "
|
||||
"${ImportConfig} not found. Set ${ReqPack}_IMPORT_CONFIG "
|
||||
"from the list: ${ImpConfigs}.")
|
||||
endif()
|
||||
endif()
|
||||
# Set _Pack_LIBS
|
||||
if (NOT IsImported OR ${ReqPack}_TARGET_FORCE)
|
||||
# Set _Pack_LIBS to be the target itself
|
||||
set(_Pack_LIBS ${TargetName})
|
||||
if (NOT ${Name}_FIND_QUIETLY)
|
||||
message(STATUS "Found ${ReqPack}: ${_Pack_LIBS} (target)")
|
||||
endif()
|
||||
else()
|
||||
# Set _Pack_LIBS from the target properties for ImportConfig
|
||||
foreach (_prop IMPORTED_LOCATION_${ImportConfig}
|
||||
IMPORTED_LINK_INTERFACE_LIBRARIES_${ImportConfig})
|
||||
get_target_property(_value ${TargetName} ${_prop})
|
||||
if (_value)
|
||||
list(APPEND _Pack_LIBS ${_value})
|
||||
endif()
|
||||
endforeach()
|
||||
if (NOT ${Name}_FIND_QUIETLY)
|
||||
message(STATUS
|
||||
"Imported ${ReqPack}[${ImportConfig}]: ${_Pack_LIBS}")
|
||||
endif()
|
||||
endif()
|
||||
# Set _Pack_INCS
|
||||
foreach (_prop INCLUDE_DIRECTORIES)
|
||||
get_target_property(_value ${TargetName} ${_prop})
|
||||
if (_value)
|
||||
list(APPEND _Pack_INCS ${_value})
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
# _Pack_LIBS and _Pack_INCS should be fully defined here
|
||||
list(APPEND ${Prefix}_LIBRARIES ${_Pack_LIBS})
|
||||
list(APPEND ${Prefix}_INCLUDE_DIRS ${_Pack_INCS})
|
||||
endif()
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if (Found AND ${Name}_REQUIRED_LIBRARIES)
|
||||
list(APPEND ${Prefix}_LIBRARIES ${${Name}_REQUIRED_LIBRARIES})
|
||||
endif()
|
||||
|
||||
if (NOT ("${${Prefix}_INCLUDE_DIRS}" STREQUAL ""))
|
||||
list(INSERT ReqVars 0 ${Prefix}_INCLUDE_DIRS)
|
||||
set(ReqHeaders 1)
|
||||
@@ -543,6 +401,12 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
if (ReqHeaders)
|
||||
list(REMOVE_DUPLICATES ${Prefix}_INCLUDE_DIRS)
|
||||
endif()
|
||||
# Write the updated values to the cache.
|
||||
set(${Prefix}_LIBRARIES ${${Prefix}_LIBRARIES} CACHE STRING
|
||||
"${LibDoc}" FORCE)
|
||||
set(${Prefix}_INCLUDE_DIRS ${${Prefix}_INCLUDE_DIRS} CACHE STRING
|
||||
"${IncDoc}" FORCE)
|
||||
set(${Prefix}_FOUND TRUE CACHE BOOL "${Name} was found." FORCE)
|
||||
|
||||
# Check for optional "CHECK_BUILD" arguments.
|
||||
set(I 9) # 9 is the number of required arguments
|
||||
@@ -560,10 +424,6 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
set(CMAKE_REQUIRED_QUIET ${${Name}_FIND_QUIETLY})
|
||||
check_cxx_source_compiles("${TestSrc}" ${TestVar})
|
||||
if (TestReq)
|
||||
if (NOT ${TestVar})
|
||||
set(Found FALSE)
|
||||
unset(${TestVar} CACHE)
|
||||
endif()
|
||||
list(APPEND ReqVars ${TestVar})
|
||||
endif()
|
||||
elseif("${ARGV${I}}" STREQUAL "ADD_COMPONENT")
|
||||
@@ -574,35 +434,22 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
|
||||
endif()
|
||||
math(EXPR I "${I}+1")
|
||||
endwhile()
|
||||
else()
|
||||
set(${Prefix}_FOUND FALSE CACHE BOOL "${Name} was not found." FORCE)
|
||||
endif()
|
||||
if ("_x_${ReqVars}" STREQUAL "_x_")
|
||||
set(${Prefix}_FOUND ${Found})
|
||||
set(ReqVars ${Prefix}_FOUND)
|
||||
endif()
|
||||
# foreach(ReqVar ${ReqVars})
|
||||
# message(STATUS " *** ${ReqVar}=${${ReqVar}}")
|
||||
# get_property(IsCached CACHE ${ReqVar} PROPERTY "VALUE" SET)
|
||||
# if (IsCached)
|
||||
# get_property(CachedVal CACHE ${ReqVar} PROPERTY "VALUE")
|
||||
# message(STATUS " *** ${ReqVar}[cached]=${CachedVal}")
|
||||
# endif()
|
||||
# message(STATUS "${ReqVar}=${${ReqVar}}")
|
||||
# endforeach()
|
||||
|
||||
include(FindPackageHandleStandardArgs)
|
||||
find_package_handle_standard_args(${Name}
|
||||
" *** ${Name} not found. Please set ${DirVar}." ${ReqVars})
|
||||
|
||||
string(TOUPPER ${Name} UName)
|
||||
if (${UName}_FOUND)
|
||||
# Write the ${Prefix}_* variables to the cache.
|
||||
set(${Prefix}_LIBRARIES ${${Prefix}_LIBRARIES} CACHE STRING
|
||||
"${LibDoc}" FORCE)
|
||||
set(${Prefix}_INCLUDE_DIRS ${${Prefix}_INCLUDE_DIRS} CACHE STRING
|
||||
"${IncDoc}" FORCE)
|
||||
set(${Prefix}_FOUND TRUE CACHE BOOL "${Name} was found." FORCE)
|
||||
if (ReqHeaders AND (NOT ${Name}_FIND_QUIETLY))
|
||||
message(STATUS "${Prefix}_INCLUDE_DIRS=${${Prefix}_INCLUDE_DIRS}")
|
||||
endif()
|
||||
if (Found AND ReqLibs AND ReqHeaders AND (NOT ${Name}_FIND_QUIETLY))
|
||||
message(STATUS "${Prefix}_INCLUDE_DIRS=${${Prefix}_INCLUDE_DIRS}")
|
||||
endif()
|
||||
|
||||
endfunction(mfem_find_package)
|
||||
|
||||
@@ -30,21 +30,7 @@
|
||||
#ifdef MFEM_USE_SUPERLU
|
||||
#error Building with SuperLU_DIST (MFEM_USE_SUPERLU=YES) requires MPI (MFEM_USE_MPI=YES)
|
||||
#endif
|
||||
#ifdef MFEM_USE_STRUMPACK
|
||||
#error Building with STRUMPACK (MFEM_USE_STRUMPACK=YES) requires MPI (MFEM_USE_MPI=YES)
|
||||
#endif
|
||||
#ifdef MFEM_USE_PETSC
|
||||
#error Building with PETSc (MFEM_USE_PETSC=YES) requires MPI (MFEM_USE_MPI=YES)
|
||||
#endif
|
||||
#ifdef MFEM_USE_PUMI
|
||||
#error Building with PUMI (MFEM_USE_PUMI=YES) requires MPI (MFEM_USE_MPI=YES)
|
||||
#endif
|
||||
#endif // MFEM_USE_MPI not defined
|
||||
|
||||
// Macro that returns its first arg when MFEM_USE_BACKENDS is defined, and its
|
||||
// second arg if it is not defined.
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
#define MFEM_IF_BACKENDS(x,y) x
|
||||
#else
|
||||
#define MFEM_IF_BACKENDS(x,y) y
|
||||
#endif
|
||||
|
||||
+5
-56
@@ -12,33 +12,6 @@
|
||||
#ifndef MFEM_CONFIG_HEADER
|
||||
#define MFEM_CONFIG_HEADER
|
||||
|
||||
// MFEM version: integer of the form: (major*100 + minor)*100 + patch.
|
||||
// #define MFEM_VERSION @MFEM_VERSION@
|
||||
|
||||
// MFEM version string of the form "3.3" or "3.3.1".
|
||||
// #define MFEM_VERSION_STRING "@MFEM_VERSION_STRING@"
|
||||
|
||||
// MFEM version type, see the MFEM_VERSION_TYPE_* constants below.
|
||||
#define MFEM_VERSION_TYPE ((MFEM_VERSION)%2)
|
||||
|
||||
// MFEM version type constants.
|
||||
#define MFEM_VERSION_TYPE_RELEASE 0
|
||||
#define MFEM_VERSION_TYPE_DEVELOPMENT 1
|
||||
|
||||
// Separate MFEM version numbers for major, minor, and patch.
|
||||
#define MFEM_VERSION_MAJOR ((MFEM_VERSION)/10000)
|
||||
#define MFEM_VERSION_MINOR (((MFEM_VERSION)/100)%100)
|
||||
#define MFEM_VERSION_PATCH ((MFEM_VERSION)%100)
|
||||
|
||||
// Description of the git commit used to build MFEM.
|
||||
// #define MFEM_GIT_STRING "@MFEM_GIT_STRING@"
|
||||
|
||||
// The absolute path of the MFEM source prefix
|
||||
// #define MFEM_SOURCE_DIR "@MFEM_SOURCE_DIR@"
|
||||
|
||||
// The absolute path of the MFEM installation prefix
|
||||
// #define MFEM_INSTALL_DIR "@MFEM_INSTALL_DIR@"
|
||||
|
||||
// Build the parallel MFEM library.
|
||||
// Requires an MPI compiler, and the libraries HYPRE and METIS.
|
||||
// #define MFEM_USE_MPI
|
||||
@@ -46,18 +19,12 @@
|
||||
// Enable debug checks in MFEM.
|
||||
// #define MFEM_DEBUG
|
||||
|
||||
// Throw an exception on errors.
|
||||
// #define MFEM_USE_EXCEPTIONS
|
||||
|
||||
// Enable gzstream in MFEM.
|
||||
// #define MFEM_USE_GZSTREAM
|
||||
|
||||
// Enable backtraces for mfem_error through libunwind.
|
||||
// #define MFEM_USE_LIBUNWIND
|
||||
|
||||
// Enable MFEM features that use the METIS library (parallel MFEM).
|
||||
// #define MFEM_USE_METIS
|
||||
|
||||
// Enable this option if linking with METIS version 5 (parallel MFEM).
|
||||
// #define MFEM_USE_METIS_5
|
||||
|
||||
@@ -75,7 +42,11 @@
|
||||
// #define MFEM_USE_MEMALLOC
|
||||
|
||||
// Which library functions to use in class StopWatch for measuring time.
|
||||
// For a list of the available options, see INSTALL.
|
||||
// The available options are:
|
||||
// 0 - use std::clock from <ctime>
|
||||
// 1 - use times from <sys/times.h>
|
||||
// 2 - use high-resolution POSIX clocks
|
||||
// 3 - use QueryPerformanceCounter from <windows.h>
|
||||
// If not defined, an option is selected automatically.
|
||||
// #define MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@
|
||||
|
||||
@@ -91,9 +62,6 @@
|
||||
// Enable MFEM functionality based on the SuperLU library.
|
||||
// #define MFEM_USE_SUPERLU
|
||||
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
// #define MFEM_USE_STRUMPACK
|
||||
|
||||
// Enable functionality based on the Gecko library
|
||||
// #define MFEM_USE_GECKO
|
||||
|
||||
@@ -103,9 +71,6 @@
|
||||
// Enable Sidre support
|
||||
// #define MFEM_USE_SIDRE
|
||||
|
||||
// Enable Conduit support
|
||||
// #define MFEM_USE_CONDUIT
|
||||
|
||||
// Enable functionality based on the NetCDF library (reading CUBIT files)
|
||||
// #define MFEM_USE_NETCDF
|
||||
|
||||
@@ -115,26 +80,10 @@
|
||||
// Enable functionality based on the MPFR library.
|
||||
// #define MFEM_USE_MPFR
|
||||
|
||||
// Enable MFEM functionality based on the PUMI library
|
||||
// #define MFEM_USE_PUMI
|
||||
|
||||
// Enable the use of MFEM backends.
|
||||
// #define MFEM_USE_BACKENDS
|
||||
|
||||
// Enable the OCCA backend.
|
||||
// #define MFEM_USE_OCCA
|
||||
|
||||
// Windows specific options
|
||||
#ifdef _WIN32
|
||||
// Macro needed to get defines like M_PI from <cmath>. (Visual Studio C++ only?)
|
||||
#define _USE_MATH_DEFINES
|
||||
#endif
|
||||
|
||||
// Version of HYPRE used for building MFEM.
|
||||
// #define MFEM_HYPRE_VERSION @MFEM_HYPRE_VERSION@
|
||||
|
||||
// Macro defined when PUMI is built with support for the Simmetrix SimModSuite
|
||||
// library.
|
||||
// #define MFEM_USE_SIMMETRIX
|
||||
|
||||
#endif // MFEM_CONFIG_HEADER
|
||||
|
||||
@@ -10,16 +10,9 @@
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Variables corresponding to defines in config.hpp (YES, NO, or value)
|
||||
MFEM_VERSION = @MFEM_VERSION@
|
||||
MFEM_VERSION_STRING = @MFEM_VERSION_STRING@
|
||||
MFEM_GIT_STRING = @MFEM_GIT_STRING@
|
||||
MFEM_SOURCE_DIR = @MFEM_SOURCE_DIR@
|
||||
MFEM_INSTALL_DIR = @MFEM_INSTALL_DIR@
|
||||
MFEM_USE_MPI = @MFEM_USE_MPI@
|
||||
MFEM_USE_METIS = @MFEM_USE_METIS@
|
||||
MFEM_USE_METIS_5 = @MFEM_USE_METIS_5@
|
||||
MFEM_DEBUG = @MFEM_DEBUG@
|
||||
MFEM_USE_EXCEPTIONS = @MFEM_USE_EXCEPTIONS@
|
||||
MFEM_USE_GZSTREAM = @MFEM_USE_GZSTREAM@
|
||||
MFEM_USE_LIBUNWIND = @MFEM_USE_LIBUNWIND@
|
||||
MFEM_USE_LAPACK = @MFEM_USE_LAPACK@
|
||||
@@ -31,17 +24,12 @@ MFEM_USE_SUNDIALS = @MFEM_USE_SUNDIALS@
|
||||
MFEM_USE_MESQUITE = @MFEM_USE_MESQUITE@
|
||||
MFEM_USE_SUITESPARSE = @MFEM_USE_SUITESPARSE@
|
||||
MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
|
||||
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
|
||||
MFEM_USE_GECKO = @MFEM_USE_GECKO@
|
||||
MFEM_USE_GNUTLS = @MFEM_USE_GNUTLS@
|
||||
MFEM_USE_NETCDF = @MFEM_USE_NETCDF@
|
||||
MFEM_USE_PETSC = @MFEM_USE_PETSC@
|
||||
MFEM_USE_MPFR = @MFEM_USE_MPFR@
|
||||
MFEM_USE_SIDRE = @MFEM_USE_SIDRE@
|
||||
MFEM_USE_CONDUIT = @MFEM_USE_CONDUIT@
|
||||
MFEM_USE_PUMI = @MFEM_USE_PUMI@
|
||||
MFEM_USE_BACKENDS = @MFEM_USE_BACKENDS@
|
||||
MFEM_USE_OCCA = @MFEM_USE_OCCA@
|
||||
|
||||
# Compiler, compile options, and link options
|
||||
MFEM_CXX = @MFEM_CXX@
|
||||
@@ -49,25 +37,17 @@ MFEM_CPPFLAGS = @MFEM_CPPFLAGS@
|
||||
MFEM_CXXFLAGS = @MFEM_CXXFLAGS@
|
||||
MFEM_TPLFLAGS = @MFEM_TPLFLAGS@
|
||||
MFEM_INCFLAGS = @MFEM_INCFLAGS@
|
||||
MFEM_PICFLAG = @MFEM_PICFLAG@
|
||||
MFEM_FLAGS = @MFEM_FLAGS@
|
||||
MFEM_EXT_LIBS = @MFEM_EXT_LIBS@
|
||||
MFEM_LIBS = @MFEM_LIBS@
|
||||
MFEM_LIB_FILE = @MFEM_LIB_FILE@
|
||||
MFEM_STATIC = @MFEM_STATIC@
|
||||
MFEM_SHARED = @MFEM_SHARED@
|
||||
MFEM_BUILD_TAG = @MFEM_BUILD_TAG@
|
||||
MFEM_PREFIX = @MFEM_PREFIX@
|
||||
MFEM_INC_DIR = @MFEM_INC_DIR@
|
||||
MFEM_LIB_DIR = @MFEM_LIB_DIR@
|
||||
|
||||
# Location of test.mk
|
||||
MFEM_TEST_MK = @MFEM_TEST_MK@
|
||||
|
||||
# Command used to launch MPI jobs
|
||||
MFEM_MPIEXEC = @MFEM_MPIEXEC@
|
||||
MFEM_MPIEXEC_NP = @MFEM_MPIEXEC_NP@
|
||||
MFEM_MPI_NP = @MFEM_MPI_NP@
|
||||
|
||||
# Optional extra configuration
|
||||
@MFEM_CONFIG_EXTRA@
|
||||
|
||||
+11
-44
@@ -18,10 +18,8 @@ if (NOT CMAKE_BUILD_TYPE)
|
||||
"Build type: Debug, Release, RelWithDebInfo, or MinSizeRel." FORCE)
|
||||
endif()
|
||||
|
||||
# MFEM options. Set to mimic the default "defaults.mk" file.
|
||||
# MFEM options. Set to mimic the default "default.mk" file.
|
||||
option(MFEM_USE_MPI "Enable MPI parallel build" OFF)
|
||||
option(MFEM_USE_METIS "Enable METIS usage" ${MFEM_USE_MPI})
|
||||
option(MFEM_USE_EXCEPTIONS "Enable the use of exceptions" OFF)
|
||||
option(MFEM_USE_GZSTREAM "Enable gzstream for compressed data streams." OFF)
|
||||
option(MFEM_USE_LIBUNWIND "Enable backtrace for errors." OFF)
|
||||
option(MFEM_USE_LAPACK "Enable LAPACK usage" OFF)
|
||||
@@ -32,21 +30,17 @@ option(MFEM_USE_SUNDIALS "Enable SUNDIALS usage" OFF)
|
||||
option(MFEM_USE_MESQUITE "Enable MESQUITE usage" OFF)
|
||||
option(MFEM_USE_SUITESPARSE "Enable SuiteSparse usage" OFF)
|
||||
option(MFEM_USE_SUPERLU "Enable SuperLU_DIST usage" OFF)
|
||||
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
|
||||
option(MFEM_USE_GECKO "Enable GECKO usage" OFF)
|
||||
option(MFEM_USE_GNUTLS "Enable GNUTLS usage" OFF)
|
||||
option(MFEM_USE_NETCDF "Enable NETCDF usage" OFF)
|
||||
option(MFEM_USE_PETSC "Enable PETSc support." OFF)
|
||||
option(MFEM_USE_MPFR "Enable MPFR usage." OFF)
|
||||
option(MFEM_USE_SIDRE "Enable Axom/Sidre usage" OFF)
|
||||
option(MFEM_USE_CONDUIT "Enable Conduit usage" OFF)
|
||||
option(MFEM_USE_PUMI "Enable PUMI" OFF)
|
||||
option(MFEM_USE_SIDRE "Enable ATK/Sidre usage" OFF)
|
||||
|
||||
# Allow a user to disable testing, examples, and/or miniapps at CONFIGURE TIME
|
||||
# if they don't want/need them (e.g. if MFEM is "just a dependency" and all they
|
||||
# need is the library, building all that stuff adds unnecessary overhead). Note
|
||||
# that the examples or miniapps can always be built using the targets 'examples'
|
||||
# or 'miniapps', respectively.
|
||||
# need is the library, building all that stuff adds unnecessary overhead). To
|
||||
# match "makefile" behavior, they are all enabled by default.
|
||||
option(MFEM_ENABLE_TESTING "Enable the ctest framework for testing" ON)
|
||||
option(MFEM_ENABLE_EXAMPLES "Build all of the examples" OFF)
|
||||
option(MFEM_ENABLE_MINIAPPS "Build all of the miniapps" OFF)
|
||||
@@ -72,7 +66,7 @@ set(METIS_DIR "${MFEM_DIR}/../metis-4.0" CACHE PATH "Path to the METIS library."
|
||||
|
||||
set(LIBUNWIND_DIR "" CACHE PATH "Path to Libunwind.")
|
||||
|
||||
set(SUNDIALS_DIR "${MFEM_DIR}/../sundials-3.0.0" CACHE PATH
|
||||
set(SUNDIALS_DIR "${MFEM_DIR}/../sundials-2.7.0" CACHE PATH
|
||||
"Path to the SUNDIALS library.")
|
||||
# The following may be necessary, if SUNDIALS was built with KLU:
|
||||
# set(SUNDIALS_REQUIRED_PACKAGES "SuiteSparse/KLU/AMD/BTF/COLAMD/config"
|
||||
@@ -97,32 +91,6 @@ set(SuperLUDist_DIR "${MFEM_DIR}/../SuperLU_DIST_5.1.0" CACHE PATH
|
||||
set(SuperLUDist_REQUIRED_PACKAGES "MPI" "BLAS" "ParMETIS" CACHE STRING
|
||||
"Additional packages required by SuperLU_DIST.")
|
||||
|
||||
set(STRUMPACK_DIR "${MFEM_DIR}/../STRUMPACK-build" CACHE PATH
|
||||
"Path to the STRUMPACK library.")
|
||||
# STRUMPACK may also depend on "OpenMP", depending on how it was compiled.
|
||||
set(STRUMPACK_REQUIRED_PACKAGES "MPI" "MPI_Fortran" "ParMETIS" "METIS"
|
||||
"ScaLAPACK" "Scotch/ptscotch/ptscotcherr/scotch/scotcherr" CACHE STRING
|
||||
"Additional packages required by STRUMPACK.")
|
||||
# If the MPI package does not find all required Fortran libraries:
|
||||
# set(STRUMPACK_REQUIRED_LIBRARIES "gfortran" "mpi_mpifh" CACHE STRING
|
||||
# "Additional libraries required by STRUMPACK.")
|
||||
|
||||
# The Scotch library, required by STRUMPACK
|
||||
set(Scotch_DIR "${MFEM_DIR}/../scotch_6.0.4" CACHE PATH
|
||||
"Path to the Scotch and PT-Scotch libraries.")
|
||||
set(Scotch_REQUIRED_PACKAGES "Threads" CACHE STRING
|
||||
"Additional packages required by Scotch.")
|
||||
# Tell the "Threads" package/module to prefer pthreads.
|
||||
set(CMAKE_THREAD_PREFER_PTHREAD TRUE)
|
||||
set(Threads_LIB_VARS CMAKE_THREAD_LIBS_INIT)
|
||||
|
||||
# The ScaLAPACK library, required by STRUMPACK
|
||||
set(ScaLAPACK_DIR "${MFEM_DIR}/../scalapack-2.0.2/lib/cmake/scalapack-2.0.2"
|
||||
CACHE PATH "Path to the configuration file scalapack-config.cmake")
|
||||
set(ScaLAPACK_TARGET_NAMES scalapack)
|
||||
# set(ScaLAPACK_TARGET_FORCE)
|
||||
# set(ScaLAPACK_IMPORT_CONFIG DEBUG)
|
||||
|
||||
set(GECKO_DIR "${MFEM_DIR}/../gecko" CACHE PATH "Path to the Gecko library.")
|
||||
|
||||
set(GNUTLS_DIR "" CACHE PATH "Path to the GnuTLS library.")
|
||||
@@ -134,20 +102,19 @@ set(NetCDF_REQUIRED_PACKAGES "" CACHE STRING
|
||||
|
||||
set(PETSC_DIR "${MFEM_DIR}/../petsc" CACHE PATH
|
||||
"Path to the PETSc main directory.")
|
||||
set(PETSC_ARCH "arch-linux2-c-debug" CACHE STRING "PETSc build architecture.")
|
||||
set(PETSC_ARCH "arch-linux2-c-debug" CACHE PATH "PETSc build architecture.")
|
||||
|
||||
set(MPFR_DIR "" CACHE PATH "Path to the MPFR library.")
|
||||
|
||||
set(CONDUIT_DIR "${MFEM_DIR}/../conduit" CACHE PATH
|
||||
"Path to the Conduit library.")
|
||||
set(Conduit_REQUIRED_PACKAGES "HDF5" CACHE STRING
|
||||
"Additional packages required by Conduit.")
|
||||
|
||||
set(AXOM_DIR "${MFEM_DIR}/../axom" CACHE PATH "Path to the Axom library.")
|
||||
set(ATK_DIR "${MFEM_DIR}/../asctoolkit" CACHE PATH "Path to the ATK library.")
|
||||
# May need to add "Boost" as requirement.
|
||||
set(Axom_REQUIRED_PACKAGES "Conduit/relay" CACHE STRING
|
||||
"Additional packages required by Axom.")
|
||||
|
||||
set(PUMI_DIR "${MFEM_DIR}/../pumi-2.1.0" CACHE STRING
|
||||
"Directory where PUMI is installed")
|
||||
set(ATK_REQUIRED_PACKAGES "Conduit/relay" CACHE STRING
|
||||
"Additional packages required by ATK.")
|
||||
|
||||
set(BLAS_INCLUDE_DIRS "" CACHE STRING "Path to BLAS headers.")
|
||||
set(BLAS_LIBRARIES "" CACHE STRING "The BLAS library.")
|
||||
|
||||
+33
-122
@@ -30,36 +30,15 @@ PREFIX = ./mfem
|
||||
# Install program
|
||||
INSTALL = /usr/bin/install
|
||||
|
||||
STATIC = YES
|
||||
SHARED = NO
|
||||
|
||||
ifneq ($(NOTMAC),)
|
||||
AR = ar
|
||||
ARFLAGS = cruv
|
||||
RANLIB = ranlib
|
||||
PICFLAG = -fPIC
|
||||
SO_EXT = so
|
||||
SO_VER = so.$(MFEM_VERSION_STRING)
|
||||
BUILD_SOFLAGS = -shared -Wl,-soname,libmfem.$(SO_VER)
|
||||
BUILD_RPATH = -Wl,-rpath,$(BUILD_REAL_DIR)
|
||||
INSTALL_SOFLAGS = $(BUILD_SOFLAGS)
|
||||
INSTALL_RPATH = -Wl,-rpath,@MFEM_LIB_DIR@
|
||||
else
|
||||
# Silence "has no symbols" warnings on Mac OS X
|
||||
AR = ar
|
||||
ARFLAGS = Scruv
|
||||
RANLIB = ranlib -no_warning_for_no_symbols
|
||||
PICFLAG = -fPIC
|
||||
SO_EXT = dylib
|
||||
SO_VER = $(MFEM_VERSION_STRING).dylib
|
||||
MAKE_SOFLAGS = -Wl,-dylib,-install_name,$(1)/libmfem.$(SO_VER),\
|
||||
-compatibility_version,$(MFEM_VERSION_STRING),\
|
||||
-current_version,$(MFEM_VERSION_STRING),\
|
||||
-undefined,dynamic_lookup
|
||||
BUILD_SOFLAGS = $(subst $1 ,,$(call MAKE_SOFLAGS,$(BUILD_REAL_DIR)))
|
||||
BUILD_RPATH = -Wl,-undefined,dynamic_lookup
|
||||
INSTALL_SOFLAGS = $(subst $1 ,,$(call MAKE_SOFLAGS,$(MFEM_LIB_DIR)))
|
||||
INSTALL_RPATH = -Wl,-undefined,dynamic_lookup
|
||||
endif
|
||||
|
||||
# Set CXXFLAGS to overwrite the default selection of DEBUG_FLAGS/OPTIM_FLAGS
|
||||
@@ -75,47 +54,31 @@ endif
|
||||
# Command used to launch MPI jobs
|
||||
MFEM_MPIEXEC = mpirun
|
||||
MFEM_MPIEXEC_NP = -np
|
||||
# Number of mpi tasks for parallel jobs
|
||||
MFEM_MPI_NP = 4
|
||||
|
||||
# MFEM configuration options: YES/NO values, which are exported to config.mk and
|
||||
# config.hpp. The values below are the defaults for generating the actual values
|
||||
# in config.mk and config.hpp.
|
||||
|
||||
MFEM_USE_MPI = NO
|
||||
# FIXME: add MFEM_USE_BACKENDS, MFEM_USE_OCCA to the CMake build system
|
||||
MFEM_USE_BACKENDS = YES
|
||||
MFEM_USE_OCCA = YES
|
||||
MFEM_USE_METIS = $(MFEM_USE_MPI)
|
||||
MFEM_USE_METIS_5 = NO
|
||||
MFEM_DEBUG = NO
|
||||
MFEM_USE_EXCEPTIONS = NO
|
||||
MFEM_USE_GZSTREAM = NO
|
||||
MFEM_USE_LIBUNWIND = NO
|
||||
MFEM_USE_LAPACK = NO
|
||||
MFEM_THREAD_SAFE = NO
|
||||
MFEM_USE_OPENMP = NO
|
||||
MFEM_USE_MEMALLOC = YES
|
||||
MFEM_TIMER_TYPE = $(if $(NOTMAC),2,4)
|
||||
MFEM_TIMER_TYPE = $(if $(NOTMAC),2,0)
|
||||
MFEM_USE_SUNDIALS = NO
|
||||
MFEM_USE_MESQUITE = NO
|
||||
MFEM_USE_SUITESPARSE = NO
|
||||
MFEM_USE_SUPERLU = NO
|
||||
MFEM_USE_STRUMPACK = NO
|
||||
MFEM_USE_GECKO = NO
|
||||
MFEM_USE_GNUTLS = NO
|
||||
MFEM_USE_NETCDF = NO
|
||||
MFEM_USE_PETSC = NO
|
||||
MFEM_USE_MPFR = NO
|
||||
MFEM_USE_SIDRE = NO
|
||||
MFEM_USE_CONDUIT = NO
|
||||
MFEM_USE_PUMI = NO
|
||||
|
||||
# Compile and link options for zlib.
|
||||
ZLIB_DIR =
|
||||
ZLIB_OPT = $(if $(ZLIB_DIR),-I$(ZLIB_DIR)/include)
|
||||
ZLIB_LIB = $(if $(ZLIB_DIR),$(ZLIB_RPATH) -L$(ZLIB_DIR)/lib ,)-lz
|
||||
ZLIB_RPATH = -Wl,-rpath,$(ZLIB_DIR)/lib
|
||||
|
||||
LIBUNWIND_OPT = -g
|
||||
LIBUNWIND_LIB = $(if $(NOTMAC),-lunwind -ldl,)
|
||||
@@ -126,7 +89,7 @@ HYPRE_OPT = -I$(HYPRE_DIR)/include
|
||||
HYPRE_LIB = -L$(HYPRE_DIR)/lib -lHYPRE
|
||||
|
||||
# METIS library configuration
|
||||
ifeq ($(MFEM_USE_SUPERLU)$(MFEM_USE_STRUMPACK),NONO)
|
||||
ifeq ($(MFEM_USE_SUPERLU),NO)
|
||||
ifeq ($(MFEM_USE_METIS_5),NO)
|
||||
METIS_DIR = @MFEM_DIR@/../metis-4.0
|
||||
METIS_OPT =
|
||||
@@ -137,7 +100,7 @@ ifeq ($(MFEM_USE_SUPERLU)$(MFEM_USE_STRUMPACK),NONO)
|
||||
METIS_LIB = -L$(METIS_DIR)/lib -lmetis
|
||||
endif
|
||||
else
|
||||
# ParMETIS: currently needed by SuperLU or STRUMPACK. We assume that METIS 5
|
||||
# ParMETIS currently needed only with SuperLU. We assume that METIS 5
|
||||
# (included with ParMETIS) is installed in the same location.
|
||||
METIS_DIR = @MFEM_DIR@/../parmetis-4.0.3
|
||||
METIS_OPT = -I$(METIS_DIR)/include
|
||||
@@ -157,10 +120,10 @@ OPENMP_LIB =
|
||||
POSIX_CLOCKS_LIB = -lrt
|
||||
|
||||
# SUNDIALS library configuration
|
||||
SUNDIALS_DIR = @MFEM_DIR@/../sundials-3.0.0
|
||||
SUNDIALS_DIR = @MFEM_DIR@/../sundials-2.7.0
|
||||
SUNDIALS_OPT = -I$(SUNDIALS_DIR)/include
|
||||
SUNDIALS_LIB = -Wl,-rpath,$(SUNDIALS_DIR)/lib -L$(SUNDIALS_DIR)/lib\
|
||||
-lsundials_arkode -lsundials_cvode -lsundials_nvecserial -lsundials_kinsol
|
||||
-lsundials_arkode -lsundials_cvode -lsundials_nvecserial -lsundials_kinsol
|
||||
|
||||
ifeq ($(MFEM_USE_MPI),YES)
|
||||
SUNDIALS_LIB += -lsundials_nvecparhyp -lsundials_nvecparallel
|
||||
@@ -177,41 +140,14 @@ MESQUITE_LIB = -L$(MESQUITE_DIR)/lib -lmesquite
|
||||
LIB_RT = $(if $(NOTMAC),-lrt,)
|
||||
SUITESPARSE_DIR = @MFEM_DIR@/../SuiteSparse
|
||||
SUITESPARSE_OPT = -I$(SUITESPARSE_DIR)/include
|
||||
SUITESPARSE_LIB = -Wl,-rpath,$(SUITESPARSE_DIR)/lib -L$(SUITESPARSE_DIR)/lib\
|
||||
-lklu -lbtf -lumfpack -lcholmod -lcolamd -lamd -lcamd -lccolamd\
|
||||
-lsuitesparseconfig $(LIB_RT) $(METIS_LIB) $(LAPACK_LIB)
|
||||
SUITESPARSE_LIB = -L$(SUITESPARSE_DIR)/lib -lklu -lbtf -lumfpack -lcholmod\
|
||||
-lcolamd -lamd -lcamd -lccolamd -lsuitesparseconfig $(LIB_RT) $(METIS_LIB)\
|
||||
$(LAPACK_LIB)
|
||||
|
||||
# SuperLU library configuration
|
||||
SUPERLU_DIR = @MFEM_DIR@/../SuperLU_DIST_5.1.0
|
||||
SUPERLU_OPT = -I$(SUPERLU_DIR)/SRC
|
||||
SUPERLU_LIB = -Wl,-rpath,$(SUPERLU_DIR)/SRC -L$(SUPERLU_DIR)/SRC -lsuperlu_dist
|
||||
|
||||
# SCOTCH library configuration (required by STRUMPACK)
|
||||
SCOTCH_DIR = @MFEM_DIR@/../scotch_6.0.4
|
||||
SCOTCH_OPT = -I$(SCOTCH_DIR)/include
|
||||
SCOTCH_LIB = -L$(SCOTCH_DIR)/lib -lptscotch -lptscotcherr -lscotch -lscotcherr\
|
||||
-lpthread
|
||||
|
||||
# SCALAPACK library configuration (required by STRUMPACK)
|
||||
SCALAPACK_DIR = @MFEM_DIR@/../scalapack-2.0.2
|
||||
SCALAPACK_OPT = -I$(SCALAPACK_DIR)/SRC
|
||||
SCALAPACK_LIB = -L$(SCALAPACK_DIR)/lib -lscalapack $(LAPACK_LIB)
|
||||
|
||||
# MPI Fortran library, needed e.g. by STRUMPACK
|
||||
# MPICH:
|
||||
MPI_FORTRAN_LIB = -lmpifort
|
||||
# OpenMPI:
|
||||
# MPI_FORTRAN_LIB = -lmpi_mpifh
|
||||
# Additional Fortan library:
|
||||
# MPI_FORTRAN_LIB += -lgfortran
|
||||
|
||||
# STRUMPACK library configuration
|
||||
STRUMPACK_DIR = @MFEM_DIR@/../STRUMPACK-build
|
||||
STRUMPACK_OPT = -I$(STRUMPACK_DIR)/include $(SCOTCH_OPT)
|
||||
# If STRUMPACK was build with OpenMP support, the following may be need:
|
||||
# STRUMPACK_OPT += $(OPENMP_OPT)
|
||||
STRUMPACK_LIB = -L$(STRUMPACK_DIR)/lib -lstrumpack $(MPI_FORTRAN_LIB)\
|
||||
$(SCOTCH_LIB) $(SCALAPACK_LIB)
|
||||
SUPERLU_LIB = -L$(SUPERLU_DIR)/SRC -lsuperlu_dist
|
||||
|
||||
# Gecko library configuration
|
||||
GECKO_DIR = @MFEM_DIR@/../gecko
|
||||
@@ -223,68 +159,43 @@ GNUTLS_OPT =
|
||||
GNUTLS_LIB = -lgnutls
|
||||
|
||||
# NetCDF library configuration
|
||||
NETCDF_DIR = $(HOME)/local
|
||||
HDF5_DIR = $(HOME)/local
|
||||
NETCDF_OPT = -I$(NETCDF_DIR)/include -I$(HDF5_DIR)/include $(ZLIB_OPT)
|
||||
NETCDF_LIB = -Wl,-rpath,$(NETCDF_DIR)/lib -L$(NETCDF_DIR)/lib\
|
||||
-Wl,-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib\
|
||||
-lnetcdf -lhdf5_hl -lhdf5 $(ZLIB_LIB)
|
||||
NETCDF_DIR = $(HOME)/local
|
||||
HDF5_DIR = $(HOME)/local
|
||||
ZLIB_DIR = $(HOME)/local
|
||||
NETCDF_OPT = -I$(NETCDF_DIR)/include
|
||||
NETCDF_LIB = -L$(NETCDF_DIR)/lib -lnetcdf -L$(HDF5_DIR)/lib -lhdf5_hl -lhdf5\
|
||||
-L$(ZLIB_DIR)/lib -lz
|
||||
|
||||
# PETSc library configuration (version greater or equal to 3.8 or the dev branch)
|
||||
PETSC_ARCH := arch-linux2-c-debug
|
||||
PETSC_DIR := $(MFEM_DIR)/../petsc/$(PETSC_ARCH)
|
||||
PETSC_VARS := $(PETSC_DIR)/lib/petsc/conf/petscvariables
|
||||
PETSC_FOUND := $(if $(wildcard $(PETSC_VARS)),YES,)
|
||||
PETSC_INC_VAR = PETSC_CC_INCLUDES
|
||||
PETSC_LIB_VAR = PETSC_EXTERNAL_LIB_BASIC
|
||||
ifeq ($(PETSC_FOUND),YES)
|
||||
PETSC_OPT := $(shell sed -n "s/$(PETSC_INC_VAR) = *//p" $(PETSC_VARS))
|
||||
PETSC_LIB := $(shell sed -n "s/$(PETSC_LIB_VAR) = *//p" $(PETSC_VARS))
|
||||
PETSC_LIB := -Wl,-rpath,$(abspath $(PETSC_DIR))/lib\
|
||||
-L$(abspath $(PETSC_DIR))/lib -lpetsc $(PETSC_LIB)
|
||||
ifeq ($(MFEM_USE_PETSC),YES)
|
||||
PETSC_DIR := $(MFEM_DIR)/../petsc/arch-linux2-c-debug
|
||||
PETSC_PC := $(PETSC_DIR)/lib/pkgconfig/PETSc.pc
|
||||
$(if $(wildcard $(PETSC_PC)),,$(error PETSc config not found - $(PETSC_PC)))
|
||||
PETSC_OPT := $(shell sed -n "s/Cflags: *//p" $(PETSC_PC))
|
||||
PETSC_LIB := $(shell sed -n "s/Libs.*: *//p" $(PETSC_PC))
|
||||
PETSC_LIB := -Wl,-rpath -Wl,$(abspath $(PETSC_DIR))/lib $(PETSC_LIB)
|
||||
endif
|
||||
|
||||
# MPFR library configuration
|
||||
MPFR_OPT =
|
||||
MPFR_LIB = -lmpfr
|
||||
|
||||
# Conduit and required libraries configuration
|
||||
CONDUIT_DIR = @MFEM_DIR@/../conduit
|
||||
CONDUIT_OPT = -I$(CONDUIT_DIR)/include/conduit
|
||||
CONDUIT_LIB = \
|
||||
-Wl,-rpath,$(CONDUIT_DIR)/lib -L$(CONDUIT_DIR)/lib \
|
||||
-lconduit -lconduit_relay -lconduit_blueprint -ldl
|
||||
|
||||
# Check if Conduit was built with hdf5 support, by looking
|
||||
# for the relay hdf5 header
|
||||
CONDUIT_HDF5_HEADER=$(CONDUIT_DIR)/include/conduit/conduit_relay_hdf5.hpp
|
||||
ifneq (,$(wildcard $(CONDUIT_HDF5_HEADER)))
|
||||
CONDUIT_OPT += -I$(HDF5_DIR)/include
|
||||
CONDUIT_LIB += -Wl,-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib \
|
||||
-lhdf5 $(ZLIB_LIB)
|
||||
endif
|
||||
|
||||
# Sidre and required libraries configuration
|
||||
# Be sure to check the HDF5_DIR (set above) is correct
|
||||
SIDRE_DIR = @MFEM_DIR@/../axom
|
||||
SIDRE_DIR = @MFEM_DIR@/../asctoolkit
|
||||
CONDUIT_DIR = @MFEM_DIR@/../conduit
|
||||
SIDRE_OPT = -I$(SIDRE_DIR)/include -I$(CONDUIT_DIR)/include/conduit\
|
||||
-I$(HDF5_DIR)/include
|
||||
SIDRE_LIB = \
|
||||
-Wl,-rpath,$(SIDRE_DIR)/lib -L$(SIDRE_DIR)/lib \
|
||||
-Wl,-rpath,$(CONDUIT_DIR)/lib -L$(CONDUIT_DIR)/lib \
|
||||
-Wl,-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib \
|
||||
-lsidre -lslic -laxom_utils -lconduit -lconduit_relay -lhdf5 $(ZLIB_LIB) -ldl
|
||||
SIDRE_LIB = -L$(SIDRE_DIR)/lib \
|
||||
-L$(CONDUIT_DIR)/lib \
|
||||
-Wl,-rpath -Wl,$(CONDUIT_DIR)/lib \
|
||||
-L$(HDF5_DIR)/lib\
|
||||
-Wl,-rpath -Wl,$(HDF5_DIR)/lib \
|
||||
-lsidre -lslic -lcommon -lconduit -lconduit_relay -lhdf5 -lz -ldl
|
||||
|
||||
# PUMI
|
||||
# Note that PUMI_DIR is needed -- it is used to check for gmi_sim.h
|
||||
PUMI_DIR = @MFEM_DIR@/../pumi-2.1.0
|
||||
PUMI_OPT = -I$(PUMI_DIR)/include
|
||||
PUMI_LIB = -L$(PUMI_DIR)/lib -lpumi -lcrv -lma -lmds -lapf -lpcu -lgmi -lparma\
|
||||
-llion -lmth -lapf_zoltan -lspr
|
||||
|
||||
OCCA_DIR = @MFEM_DIR@/../occa
|
||||
OCCA_OPT = -I$(OCCA_DIR)/include
|
||||
OCCA_LIB = -Wl,-rpath,$(OCCA_DIR)/lib -L$(OCCA_DIR)/lib -locca
|
||||
ifeq ($(MFEM_USE_MPI),YES)
|
||||
SIDRE_LIB += -lspio -lcommon
|
||||
endif
|
||||
|
||||
# If YES, enable some informational messages
|
||||
VERBOSE = NO
|
||||
|
||||
@@ -1,48 +0,0 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "HYPRE_config.h"
|
||||
#include <cstdio>
|
||||
|
||||
#ifdef HYPRE_RELEASE_VERSION
|
||||
#define HYPRE_VERSION_STRING HYPRE_RELEASE_VERSION
|
||||
#elif defined(HYPRE_PACKAGE_VERSION)
|
||||
#define HYPRE_VERSION_STRING HYPRE_PACKAGE_VERSION
|
||||
#endif
|
||||
|
||||
// Macros to expand a macro as a string
|
||||
#define STR_EXPAND(s) #s
|
||||
#define STR(s) STR_EXPAND(s)
|
||||
|
||||
// Convert the HYPRE_RELEASE_VERSION macro (string) to integer.
|
||||
// Examples: "2.10.0b" --> 21000, "2.11.2" --> 21102
|
||||
int main()
|
||||
{
|
||||
#ifdef HYPRE_VERSION_STRING
|
||||
const char *ptr = STR(HYPRE_VERSION_STRING);
|
||||
if (*ptr == '"') { ptr++; }
|
||||
int version = 0;
|
||||
for (int i = 0; i < 3; i++, ptr++)
|
||||
{
|
||||
int pv = 0;
|
||||
for (char d; d = *ptr, '0' <= d && d <= '9'; ptr++)
|
||||
{
|
||||
pv = 10*pv + (d - '0');
|
||||
if (pv >= 100) { return 1; }
|
||||
}
|
||||
version = 100*version + pv;
|
||||
}
|
||||
printf("%i\n", version);
|
||||
return 0;
|
||||
#else
|
||||
return 2;
|
||||
#endif
|
||||
}
|
||||
+5
-32
@@ -31,44 +31,17 @@ CONFIG_HPP = _config.hpp
|
||||
CONFIG_MK = config.mk
|
||||
|
||||
.SUFFIXES:
|
||||
.PHONY: all get-hypre-version header config-mk
|
||||
.PHONY: all header config-mk
|
||||
|
||||
all: header config-mk
|
||||
|
||||
MPI = $(MFEM_USE_MPI:NO=)
|
||||
GHV = get_hypre_version
|
||||
GHV_FLAGS = $(subst @MFEM_DIR@,$(if $(MFEM_DIR),$(MFEM_DIR),..),$(HYPRE_OPT))
|
||||
SMX = $(if $(MFEM_USE_PUMI:NO=),MFEM_USE_SIMMETRIX)
|
||||
SMX_PATH = $(PUMI_DIR)/include/gmi_sim.h
|
||||
SMX_FILE = $(subst @MFEM_DIR@,$(if $(MFEM_DIR),$(MFEM_DIR),..),$(SMX_PATH))
|
||||
|
||||
$(GHV): $(SRC)$(GHV).cpp
|
||||
$(call mfem-info, Determining HYPRE version ...)
|
||||
$(MFEM_CXX) ${GHV_FLAGS} $(SRC)$(GHV).cpp -o $(GHV)
|
||||
$(GHV).out: $(GHV)
|
||||
./$(GHV) > $(GHV).out
|
||||
.INTERMEDIATE: $(GHV) $(GHV).out
|
||||
|
||||
get-hypre-version: $(GHV).out
|
||||
$(eval MFEM_HYPRE_VERSION:=$(shell cat $(GHV).out))
|
||||
$(if $(MFEM_HYPRE_VERSION),$(eval export MFEM_HYPRE_VERSION)\
|
||||
$(info HYPRE version: $(MFEM_HYPRE_VERSION)),\
|
||||
$(error Unable to determine HYPRE version))
|
||||
|
||||
check-smx:
|
||||
$(call mfem-info, Checking for Simmetrix header [$(SMX_FILE)] ...)
|
||||
$(eval MFEM_USE_SIMMETRIX:=$(if $(wildcard $(SMX_FILE)),YES,NO))
|
||||
$(call mfem-info, MFEM_USE_SIMMETRIX = $(MFEM_USE_SIMMETRIX))
|
||||
$(eval export MFEM_USE_SIMMETRIX)
|
||||
|
||||
header: $(if $(MPI),get-hypre-version,) $(if $(SMX),check-smx)
|
||||
header:
|
||||
$(call mfem-info, Writing $(CONFIG_HPP) ...)
|
||||
@set -- && \
|
||||
for def in $${MFEM_DEFINES} $(if $(MPI),MFEM_HYPRE_VERSION) $(SMX); do \
|
||||
for def in $${MFEM_DEFINES}; do \
|
||||
eval var=\$$$$def && \
|
||||
if [ "NO" != "$${var}" ]; then \
|
||||
set -- "$$@" -e "s|// \(#define $${def} \)|\1|" && \
|
||||
set -- "$$@" -e "s|// \(#define $${def}\)$$|\1|" && \
|
||||
set -- "$$@" -e "s|// \(#define $${def}\)|\1|" && \
|
||||
set -- "$$@" -e "s#@$${def}@#$${var}#g"; \
|
||||
fi; \
|
||||
done && \
|
||||
@@ -89,4 +62,4 @@ config-mk:
|
||||
sed "$$@" $(SRC)config.mk.in > $(CONFIG_MK)
|
||||
|
||||
clean:
|
||||
rm -f $(CONFIG_HPP) $(CONFIG_MK) sample-runs-build.log
|
||||
rm -f $(CONFIG_HPP) $(CONFIG_MK)
|
||||
|
||||
@@ -1,525 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
make="${MAKE:-make}"
|
||||
mpiexec="${MPIEXEC:-mpirun}"
|
||||
mpiexec_np="${MPIEXEC_NP:--np}"
|
||||
run_prefix=""
|
||||
run_vg="valgrind --leak-check=full --show-reachable=yes --track-origins=yes"
|
||||
run_suffix="-no-vis"
|
||||
skip_gen_meshes="yes"
|
||||
cur_dir="${PWD}"
|
||||
mfem_dir="$(cd "$(dirname "$0")"/.. && pwd)"
|
||||
mfem_build_dir=""
|
||||
build_log=""
|
||||
output_dir=""
|
||||
output_sfx=".out"
|
||||
# The group format is: '"group-name" "group-summary-title" "group-directory"
|
||||
# "group-source-patterns"'
|
||||
groups_serial=(
|
||||
'"examples"
|
||||
"Examples:"
|
||||
"examples"
|
||||
"ex{,1}[0-9].cpp"'
|
||||
# "ex1.cpp"'
|
||||
'"sundials"
|
||||
"SUNDIALS examples:"
|
||||
"examples/sundials"
|
||||
"ex{9,10,16}.cpp"'
|
||||
'"performance"
|
||||
"Performance miniapps:"
|
||||
"miniapps/performance"
|
||||
"ex1.cpp"'
|
||||
# ""'
|
||||
'"meshing"
|
||||
"Meshing miniapps:"
|
||||
"miniapps/meshing"
|
||||
"mobius-strip.cpp klein-bottle.cpp mesh-optimizer.cpp"'
|
||||
)
|
||||
# Parallel groups
|
||||
groups_parallel=(
|
||||
'"examples"
|
||||
"Examples:"
|
||||
"examples"
|
||||
"ex{,1}[0-9]p.cpp"'
|
||||
# "ex1p.cpp"'
|
||||
'"sundials"
|
||||
"SUNDIALS examples:"
|
||||
"examples/sundials"
|
||||
"ex{9,10,16}p.cpp"'
|
||||
'"petsc"
|
||||
"PETSc examples:"
|
||||
"examples/petsc"
|
||||
"ex{,1}[0-9]p.cpp"'
|
||||
'"performance"
|
||||
"Performance miniapps:"
|
||||
"miniapps/performance"
|
||||
"ex1p.cpp"'
|
||||
# ""'
|
||||
'"meshing"
|
||||
"Meshing miniapps:"
|
||||
"miniapps/meshing"
|
||||
"pmesh-optimizer.cpp"'
|
||||
'"electromagnetics"
|
||||
"Electromagnetics miniapps:"
|
||||
"miniapps/electromagnetics"
|
||||
"joule.cpp"'
|
||||
# "{volta,tesla,joule}.cpp"' # todo: multiline sample runs
|
||||
)
|
||||
# All groups serial + parallel runs mixed in the same group:
|
||||
groups_all=(
|
||||
'"examples"
|
||||
"Examples:"
|
||||
"examples"
|
||||
"ex\"{,1}[0-9]\"{,p}.cpp"'
|
||||
'"sundials"
|
||||
"SUNDIALS examples:"
|
||||
"examples/sundials"
|
||||
"ex\"{9,10,16}\"{,p}.cpp"'
|
||||
'"petsc"
|
||||
"PETSc examples:"
|
||||
"examples/petsc"
|
||||
"ex{,1}[0-9]p.cpp"'
|
||||
'"performance"
|
||||
"Performance miniapps:"
|
||||
"miniapps/performance"
|
||||
"ex1{,p}.cpp"'
|
||||
'"meshing"
|
||||
"Meshing miniapps:"
|
||||
"miniapps/meshing"
|
||||
"mobius-strip.cpp klein-bottle.cpp {,p}mesh-optimizer.cpp"'
|
||||
'"electromagnetics"
|
||||
"Electromagnetics miniapps:"
|
||||
"miniapps/electromagnetics"
|
||||
"joule.cpp"'
|
||||
# "{volta,tesla,joule}.cpp"' # todo: multiline sample runs
|
||||
)
|
||||
make_all="all"
|
||||
base_timeformat=$'real: %3Rs user: %3Us sys: %3Ss %%cpu: %P'
|
||||
# separator
|
||||
sep='----------------------------------------------------------------'
|
||||
|
||||
# Command line parameters:
|
||||
opt_help="no"
|
||||
opt_show="no"
|
||||
mfem_config="MFEM_USE_MPI=NO MFEM_DEBUG=NO"
|
||||
groups=()
|
||||
group_name=""
|
||||
group_title=""
|
||||
valgrind="no"
|
||||
make_j="-j $(getconf _NPROCESSORS_ONLN)"
|
||||
color="no"
|
||||
built="no"
|
||||
timing="no"
|
||||
|
||||
# Read the sample runs from the source "$1" and put them in the array variable
|
||||
# "runs".
|
||||
function extract_sample_runs()
|
||||
{
|
||||
local old_IFS="${IFS}" sruns="" pruns=""
|
||||
local src="$1"
|
||||
if [ "${src}" == "" ]; then runs=(); return 1; fi
|
||||
local app=${src%.cpp}
|
||||
local vg_app="${app}"
|
||||
if [ "${valgrind}" == "yes" ]; then vg_app="${run_vg} ${app}"; fi
|
||||
# parallel sample runs are lines matching "^//.* mpirun .* ${app}" with
|
||||
# everything in front of "mpirun" removed:
|
||||
pruns=`grep "^//.* mpirun .* ${app}" "${src}" |
|
||||
sed -e "s/.* mpirun \(.*\) ${app}/${mpiexec} \1 ${app}/g" \
|
||||
-e "s/ -np\(.*\) ${app}/ ${mpiexec_np}\1 ${vg_app}/g"`
|
||||
# serial sample runs are lines that are not parallel sample runs and matching
|
||||
# "^//.* ${app}" with everything in front of "${app}" removed:
|
||||
sruns=`grep -v "^//.* mpirun .* ${app}" "${src}" |
|
||||
grep "^//.* ${app}" |
|
||||
sed -e "s/.* ${app}/${vg_app}/g"`
|
||||
runs="${sruns}${pruns}"
|
||||
if [ "$skip_gen_meshes" == "yes" ]; then
|
||||
runs=`printf "%s" "$runs" | grep -v ".* -m .*\.gen"`
|
||||
fi
|
||||
IFS=$'\n'
|
||||
runs=(${runs})
|
||||
IFS="${old_IFS}"
|
||||
}
|
||||
|
||||
# Echo usage information
|
||||
function help_message()
|
||||
{
|
||||
cat <<EOF
|
||||
|
||||
$0 [options]
|
||||
|
||||
Options:
|
||||
-h|-help Print this usage information and exit
|
||||
-p|-par Build the parallel MFEM library + examples + miniapps.
|
||||
The default is to build the serial MFEM library + examples +
|
||||
miniapps. The build can be customized by setting the variable
|
||||
'mfem_config' or by building separately and using '-b'
|
||||
-g <dir> <pattern>
|
||||
Specify explicitly a group (dir + file pattern) to run; This
|
||||
option can be used multiple times to define multiple groups
|
||||
-v Enable valgrind
|
||||
-o <dir> [${output_dir:-"<empty>: output goes to stdout"}]
|
||||
If not empty, save output to files inside <dir>
|
||||
-j <np> [${make_j}] Specify the number of jobs to use for building
|
||||
-c|-color Always use colors for the status messages: OK, FAILED, etc
|
||||
-b|-built Do NOT rebuild the library and the executables
|
||||
-t|-time Measure and print execution time for each sample run
|
||||
-s|-show Show all configured sample runs and exit
|
||||
-n Dry run: replace "\$sample_run" with "echo \$sample_run"
|
||||
<var>=<value>
|
||||
Set a shell script varible; see below for valid variables
|
||||
* Any other parameter is treated as <mfem_dir>
|
||||
<mfem_dir> [${mfem_dir}] is the MFEM source directory
|
||||
|
||||
This script tests all the sample runs listed in the begining comments of
|
||||
MFEM's serial or parallel example and miniapp codes. The list of sample runs
|
||||
is auto-generated and can be viewed with the -s|-show option.
|
||||
|
||||
The following shell script variables can be set with <var>=<value>:
|
||||
output_dir [${output_dir}]
|
||||
Same as '-o': if not empty, save output to files in that directory
|
||||
output_sfx [${output_sfx}]
|
||||
Suffix to append to the output files
|
||||
mfem_config [${mfem_config}]
|
||||
Set MFEM configuration options
|
||||
make [${make}], mpiexec [${mpiexec}], mpiexec_np [${mpiexec_np}]
|
||||
Their values can also set using the respective uppercase environment
|
||||
variable
|
||||
mfem_build_dir [${mfem_build_dir}]
|
||||
Set this variable to something different from <mfem_dir> to use an
|
||||
out-of-source build
|
||||
|
||||
For other valid variables, see the script source.
|
||||
|
||||
The following environment variables, if non-empty, are used:
|
||||
MAKE, MPIEXEC, MPIEXEC_NP
|
||||
|
||||
Example usage:
|
||||
$0 -s [ Show all configured sample runs ]
|
||||
$0 -o baseline [ Serial build; run and save all built sample runs ]
|
||||
$0 -p -o baseline [ Parallel build; run and save all sample runs ]
|
||||
$0 -b -g examples ex8.cpp [ Use the existing build; run ex8 sample runs ]
|
||||
|
||||
EOF
|
||||
}
|
||||
|
||||
function show_runs()
|
||||
{
|
||||
echo "${sep}"
|
||||
for group_params in "${groups[@]}"; do
|
||||
eval params=(${group_params})
|
||||
name="${params[0]}"
|
||||
title="${params[1]}"
|
||||
group_dir="${mfem_dir}/${params[2]}"
|
||||
pattern="${params[3]}"
|
||||
printf "group name: [%s]\n" "${name}"
|
||||
printf "summary title: [%s]\n" "${title}"
|
||||
printf "directory: [%s]\n" "${group_dir}"
|
||||
printf "pattern: [%s]\n" "${pattern}"
|
||||
cd "${cur_dir}"; cd "${group_dir}" || exit 1
|
||||
eval sources=(${pattern})
|
||||
eval sources=("${sources[@]}")
|
||||
printf "sources: (%s)\n" "${sources[*]}"
|
||||
printf "sample runs:\n"
|
||||
for src in "${sources[@]}"; do
|
||||
extract_sample_runs "${src}"
|
||||
for run in "${runs[@]}"; do
|
||||
printf " %s\n" "${run}"
|
||||
done
|
||||
done
|
||||
echo "${sep}"
|
||||
done
|
||||
}
|
||||
|
||||
# Process command line parameters
|
||||
while [ $# -gt 0 ]; do
|
||||
|
||||
case "$1" in
|
||||
-h|-help)
|
||||
opt_help="yes"
|
||||
;;
|
||||
-p|-parallel)
|
||||
mfem_config="MFEM_USE_MPI=YES MFEM_DEBUG=NO"
|
||||
;;
|
||||
-g)
|
||||
gbasename="$(basename "$2")"
|
||||
gname="${group_name:-${gbasename}}"
|
||||
gtitle="${group_title:-"Group <${gbasename}>:"}"
|
||||
test_group="\"${gname}\" \"${gtitle}\" \"$2\" \"$3\""
|
||||
groups=("${groups[@]}" "${test_group}")
|
||||
shift 2
|
||||
;;
|
||||
-v)
|
||||
valgrind="yes"
|
||||
;;
|
||||
-o)
|
||||
shift
|
||||
output_dir="$1"
|
||||
;;
|
||||
-j)
|
||||
shift
|
||||
make_j="-j $1"
|
||||
;;
|
||||
-c|-color)
|
||||
color="yes"
|
||||
;;
|
||||
-b|-built)
|
||||
built="yes"
|
||||
;;
|
||||
-t|-time)
|
||||
timing="yes"
|
||||
;;
|
||||
-s|-show)
|
||||
opt_show="yes"
|
||||
;;
|
||||
-n)
|
||||
run_prefix="echo"
|
||||
;;
|
||||
*=*)
|
||||
eval $1
|
||||
;;
|
||||
*)
|
||||
mfem_dir="$1"
|
||||
;;
|
||||
esac
|
||||
|
||||
shift
|
||||
done # while ...
|
||||
|
||||
mfem_build_dir="${mfem_build_dir:-${mfem_dir}}"
|
||||
|
||||
build_log="${build_log:-${mfem_build_dir}/config/sample-runs-build.log}"
|
||||
|
||||
if [ 0 -eq ${#groups[*]} ]; then
|
||||
groups=("${groups_all[@]}")
|
||||
# These can be used as command line arguments:
|
||||
# 'groups=("${groups_serial[@]}")'
|
||||
# 'groups=("${groups_parallel[@]}")'
|
||||
fi
|
||||
|
||||
if [ "${opt_help}" == "yes" ]; then
|
||||
help_message
|
||||
exit
|
||||
fi
|
||||
|
||||
if [ "${opt_show}" == "yes" ]; then
|
||||
show_runs
|
||||
exit
|
||||
fi
|
||||
|
||||
# Setup colors
|
||||
if [ -t 1 ] && [ -z "${output_dir}" ] || [ "${color}" == "yes" ]; then
|
||||
red='\033[0;31m'
|
||||
green='\033[0;32m'
|
||||
yellow='\033[0;33m'
|
||||
magenta='\033[0;35m'
|
||||
cyan='\033[0;36m'
|
||||
none='\033[0m'
|
||||
else
|
||||
red=
|
||||
green=
|
||||
yellow=
|
||||
magenta=
|
||||
cyan=
|
||||
none=
|
||||
fi
|
||||
|
||||
# Run the given command, saving the rune time in the variable "timer".
|
||||
function timed_run()
|
||||
{
|
||||
timer="$({ time "$@" 1>&3 2>&4; } 2>&1)"
|
||||
} 3>&1 4>&2
|
||||
|
||||
# This function is used to execute the sample runs
|
||||
function go()
|
||||
{
|
||||
local cmd=("$@")
|
||||
local res=""
|
||||
echo $sep
|
||||
echo "<${group}>" "${cmd[@]}"
|
||||
echo $sep
|
||||
if [ "${timing}" == "yes" ]; then
|
||||
timed_run "${cmd[@]}"
|
||||
else
|
||||
"${cmd[@]}"
|
||||
fi
|
||||
if [ "$?" -eq 0 ]; then
|
||||
res="${green} OK ${none}"
|
||||
else
|
||||
res="${red}FAILED${none}"
|
||||
fi
|
||||
printf "[${res}] <${group}> ${cmd[*]}\n"
|
||||
if [ "${timing}" == "yes" ]; then
|
||||
printf "Run time: %s\n" "${timer}"
|
||||
timer=(${timer})
|
||||
timer="${timer[1]}"
|
||||
printf -v line "[$res](%8s) ${cmd[*]}" "$timer"
|
||||
summary=("${summary[@]}" "$line")
|
||||
else
|
||||
summary=("${summary[@]}" "[${res}] ${cmd[*]}")
|
||||
fi
|
||||
echo $sep
|
||||
}
|
||||
|
||||
# This function is used to run a group of sample runs (in the same directory)
|
||||
function go_group()
|
||||
{
|
||||
local res=""
|
||||
if [ $# -eq 0 ]; then return 0; fi
|
||||
local group_output_dir="" output_file="" output=""
|
||||
if [ ! -z "$output_dir" ]; then
|
||||
group_output_dir="${output_dir}/${group_dir}"
|
||||
mkdir -p "${group_output_dir}" || exit 1
|
||||
fi
|
||||
for src in "$@"; do
|
||||
cd "${mfem_dir}/${group_dir}" || exit 1
|
||||
extract_sample_runs "${src}" || continue
|
||||
[ "${#runs[@]}" -eq 0 ] && continue
|
||||
cd "${mfem_build_dir}/${group_dir}" || exit 1
|
||||
if [ ! -x "${src%.cpp}" ]; then
|
||||
res="${magenta} SKIP ${none}"
|
||||
echo $sep
|
||||
printf "[${res}] <${group}> <${src}>\n"
|
||||
echo $sep
|
||||
summary=("${summary[@]}" "[${res}] <${src}>")
|
||||
continue
|
||||
fi
|
||||
if [ ! -z "$output_dir" ]; then
|
||||
output_file="${group_output_dir}/${src}${output_sfx}"
|
||||
: > "${output_file}"
|
||||
output=">> \"${output_file}\" 2>&1"
|
||||
fi
|
||||
for run in "${runs[@]}"; do
|
||||
if [ "${run}" == "" ]; then continue; fi
|
||||
eval go \${run_prefix} \${run} \${run_suffix} $output
|
||||
done
|
||||
done
|
||||
${make} clean-exec
|
||||
}
|
||||
|
||||
# Make sure $mfem_dir exists and we can cd into it
|
||||
cd "$mfem_dir" || exit 1
|
||||
# Make sure $mfem_dir is an absolute path
|
||||
mfem_dir="$PWD"
|
||||
cd "${cur_dir}"
|
||||
if [ "${built}" == "no" ]; then
|
||||
mkdir -p "${mfem_build_dir}" || exit 1
|
||||
fi
|
||||
# Make sure $mfem_build_dir exists and we can cd into it
|
||||
cd "${mfem_build_dir}" || exit 1
|
||||
# Make sure $mfem_build_dir is an absolute path
|
||||
mfem_build_dir="$PWD"
|
||||
# Setup 'output_dir'
|
||||
if [ ! -z "$output_dir" ]; then
|
||||
cd "${cur_dir}"
|
||||
mkdir -p "${output_dir}" && cd "${output_dir}" || exit 1
|
||||
output_dir="$PWD"
|
||||
echo "Sending output to files in: [${output_dir}]"
|
||||
echo "Using suffix: [${output_sfx}]"
|
||||
fi
|
||||
|
||||
TIMEFORMAT="${base_timeformat}"
|
||||
|
||||
function set_echo_log()
|
||||
{
|
||||
local dirname=`dirname "$1"`
|
||||
cd "${cur_dir}"
|
||||
mkdir -p "${dirname}" && cd "${dirname}" || exit 1
|
||||
echo_log="$PWD"/`basename "$1"`
|
||||
}
|
||||
|
||||
# Echo the given command line; then run it sending all output to $echo_log
|
||||
function echo_run()
|
||||
{
|
||||
echo " $@"
|
||||
{ echo " $@"; echo "$sep";
|
||||
"$@"
|
||||
echo "$sep"; } >> "$echo_log" 2>&1
|
||||
}
|
||||
|
||||
# Function that builds the mfem library, examples and miniapps
|
||||
function build_all()
|
||||
{
|
||||
printf "Building MFEM with all examples and miniapps:\n"
|
||||
set_echo_log "${build_log}"
|
||||
echo " ### build log: [$echo_log]"
|
||||
{ echo "$sep"; echo " MFEM build log"; echo "$sep"; } > "$echo_log"
|
||||
echo_run cd "${mfem_build_dir}"
|
||||
if [ "${mfem_dir}" != "${mfem_build_dir}" ]; then
|
||||
echo_run ${make} -f "${mfem_dir}"/makefile config
|
||||
fi
|
||||
# Don't use 'make distclean' as it will delete the default $build_log
|
||||
echo_run ${make} clean || exit 1
|
||||
echo_run ${make} config ${mfem_config} || exit 1
|
||||
echo_run ${make} ${make_j} || exit 1
|
||||
echo_run ${make} ${make_all} ${make_j} || exit 1
|
||||
}
|
||||
|
||||
# Function that runs all sample runs, given by the array variable "groups".
|
||||
function all_go()
|
||||
{
|
||||
for group_params in "${groups[@]}"; do
|
||||
eval params=(${group_params})
|
||||
group="${params[0]}"
|
||||
group_dir="${params[2]}"
|
||||
cd "${mfem_dir}/${group_dir}" || exit 1
|
||||
eval sources=(${params[3]})
|
||||
eval sources=("${sources[@]}")
|
||||
summary=("${summary[@]}" "${params[1]}")
|
||||
go_group "${sources[@]}"
|
||||
done
|
||||
|
||||
printf "Summary:\n--------\n"
|
||||
for line in "${summary[@]}"; do
|
||||
printf "${line}\n"
|
||||
done
|
||||
}
|
||||
|
||||
function main()
|
||||
{
|
||||
# Build all mfem examples and miniapps
|
||||
if [ "${built}" == "no" ]; then
|
||||
if [ "${timing}" == "yes" ]; then
|
||||
timed_run build_all
|
||||
printf "Build time: %s\n" "${timer}"
|
||||
else
|
||||
build_all
|
||||
fi
|
||||
fi
|
||||
|
||||
summary=()
|
||||
PATH=.:$PATH
|
||||
|
||||
# Print the MFEM configuration info
|
||||
cd "${mfem_build_dir}"
|
||||
echo "$sep"
|
||||
echo "MFEM configuration"
|
||||
echo "$sep"
|
||||
${make} info
|
||||
echo "$sep"
|
||||
|
||||
# Run all sample runs.
|
||||
if [ "${timing}" == "yes" ]; then
|
||||
timed_run all_go
|
||||
printf "Total run time: %s\n" "${timer}"
|
||||
else
|
||||
all_go
|
||||
fi
|
||||
echo
|
||||
}
|
||||
|
||||
output=""
|
||||
if [ ! -z "$output_dir" ]; then
|
||||
output=">> \"${output_dir}/main${output_sfx}\" 2>&1"
|
||||
fi
|
||||
eval main $output
|
||||
+7
-55
@@ -10,49 +10,14 @@
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Utilities for the "make test" and "make check" targets.
|
||||
|
||||
# Colors used below:
|
||||
# green '\033[0;32m'
|
||||
# red '\033[0;31m'
|
||||
# yellow '\033[0;33m'
|
||||
# no color '\033[0m'
|
||||
COLOR_PRINT = if [ -t 1 ]; then \
|
||||
printf $(1)$(2)'\033[0m'$(3); else printf $(2)$(3); fi
|
||||
PRINT_OK = $(call COLOR_PRINT,'\033[0;32m',OK," ($$1 $$2)\n")
|
||||
PRINT_FAILED = $(call COLOR_PRINT,'\033[0;31m',FAILED," ($$1 $$2)\n")
|
||||
PRINT_SKIP = $(call COLOR_PRINT,'\033[0;33m',SKIP,"\n")
|
||||
|
||||
# Timing support
|
||||
define TIMECMD_detect
|
||||
timecmd=$$(which time 2> /dev/null);
|
||||
if [ -n "$$timecmd" ]; then
|
||||
if $$timecmd --version > /dev/null 2>&1; then
|
||||
echo "$$timecmd" GNU; else echo "$$timecmd" NOTGNU; fi;
|
||||
else timecmd=$$(command -v time);
|
||||
if [ "$$timecmd" = time ]; then
|
||||
echo "$$timecmd" BASH; else echo X NONE; fi;
|
||||
fi
|
||||
endef
|
||||
define TIMECMD.GNU
|
||||
export TIME='%es %MkB %x'; \
|
||||
set -- $$($(1) $(SHELL) -c "$(2)" 2>&1); while [ "$$#" -gt 3 ]; do shift; done
|
||||
endef
|
||||
define TIMECMD.NOTGNU
|
||||
set -- $$($(1) -l $(SHELL) -c "$(2)" 2>&1; echo $$?); \
|
||||
set -- "$$1"s "$$(($$7/1024))"kB "$${60}"
|
||||
endef
|
||||
define TIMECMD.BASH
|
||||
TIMEFORMAT=$$'%3Rs'; \
|
||||
set -- $$({ time $(2); } 2>&1; echo $$?); set -- "$$1" "" "$$2"
|
||||
endef
|
||||
define TIMECMD.NONE
|
||||
$(2); set -- "" "" "$$?"
|
||||
endef
|
||||
TIMECMD := $(shell $(TIMECMD_detect))
|
||||
TIMEFUN := TIMECMD.$(word 2,$(TIMECMD))
|
||||
TIMECMD := $(word 1,$(TIMECMD))
|
||||
# Sample use of the timing macro: (returns shell commands as text)
|
||||
# $(call $(TIMEFUN),$(TIMECMD),$(MY_SHELL_COMMANDS))
|
||||
PRINT_OK = $(call COLOR_PRINT,'\033[0;32m',OK,"\n")
|
||||
PRINT_FAILED = $(call COLOR_PRINT,'\033[0;31m',FAILED,"\n")
|
||||
|
||||
ifneq (,$(filter test%,$(MAKECMDGOALS)))
|
||||
MAKEFLAGS += -k
|
||||
@@ -60,20 +25,16 @@ endif
|
||||
# Test runs of the examples/miniapps with parameters - check exit code
|
||||
mfem-test = \
|
||||
printf " $(3) [$(2) $(1) ... ]: "; \
|
||||
$(call $(TIMEFUN),$(TIMECMD),$(2) ./$(1) -no-vis $(4) > $(1).stderr 2>&1); \
|
||||
if [ "$$3" = 0 ]; \
|
||||
then $(PRINT_OK); else $(PRINT_FAILED); cat $(1).stderr; fi; \
|
||||
rm -f $(1).stderr; exit $$3
|
||||
if ($(2) ./$(1) -no-vis $(4) > /dev/null); \
|
||||
then $(PRINT_OK); else $(PRINT_FAILED); exit 1; fi
|
||||
|
||||
# Test runs of the examples/miniapps - check exit code and if a file exists
|
||||
mfem-test-file = \
|
||||
printf " $(3) [$(2) $(1) ... ]: "; \
|
||||
$(call $(TIMEFUN),$(TIMECMD),$(2) ./$(1) -no-vis > $(1).stderr 2>&1); \
|
||||
if [ "$$3" = 0 ] && [ -e $(4) ]; \
|
||||
then $(PRINT_OK); else $(PRINT_FAILED); cat $(1).stderr; fi; \
|
||||
rm -f $(1).stderr; exit $$3
|
||||
if ($(2) ./$(1) -no-vis > /dev/null) && [ -e $(4) ]; \
|
||||
then $(PRINT_OK); else $(PRINT_FAILED); exit 1; fi
|
||||
|
||||
.PHONY: test test-par-YES test-par-NO test-ser test-par test-clean test-print
|
||||
.PHONY: test test-par-YES test-par-NO
|
||||
|
||||
# What sets of tests to run in serial and parallel
|
||||
test-par-YES: $(PAR_$(MFEM_TESTS):=-test-par) $(SEQ_$(MFEM_TESTS):=-test-seq)
|
||||
@@ -81,12 +42,3 @@ test-par-NO: $(SEQ_$(MFEM_TESTS):=-test-seq)
|
||||
test-ser: test-par-NO
|
||||
test-par: test-par-YES
|
||||
test: all test-par-$(MFEM_USE_MPI) clean-exec
|
||||
test-clean: ; @rm -f *.stderr
|
||||
test-print: mfem-test=printf " $(3) [$(2) ./$(1) -no-vis $(if $(4),$(4) )]\n"
|
||||
test-print: mfem-test-file=printf " $(3) [$(2) ./$(1) -no-vis ]\n"
|
||||
test-print: test-par-$(MFEM_USE_MPI)
|
||||
ifeq ($(MAKECMDGOALS),test-print)
|
||||
.PHONY: $(PAR_$(MFEM_TESTS)) $(SEQ_$(MFEM_TESTS))
|
||||
endif
|
||||
|
||||
clean-exec: test-clean
|
||||
|
||||
@@ -1,122 +0,0 @@
|
||||
MFEM mesh v1.1
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see mesh/geom.hpp):
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
#
|
||||
|
||||
dimension
|
||||
2
|
||||
|
||||
elements
|
||||
17
|
||||
1 3 0 23 30 26
|
||||
1 3 26 30 25 21
|
||||
1 3 30 24 22 25
|
||||
1 3 23 18 24 30
|
||||
1 3 18 1 19 22
|
||||
1 3 22 19 10 20
|
||||
1 3 31 27 20 28
|
||||
1 3 25 22 27 31
|
||||
1 3 21 25 31 29
|
||||
1 3 29 31 28 9
|
||||
1 3 1 2 11 10
|
||||
1 3 2 3 12 11
|
||||
1 3 3 4 13 12
|
||||
2 3 4 5 14 13
|
||||
2 3 5 6 15 14
|
||||
2 3 6 7 16 15
|
||||
2 3 7 8 17 16
|
||||
|
||||
boundary
|
||||
25
|
||||
3 1 0 23
|
||||
1 1 26 0
|
||||
1 1 21 26
|
||||
3 1 23 18
|
||||
3 1 18 1
|
||||
3 1 10 20
|
||||
3 1 20 28
|
||||
1 1 29 21
|
||||
3 1 28 9
|
||||
1 1 9 29
|
||||
3 1 1 2
|
||||
3 1 11 10
|
||||
3 1 2 3
|
||||
3 1 12 11
|
||||
3 1 3 4
|
||||
3 1 13 12
|
||||
3 1 4 5
|
||||
3 1 14 13
|
||||
3 1 5 6
|
||||
3 1 15 14
|
||||
3 1 6 7
|
||||
3 1 16 15
|
||||
3 1 7 8
|
||||
2 1 8 17
|
||||
3 1 17 16
|
||||
|
||||
vertex_parents
|
||||
14
|
||||
18 0 1
|
||||
19 1 10
|
||||
20 9 10
|
||||
21 0 9
|
||||
22 18 20
|
||||
23 0 18
|
||||
24 18 22
|
||||
25 21 22
|
||||
26 0 21
|
||||
27 20 22
|
||||
28 9 20
|
||||
29 9 21
|
||||
30 23 25
|
||||
31 25 28
|
||||
|
||||
coarse_elements
|
||||
3
|
||||
3 0 3 2 1
|
||||
3 8 7 6 9
|
||||
3 17 4 5 18
|
||||
|
||||
vertices
|
||||
32
|
||||
2
|
||||
0 0
|
||||
1 0
|
||||
2 0
|
||||
3 0
|
||||
4 0
|
||||
5 0
|
||||
6 0
|
||||
7 0
|
||||
8 0
|
||||
0 1
|
||||
1 1
|
||||
2 1
|
||||
3 1
|
||||
4 1
|
||||
5 1
|
||||
6 1
|
||||
7 1
|
||||
8 1
|
||||
0.5 0
|
||||
1 0.5
|
||||
0.5 1
|
||||
0 0.5
|
||||
0.5 0.5
|
||||
0.25 0
|
||||
0.5 0.25
|
||||
0.25 0.5
|
||||
0 0.25
|
||||
0.5 0.75
|
||||
0.25 1
|
||||
0 0.75
|
||||
0.25 0.25
|
||||
0.25 0.75
|
||||
@@ -38,7 +38,7 @@ PROJECT_NAME = "MFEM"
|
||||
# could be handy for archiving the generated documentation or if some version
|
||||
# control system is used.
|
||||
|
||||
PROJECT_NUMBER = v3.4.1
|
||||
PROJECT_NUMBER = v3.3
|
||||
|
||||
# Using the PROJECT_BRIEF tag one can provide an optional one line description
|
||||
# for a project that appears at the top of each page and should give viewer a
|
||||
@@ -140,7 +140,7 @@ INLINE_INHERITED_MEMB = NO
|
||||
# shortest path that makes the file name unique will be used
|
||||
# The default value is: YES.
|
||||
|
||||
FULL_PATH_NAMES = YES
|
||||
FULL_PATH_NAMES = NO
|
||||
|
||||
# The STRIP_FROM_PATH tag can be used to strip a user-defined part of the path.
|
||||
# Stripping is only done if one of the specified strings matches the left-hand
|
||||
@@ -152,7 +152,7 @@ FULL_PATH_NAMES = YES
|
||||
# will be relative from the directory where doxygen is started.
|
||||
# This tag requires that the tag FULL_PATH_NAMES is set to YES.
|
||||
|
||||
STRIP_FROM_PATH = @MFEM_SOURCE_DIR@
|
||||
STRIP_FROM_PATH =
|
||||
|
||||
# The STRIP_FROM_INC_PATH tag can be used to strip a user-defined part of the
|
||||
# path mentioned in the documentation of a class, which tells the reader which
|
||||
@@ -760,8 +760,6 @@ WARN_LOGFILE =
|
||||
|
||||
INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
|
||||
@MFEM_SOURCE_DIR@/mfem.hpp \
|
||||
@MFEM_SOURCE_DIR@/backends/base \
|
||||
@MFEM_SOURCE_DIR@/backends/occa \
|
||||
@MFEM_SOURCE_DIR@/config \
|
||||
@MFEM_SOURCE_DIR@/general \
|
||||
@MFEM_SOURCE_DIR@/linalg \
|
||||
@@ -769,12 +767,10 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
|
||||
@MFEM_SOURCE_DIR@/fem \
|
||||
@MFEM_SOURCE_DIR@/examples \
|
||||
@MFEM_SOURCE_DIR@/examples/petsc \
|
||||
@MFEM_SOURCE_DIR@/examples/pumi \
|
||||
@MFEM_SOURCE_DIR@/examples/sundials \
|
||||
@MFEM_SOURCE_DIR@/miniapps/common \
|
||||
@MFEM_SOURCE_DIR@/miniapps/meshing \
|
||||
@MFEM_SOURCE_DIR@/miniapps/tools \
|
||||
@MFEM_SOURCE_DIR@/miniapps/nurbs \
|
||||
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
|
||||
@MFEM_SOURCE_DIR@/miniapps/performance
|
||||
|
||||
@@ -812,8 +808,7 @@ RECURSIVE = NO
|
||||
# Note that relative paths are relative to the directory from which doxygen is
|
||||
# run.
|
||||
|
||||
EXCLUDE = @MFEM_SOURCE_DIR@/config/_config.hpp \
|
||||
@MFEM_SOURCE_DIR@/config/get_hypre_version.cpp
|
||||
EXCLUDE =
|
||||
|
||||
# The EXCLUDE_SYMLINKS tag can be used to select whether or not files or
|
||||
# directories that are symbolic links (a Unix file system feature) are excluded
|
||||
@@ -1445,7 +1440,7 @@ FORMULA_TRANSPARENT = YES
|
||||
# The default value is: NO.
|
||||
# This tag requires that the tag GENERATE_HTML is set to YES.
|
||||
|
||||
USE_MATHJAX = YES
|
||||
USE_MATHJAX = NO
|
||||
|
||||
# When MathJax is enabled you can set the default output format to be used for
|
||||
# the MathJax output. See the MathJax site (see:
|
||||
@@ -1468,14 +1463,14 @@ MATHJAX_FORMAT = HTML-CSS
|
||||
# The default value is: http://cdn.mathjax.org/mathjax/latest.
|
||||
# This tag requires that the tag USE_MATHJAX is set to YES.
|
||||
|
||||
MATHJAX_RELPATH = https://cdn.llnl.gov/mathjax/2.7.2
|
||||
MATHJAX_RELPATH = http://www.mathjax.org/mathjax
|
||||
|
||||
# The MATHJAX_EXTENSIONS tag can be used to specify one or more MathJax
|
||||
# extension names that should be enabled during MathJax rendering. For example
|
||||
# MATHJAX_EXTENSIONS = TeX/AMSmath TeX/AMSsymbols
|
||||
# This tag requires that the tag USE_MATHJAX is set to YES.
|
||||
|
||||
MATHJAX_EXTENSIONS = TeX/AMSmath TeX/AMSsymbols
|
||||
MATHJAX_EXTENSIONS =
|
||||
|
||||
# The MATHJAX_CODEFILE tag can be used to specify a file with javascript pieces
|
||||
# of code that will be used on startup of the MathJax code. See the MathJax site
|
||||
|
||||
@@ -36,8 +36,8 @@ namespace mfem {
|
||||
* - HypreSolver and other \link hypre.hpp hypre classes\endlink
|
||||
*
|
||||
* <H3>Example codes</H3>
|
||||
* - <a class="el" href="examples_2ex1_8cpp_source.html">Example 1</a>: nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="examples_2ex1p_8cpp_source.html">Example 1p</a>: parallel nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="ex1_8cpp_source.html">Example 1</a>: nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="ex1p_8cpp_source.html">Example 1p</a>: parallel nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="ex2_8cpp_source.html">Example 2</a>: vector FEM for linear elasticity
|
||||
* - <a class="el" href="ex2p_8cpp_source.html">Example 2p</a>: parallel vector FEM for linear elasticity
|
||||
* - <a class="el" href="ex3_8cpp_source.html">Example 3</a>: Nedelec H(curl) FEM for the definite Maxwell problem
|
||||
@@ -56,7 +56,7 @@ namespace mfem {
|
||||
* - <a class="el" href="ex9p_8cpp_source.html">Example 9p</a>: parallel Discontinuous Galerkin (DG) time-dependent advection
|
||||
* - <a class="el" href="ex10_8cpp_source.html">Example 10</a>: time-dependent implicit nonlinear elasticity
|
||||
* - <a class="el" href="ex10p_8cpp_source.html">Example 10p</a>: parallel time-dependent implicit nonlinear elasticity
|
||||
* - <a class="el" href="examples_2ex11p_8cpp_source.html">Example 11p</a>: parallel Laplace eigensolver
|
||||
* - <a class="el" href="ex11p_8cpp_source.html">Example 11p</a>: parallel Laplace eigensolver
|
||||
* - <a class="el" href="ex12p_8cpp_source.html">Example 12p</a>: parallel linear elasticity eigensolver
|
||||
* - <a class="el" href="ex13p_8cpp_source.html">Example 13p</a>: parallel Maxwell eigensolver
|
||||
* - <a class="el" href="ex14_8cpp_source.html">Example 14</a>: Discontinuous Galerkin (DG) for the Laplace problem
|
||||
@@ -67,20 +67,14 @@ namespace mfem {
|
||||
* - <a class="el" href="ex16p_8cpp_source.html">Example 16p</a>: parallel time-dependent nonlinear heat equation
|
||||
* - <a class="el" href="ex17_8cpp_source.html">Example 17</a>: Discontinuous Galerkin (DG) for linear elasticity
|
||||
* - <a class="el" href="ex17p_8cpp_source.html">Example 17p</a>: parallel Discontinuous Galerkin (DG) for linear elasticity
|
||||
* - <a class="el" href="ex18_8cpp_source.html">Example 18</a>: Discontinuous Galerkin (DG) for the Euler equations
|
||||
* - <a class="el" href="ex18p_8cpp_source.html">Example 18p</a>: parallel Discontinuous Galerkin (DG) for the Euler equations
|
||||
* - <a class="el" href="ex19_8cpp_source.html">Example 19</a>: incompressible nonlinear elasticity
|
||||
* - <a class="el" href="ex19p_8cpp_source.html">Example 19p</a>: parallel incompressible nonlinear elasticity
|
||||
*
|
||||
* <H4>SUNDIALS Examples</H4>
|
||||
* - Variants of Examples
|
||||
* <a class="el" href="sundials_2ex9_8cpp_source.html">9</a>,
|
||||
* <a class="el" href="sundials_2ex9p_8cpp_source.html">9p</a>,
|
||||
* <a class="el" href="sundials_2ex10_8cpp_source.html">10</a>,
|
||||
* <a class="el" href="sundials_2ex10p_8cpp_source.html">10p</a>,
|
||||
* <a class="el" href="sundials_2ex16_8cpp_source.html">16</a>,
|
||||
* and
|
||||
* <a class="el" href="sundials_2ex16p_8cpp_source.html">16p</a>
|
||||
* <a class="el" class="el" href="sundials_2ex10p_8cpp_source.html">10p</a>
|
||||
* demonstrating the use of MFEM's \link sundials.hpp SUNDIALS classes\endlink
|
||||
*
|
||||
* <H4>PETSc Examples</H4>
|
||||
@@ -96,28 +90,14 @@ namespace mfem {
|
||||
* <a class="el" href="petsc_2ex10p_8cpp_source.html">10p</a>
|
||||
* demonstrating the use of MFEM's \link petsc.hpp PETSc classes\endlink
|
||||
*
|
||||
* <H4>PUMI Examples</H4>
|
||||
* - Variants of Examples
|
||||
* <a class="el" href="examples_2pumi_2ex1_8cpp_source.html">1</a>,
|
||||
* <a class="el" href="examples_2pumi_2ex1p_8cpp_source.html">1p</a>,
|
||||
* <a class="el" href="pumi_2ex2_8cpp_source.html">2</a>,
|
||||
* and
|
||||
* <a class="el" href="pumi_2ex6p_8cpp_source.html">6p</a>
|
||||
* demonstrating the use of MFEM's \link pumi.hpp PUMI classes\endlink
|
||||
*
|
||||
* <H3>Miniapps</H3>
|
||||
* - <a class="el" href="volta_8cpp_source.html">Volta</a>: simple electrostatics simulation code
|
||||
* - <a class="el" href="tesla_8cpp_source.html">Tesla</a>: simple magnetostatics simulation code
|
||||
* - <a class="el" href="maxwell_8cpp_source.html">Maxwell</a>: simple transient full-wave electromagnetics simulation code
|
||||
* - <a class="el" href="joule_8cpp_source.html">Joule</a>: transient magnetics and Joule heating miniapp
|
||||
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
|
||||
* - <a class="el" href="klein-bottle_8cpp_source.html">Klein Bottle</a>: generate three types of Klein bottle surfaces
|
||||
* - <a class="el" href="shaper_8cpp_source.html">Shaper</a>: resolve material interfaces by mesh refinement
|
||||
* - <a class="el" href="mesh-explorer_8cpp_source.html">Mesh Explorer</a>: visualize and manipulate meshes
|
||||
* - <a class="el" href="mesh-optimizer_8cpp_source.html">Mesh Optimizer</a>: optimize high-order meshes, <a class="el" href="mesh-optimizer_8cpp_source.html">serial</a> and <a class="el" href="pmesh-optimizer_8cpp_source.html">parallel</a> versions
|
||||
* - <a class="el" href="display-basis_8cpp_source.html">Display Basis</a>: visualize finite element basis functions
|
||||
* - <a class="el" href="load-dc_8cpp_source.html">Load DC</a>: visualize fields saved via DataCollection classes
|
||||
* - <a class="el" href="convert-dc_8cpp_source.html">Convert DC</a>: convert between diffirent DataCollection formats
|
||||
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Laplace problem
|
||||
*
|
||||
|
||||
+6
-6
@@ -10,17 +10,17 @@
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
MFEM_DIR ?= ..
|
||||
DOXYGEN_CONF = CodeDocumentation.conf
|
||||
DOXYGEN_CONG = CodeDocumentation.conf
|
||||
|
||||
# doxygen uses: graphviz, latex
|
||||
html: $(DOXYGEN_CONF)
|
||||
doxygen $(DOXYGEN_CONF)
|
||||
html: $(DOXYGEN_CONG)
|
||||
doxygen $(DOXYGEN_CONG)
|
||||
rm -f CodeDocumentation.html
|
||||
ln -s CodeDocumentation/html/index.html CodeDocumentation.html
|
||||
|
||||
clean:
|
||||
rm -rf $(DOXYGEN_CONF) CodeDocumentation CodeDocumentation.html *~
|
||||
rm -rf $(DOXYGEN_CONG) CodeDocumentation CodeDocumentation.html *~
|
||||
|
||||
$(DOXYGEN_CONF): $(MFEM_DIR)/doc/$(DOXYGEN_CONF).in
|
||||
$(DOXYGEN_CONG): $(MFEM_DIR)/doc/$(DOXYGEN_CONG).in
|
||||
sed -e 's%@MFEM_SOURCE_DIR@%$(MFEM_DIR)%g' $(<) \
|
||||
> $(DOXYGEN_CONF)
|
||||
> $(DOXYGEN_CONG)
|
||||
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 34 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 12 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 44 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 95 KiB |
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user