Compare commits

..
332 changed files with 9154 additions and 54351 deletions
-49
View File
@@ -1,49 +0,0 @@
version: '{build}'
# https://www.appveyor.com/docs/build-environment/#build-worker-images
image: Visual Studio 2017
install:
# Install MS-MPI
- ps: Start-FileDownload 'https://download.microsoft.com/download/B/2/E/B2EB83FE-98C2-4156-834A-E1711E6884FB/MSMpiSetup.exe'
- MSMpiSetup.exe -unattend
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
# Install MS-MPI SDK
- ps: Start-FileDownload 'https://download.microsoft.com/download/B/2/E/B2EB83FE-98C2-4156-834A-E1711E6884FB/msmpisdk.msi'
- msmpisdk.msi /passive
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
# Install METIS
- ps: Start-FileDownload 'http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz'
- 7z x metis-5.1.0.tar.gz -so | 7z x -si -ttar > nul
- cd metis-5.1.0
- ps: ( get-content "GKlib\gk_arch.h") | % { If ($_.ReadCount -ge 52) {$_ -replace "#ifdef __MSC__","#ifdef DISABLE_THIS_ANCIENT_MSC_CHECK"} Else {$_} } | set-content "GKlib\gk_arch.h"
- cmake -H. -Bbuild
# -DCMAKE_BUILD_TYPE=Release
- cmake --build build
- cd ..
# Install hypre
- ps: Start-FileDownload 'https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz'
- 7z x hypre-2.10.0b.tar.gz -so | 7z x -si -ttar > nul
- cd hypre-2.10.0b
- cmake -Hsrc -Bbuild -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
# - cmake -Hsrc -Bbuild -DCMAKE_BUILD_TYPE=Release -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
- cmake --build build
- cmake --build build --target install
- cd ..
# MFEM
before_build:
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2.10.0b\src\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2.10.0b\src\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_serial -DMFEM_USE_MPI=FALSE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2.10.0b\src\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2.10.0b\src\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
build_script:
- cmake --build build_parallel
- cmake --build build_serial
after_build:
# - cmake --build build_parallel --target check
- cmake --build build_serial --target check
-158
View File
@@ -1,158 +0,0 @@
# ------------------------------------------------------------------------------
# Ignore files that are generated from the repository sources by either building
# the code or running it. These should be the same as the files erased by
# `make distclean`.
#
# Also ignore OS-specific files like .DS_Store on Mac
# ------------------------------------------------------------------------------
# Object and library files
*.o
/libmfem.*
# CMake generated files
CMakeCache.txt
CMakeFiles/
# Backup files
*~
# Default install location
/mfem/
# Generated files in main directory, config/ and docs/
/deps.mk
config/_config.hpp
config/config.mk
config/sample-runs-build.log
doc/CodeDocumentation.conf
doc/CodeDocumentation.html
doc/CodeDocumentation
# Temporary files created by the tests.
*.stderr
# Totalview breakpoint files
*.TVD.*breakpoints
# OS-specific: Mac
*.dSYM
.DS_Store
# Example and miniapp binaries and outputs
examples/ex[1-9]
examples/ex[1-9]p
examples/ex1[04-9]
examples/ex1[0-9]p
examples/refined.mesh
examples/displaced.mesh
examples/mesh.*
examples/ex5.mesh
examples/Example5*
examples/Example9*
examples/Example15*
examples/Example16*
examples/sphere_refined.*
examples/sol.*
examples/sol_u.*
examples/sol_p.*
examples/ex9.mesh
examples/ex9-mesh.*
examples/ex9-init.*
examples/ex9-final.*
examples/deformed.*
examples/velocity.*
examples/elastic_energy.*
examples/mode_*
examples/ex16.mesh
examples/ex16-mesh.*
examples/ex16-init.*
examples/ex16-final.*
examples/vortex-mesh.*
examples/vortex.mesh
examples/vortex-?-init.*
examples/vortex-?-final.*
examples/deformation.*
examples/pressure.*
examples/sundials/ex9
examples/sundials/ex1[06]
examples/sundials/ex9p
examples/sundials/ex1[06]p
examples/sundials/ex9.mesh
examples/sundials/ex9-mesh.*
examples/sundials/ex9-init.*
examples/sundials/ex9-final.*
examples/sundials/Example9*
examples/sundials/deformed.*
examples/sundials/velocity.*
examples/sundials/elastic_energy.*
examples/sundials/ex16.mesh
examples/sundials/ex16-mesh.*
examples/sundials/ex16-init.*
examples/sundials/ex16-final.*
examples/sundials/Example16*
examples/petsc/ex[1-69]p
examples/petsc/ex10p
examples/petsc/mesh.*
examples/petsc/sol.*
examples/petsc/sol_p.*
examples/petsc/sol_u.*
examples/petsc/Example5*
examples/petsc/ex9-mesh.*
examples/petsc/ex9-init.*
examples/petsc/ex9-final.*
examples/petsc/Example9*
examples/petsc/deformed.*
examples/petsc/velocity.*
examples/petsc/elastic_energy.*
miniapps/electromagnetics/volta
miniapps/electromagnetics/tesla
miniapps/electromagnetics/maxwell
miniapps/electromagnetics/joule
miniapps/electromagnetics/Volta-AMR*
miniapps/electromagnetics/Tesla-AMR*
miniapps/electromagnetics/Maxwell-Parallel*
miniapps/electromagnetics/Joule_*
miniapps/meshing/mobius-strip
miniapps/meshing/klein-bottle
miniapps/meshing/mesh-explorer
miniapps/meshing/shaper
miniapps/meshing/mesh-optimizer
miniapps/meshing/pmesh-optimizer
miniapps/meshing/mobius-strip.mesh
miniapps/meshing/klein-bottle.mesh
miniapps/meshing/mesh-explorer.mesh
miniapps/meshing/partitioning.txt
miniapps/meshing/shaper.mesh
miniapps/meshing/optimized*
miniapps/meshing/perturbed*
miniapps/performance/ex1
miniapps/performance/ex1p
miniapps/performance/refined.mesh
miniapps/performance/mesh.*
miniapps/performance/sol.*
miniapps/tools/display-basis
miniapps/tools/load-dc
miniapps/tools/convert-dc
miniapps/nurbs/ex1
miniapps/nurbs/ex1p
miniapps/nurbs/ex11p
miniapps/nurbs/refined.mesh
miniapps/nurbs/mesh.*
miniapps/nurbs/sol.*
miniapps/nurbs/mode_*
miniapps/nurbs/Example1*
+67 -230
View File
@@ -1,207 +1,55 @@
sudo: false
language: cpp
matrix:
include:
#
# Linux
#
- os: linux
compiler: gcc
env: DEBUG=YES
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=check
#
- os: linux
compiler: gcc
env: DEBUG=NO
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=test
#
- os: linux
compiler: gcc
addons:
apt:
# sources:
# - ubuntu-toolchain-r-test
packages:
# GCC 4.9
# - g++-4.9
# MPICH
- mpich
- libmpich-dev
# OpenMPI
# - openmpi-bin
# - libopenmpi-dev
env: DEBUG=YES
MPI=YES
CODECOV=NO
MFEM_TEST_TARGET=check
NPROCS=2
cache:
directories:
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../metis-4.0
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
#
- os: linux
compiler: gcc
addons:
apt:
# sources:
# - ubuntu-toolchain-r-test
packages:
# GCC 4.9
# - g++-4.9
# MPICH
- mpich
- libmpich-dev
# OpenMPI
# - openmpi-bin
# - libopenmpi-dev
env: DEBUG=NO
MPI=YES
CODECOV=YES
MFEM_TEST_TARGET=test
NPROCS=2
cache:
directories:
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../metis-4.0
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
#
# Mac OS X
#
- os: osx
# osx_image: xcode7.3
compiler: clang
env: DEBUG=YES
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=check
#
- os: osx
# osx_image: xcode7.3
compiler: clang
env: DEBUG=NO
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=test
#
- os: osx
# osx_image: xcode7.3
compiler: clang
env: DEBUG=YES
MPI=YES
CODECOV=NO
MFEM_TEST_TARGET=check
NPROCS=4
TMPDIR=/tmp
cache:
directories:
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../metis-4.0
- $HOME/local-cached
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
#
- os: osx
# osx_image: xcode7.3
compiler: clang
env: DEBUG=NO
MPI=YES
CODECOV=YES
MFEM_TEST_TARGET=test
NPROCS=4
TMPDIR=/tmp
cache:
directories:
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../metis-4.0
- $HOME/local-cached
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
compiler:
- gcc
- clang
os:
- linux
- osx
env:
- DEBUG=YES MPI=YES TMPDIR=/tmp
- DEBUG=NO MPI=YES TMPDIR=/tmp
- DEBUG=YES MPI=NO
- DEBUG=NO MPI=NO
# Test with GCC on Linux an Clang on Mac
matrix:
exclude:
- compiler: clang
os: linux
- compiler: gcc
os: osx
before_install:
# No addon for brew yet, have to install OSX packages this way.
# - if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
# brew install open-mpi;
# fi
# On Mac OS X, build and cache OpenMPI 2.1.1:
- if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
if [ ! -e $HOME/local-cached/bin/mpicc ]; then
mkdir -p $HOME/builds && cd $HOME/builds &&
wget https://www.open-mpi.org/software/ompi/v2.1/downloads/openmpi-2.1.1.tar.bz2 &&
tar jxf openmpi-2.1.1.tar.bz2 &&
mkdir openmpi-build && cd openmpi-build &&
../openmpi-2.1.1/configure --prefix=$HOME/local-cached &&
make -j3 all && make install;
fi;
PATH=$HOME/local-cached/bin:$PATH;
cd $TRAVIS_BUILD_DIR;
fi
# Update environment to find g++ 4.9 installation first.
# - if [ $TRAVIS_OS_NAME == "linux" ]; then
# mkdir -p latest-gcc-symlinks;
# ln -s /usr/bin/g++-4.9 latest-gcc-symlinks/g++;
# ln -s /usr/bin/gcc-4.9 latest-gcc-symlinks/gcc;
# ln -s /usr/bin/gcov-4.9 latest-gcc-symlinks/gcov;
# export PATH=$PWD/latest-gcc-symlinks:$PATH;
# fi
# Install tool to upload code coverage reports to coveralls.io
- if [ "$CODECOV" == "YES" ]; then
export PYTHONUSERBASE=$HOME/local;
pip install --user cpp-coveralls;
pip install --user pyyaml;
PATH=$HOME/local/bin:$PATH;
fi
- if [ $TRAVIS_OS_NAME == "linux" -a "$CXX" == "g++" ]; then sudo add-apt-repository -y ppa:ubuntu-toolchain-r/test; fi
- if [ $TRAVIS_OS_NAME == "linux" ]; then sudo apt-get update; fi || true
install:
# Set MPI compilers, print compiler version
- if [ $MPI == "YES" ]; then
if [ "$TRAVIS_OS_NAME" == "linux" ]; then
export MPICH_CC="$CC";
export MPICH_CXX="$CXX";
else
export OMPI_CC="$CC";
export OMPI_CXX="$CXX";
mpic++ --showme:version;
fi;
mpic++ -v;
else
$CXX -v;
fi
# g++-4.9
- if [ $TRAVIS_OS_NAME == "linux" -a "$CXX" == "g++" ]; then sudo apt-get install -qq g++-4.9; fi
- if [ $TRAVIS_OS_NAME == "linux" -a "$CXX" == "g++" ]; then export CXX="g++-4.9"; fi
# Back out of the mfem directory to install the libraries
- cd ..
# OpenMPI
- if [ $TRAVIS_OS_NAME == "linux" ]; then
sudo apt-get install openmpi-bin openmpi-common openssh-client openssh-server libopenmpi1.3 libopenmpi-dbg libopenmpi-dev;
else
travis_wait brew install open-mpi;
fi
# hypre
- if [ $MPI == "YES" ]; then
if [ ! -e hypre-2.10.0b/src/hypre/lib/libHYPRE.a ]; then
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
rm -rf hypre-2.10.0b;
tar xvzf hypre-2.10.0b.tar.gz;
cd hypre-2.10.0b/src;
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
make -j3;
cd ../..;
if [ ! -d hypre-2.10.0b ]; then
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
tar xvzf hypre-2.10.0b.tar.gz;
cd hypre-2.10.0b/src;
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
make -j 4;
cd ../..;
else
echo "Reusing cached hypre-2.10.0b/";
fi;
@@ -210,54 +58,43 @@ install:
fi
# METIS
- if [ $MPI == "YES" ]; then
if [ ! -e metis-4.0/libmetis.a ]; then
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
tar xvzf metis-4.0.3.tar.gz;
make -j3 -C metis-4.0.3/Lib CC="$CC" OPTFLAGS="-O2";
rm -rf metis-4.0;
mv metis-4.0.3 metis-4.0;
else
echo "Reusing cached metis-4.0/";
fi;
- if [ ! -d metis-4.0 ]; then
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
tar xvzf metis-4.0.3.tar.gz;
cd metis-4.0.3;
make -j 4;
cd ..;
mv metis-4.0.3 metis-4.0;
else
echo "Reusing cached metis-4.0/";
fi
# # Delete an expired cache here: https://travis-ci.org/mfem/mfem/caches
# cache:
# directories:
# - $TRAVIS_BUILD_DIR/../hypre-2.10.0b
# - $TRAVIS_BUILD_DIR/../metis-4.0
script:
# Compiler
- if [ $MPI == "YES" ]; then
export MYCXX=mpic++;
export OMPI_CXX="$CXX";
$MYCXX --showme:version;
else
export MYCXX="$CXX";
fi
# Print the compiler version
- $MYCXX -v
# Set some variables
- cd $TRAVIS_BUILD_DIR;
CPPFLAGS="";
SKIP_TEST_DIRS="";
if [ "$CODECOV" == "YES" ]; then
CPPFLAGS="--coverage -g";
fi;
if [ "$CXX" == "clang++" ]; then
export MFEM_PERF_SW=clang;
fi
# Configure the library
- make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG MFEM_CXX="$MYCXX"
MFEM_MPI_NP=$NPROCS CPPFLAGS="$CPPFLAGS"
# Show the configuration
- make info
# Build the library
- make -j3
# Build the examples and the miniapps
- make -j3 all
# Run tests
- make $MFEM_TEST_TARGET SKIP_TEST_DIRS="$SKIP_TEST_DIRS"
after_success:
- if [ "$CODECOV" == "YES" ]; then
coveralls --include fem --include general --include linalg --include
mesh --exclude /usr --gcov-options '\-lp' --root $TRAVIS_BUILD_DIR;
# Build the code and do a quick check (debug mode) or a full tests run (non-debug mode)
- if [ $DEBUG == "NO" ]; then
export MFEM_TEST_TARGET="test";
else
export MFEM_TEST_TARGET="check";
fi
# Build and check/test MFEM, its examples and miniapps
- cd $TRAVIS_BUILD_DIR &&
make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG MFEM_CXX="$MYCXX" &&
make info &&
make all -j 4 &&
make $MFEM_TEST_TARGET
+8 -271
View File
@@ -8,279 +8,16 @@
http://mfem.org
Version 3.3.3 (development)
===========================
Development version, not released yet
=====================================
More efficient non-conforming adaptive mesh refinement
------------------------------------------------------
- Significantly reduced MPI communication in the construction of the parallel
prolongation matrix in ParFiniteElementSpace, for much improved parallel
scaling of non-conforming AMR on hundreds of thousands of MPI tasks. The
memory footprint of the ParNCMesh class has also been reduced.
- In FiniteElementSpace, the fully assembled refinement matrix is now replaced
by default by a specialized refinement operator. The operator option is both
faster and more memory efficient than using the fully assembled matrix. The
old approach is still available and can be enabled, if needed, using the new
method FiniteElementSpace::SetUpdateOperatorType().
Discretization improvements
---------------------------
- Added support for a general "high-order"-to-"low-order refined" transfer of
GridFunction and true-dof data from a "high-order" finite element space
defined on a coarse mesh, to a "low-order refined" space defined on a refined
mesh. The new methods, GetTransferOperator and GetTrueTransferOperator in the
FiniteElementSpace classes, work in both serial and parallel and support
matrix-based as well as matrix-free transfer operator representations. They
use a new method, GetTransferMatrix, in the FiniteElement class similar to
GetLocalInterpolation, that allows the coarse FiniteElement to be different
from the fine FiniteElement.
- Added class ComplexOperator, that implements the action of a complex operator
through the equivalent 2x2 real formulation. Both symmetric and antisymmetric
block structures are supported.
- Added classes for general block nonlinear finite element operators (deriving
from BlockNonlinearForm and ParBlockNonlinearForm) enabling solution of
nonlinear systems with multiple unknowns in different function spaces. Such
operators have assemble-based action and also support assembly of the gradient
operator to enable inversion with Newton iteration.
- Added variable order NURBS: for each space each knot vector in the mesh can
have a different order. The order information is now part of the finite
element space header in the NURBS mesh output, so NURBS meshes in the old
format need to be updated.
- In the classes NonlinearForm and ParNonlinearForm, added support for
non-conforming AMR meshes; see also the "API changes" section.
- New specialized time integrators: symplectic integrators of orders 1-4 for
systems of first order ODEs derived from a Hamiltonian and generalized-alpha
ODE solver for the filtered NavierStokes equations with stabilization. See
classes SIASolver and GeneralizedAlphaSolver in linalg/ode.hpp.
- Inherit finite element classes from the new base class TensorBasisElement,
whenever the basis can be represented by a tensor product of 1D bases.
- Added support for elimination of boundary conditions in block matrices.
New and updated examples and miniapps
-------------------------------------
- Added a new serial and parallel example (ex19) that solves the quasi-static
incompressible hyperelastic equations. The example demonstrates the use of
block nonlinear forms as well as custom block preconditioners.
- Added a new electromagnetics miniapp, Maxwell, for simulating time-domain
electromagnetics phenomena as a coupled first order system of equations.
- A simple local refinement option has been added to the mesh-explorer miniapp
(menu option 'r', sub-option 'l') that selects elements for refinement based
on their spatial location - see the function 'region()' in the source file.
- Added a set of miniapps specifically focused on Isogeometric Analysis (IGA) on
NURBS meshes in the miniapps/nurbs directory. Currently the directory contains
variable order NURBS versions of examples 1, 1p and 11p.
- Added two new miniapps related to DataCollection I/O in miniapps/tools:
load-dc.cpp can be used to visualize fields saved via DataCollection classes;
convert-dc.cpp demonstrates how to convert between MFEM's different concrete
DataCollection options.
- Example 10p with its SUNDIALS and PETSc versions have been updated to reflect
the change in the behavior of the method ParNonlinearForm::GetLocalGradient()
(see the "API changes" section) and now works correctly on non-conforming AMR
meshes. Example 10 and its SUNDIALS version have also been updated to support
non-conforming ARM meshes.
Miscellaneous
-------------
- Documented project workflow and provided contribution guidelines in the new
top-level file, CONTRIBUTING.md.
- Added (optional) Conduit Mesh Blueprint support of MFEM data for both in-core
and I/O use cases. This includes a new ConduitDataCollection that provides
json, simple binary, and HDF5-based I/O. Support requires Conduit >= v0.3.1
and VisIt >= v2.13.1 will read the new Data Collection outputs.
- Added a new developer tool, config/sample-runs.sh, that extracts the sample
runs from all examples and miniapps and runs them. Optionally, it can save the
output from the execution to files, allowing comparison between different
versions and builds of the library.
- Support for building a shared version of the MFEM library with GNU make.
- Added a build option, MFEM_USE_EXCEPTIONS=YES, to throw an exception instead
of calling abort on mfem errors.
- When building with the GnuTLS library, switch to using X.509 certificates for
secure socket authentication. Support for the previously used OpenPGP keys has
been deprecated in GnuTLS 3.5.x and removed in 3.6.0. For secure communication
with the visualization tool GLVis, a new set of certificates can be generated
using the latest version of the script 'glvis-keygen.sh' from GLVis.
- Upgraded MFEM to support Axom 0.2.8. Prior versions are no longer supported.
API changes
-----------
- Introduced a new enum, Matrix::DiagonalPolicy, that replaces the integer
parameters in many methods that perform elimination of rows and/or columns in
matrices. Some examples of such methods are:
* class SparseMatrix: EliminateRow(), EliminateCol(), EliminateRowCol(), ...
* class BilinearForm: EliminateEssentialBC(), EliminateVDofs(), ...
* class StaticCondensation: EliminateReducedTrueDofs()
* class BlockMatrix: EliminateRowCol()
Calling these methods with an explicitly given (integer) constants, will now
generate compilation errors, please use one of the new enum constants instead.
- Modified the virtual method AbstractSparseMatrix::EliminateZeroRows() and its
implementations in derived classes, to accept an optional 'threshold'
parameter, replacing previously hard-coded threshold values.
- In the classes NonlinearForm and ParNonlinearForm:
* The method GetLocalGradient() no longer imposes boundary conditions. The
motivation for the change is that, in the case of non-conforming AMR,
performing the elimination at the local level is incorrect - it must be
applied at the true-dof level.
* The method SetEssentialVDofs() is now deprecated.
Version 3.3.2, released on Nov 10, 2017
=======================================
High-order mesh optimization
----------------------------
- Added support for mesh optimization via node-movement based on the Target-
Matrix Optimization Paradigm (TMOP) developed by P.Knupp et al. A variety of
mesh quality metrics, with their first and second derivatives have been
implemented. The combination of targets & quality metrics is used to optimize
the physical node positions, i.e., they must be as close as possible to the
shape, size and/or alignment of their targets. The optimization of arbitrary
high-order meshes in 2D, 3D, serial and parallel is supported.
- The new Mesh Optimizer miniapp can be used to perform mesh optimization with
TMOP in serial and parallel versions. The miniapp also demonstrates the use of
nonlinear operators and their coupling to Newton methods for solving
minimization problems.
New and improved solvers and preconditioners
--------------------------------------------
- MFEM is now included in the xSDK project, the Extreme-scale Scientific
Software Development Kit, as of xSDK-0.3.0. Various changes were made to
comply with xSDK's community policies, https://xsdk.info/policies, including:
xSDK-specific options in CMake, support for user-provided MPI communicators,
runtime API for version number, and the ability to disable/redirect output.
For more details, see general/globals.hpp and in particular the mfem::err and
mfem::out streams replacing std::err and std::out respectively.
- Added (optional) support for the STRUMPACK parallel sparse direct solver and
preconditioner. STRUMPACK uses Hierarchically Semi-Separable (HSS) compression
in a fully algebraic manner, with interface similar to SuperLU_DIST. See
http://portal.nersc.gov/project/sparse/strumpack for more details.
- Added a block lower triangular preconditioner based (only) on the actions of
each block, see class BlockLowerTriangularPreconditioner.
- Added an optional operator in LOBPCG to projects vectors onto a desired
subspace (e.g. divergence-free). Other small changes in LOBPCG include the
ability to set the starting vectors and support for relative tolerance.
- The Newton solver supports an optional scaling factor, that can limit the
increment in the Newton step, see e.g. the Mesh Optimizer miniapp.
- Updated MFEM integration to support the new SUNDIALS 3.0.0 interface.
New and updated examples and miniapps
-------------------------------------
- Added a new serial and parallel example (ex18) that solves the transient Euler
equations on a periodic domain with explicit time integrators. In the process
extended the NonlinearForm class to allow for integrals over faces and
exchanging face-neighbor data in parallel.
- Added a new meshing miniapp, Shaper, that can be used to resolve complicated
material interfaces by mesh refinement, e.g. as a tool for initial mesh
generation from prescribed "material()" function. Both conforming and
non-conforming (isotropic and anisotropic) refinements are supported.
- Added a new meshing miniapp, Mesh Optimizer, that demonstrates the use of TMOP
for mesh optimization (serial and parallel version.)
- Added SUNDIALS version of Example 16/16p.
Discretization improvements
---------------------------
- Added a FindPoints method of the Mesh and ParMesh classes that returns the
elements that contain a given set of points, together with the coordinates of
the points in the reference space of the corresponding element. In parallel,
if a point is shared by multiple processors, only one of them will mark that
point as found. Note that the current implementation of this method is not
optimal and/or 100% reliable. See the mesh-explorer miniapp for an example.
- Added a new class InverseElementTransformation, that supports a number of
algorithms for inversion of general ElementTransformations. This class can be
used as a more flexible and extensible alternative to ElementTransformation's
TransformBack method. It is also used in the FindPoints methods as a tunable
and customizable inversion algorithm.
- Memory optimizations in the NCMesh class, which now uses 50% less memory than
before. The average cost of an element in a uniformly refined mesh (including
the refinement hierarchy, but excluding the temporary face_list and edge_list)
- Memory optimizations in the NCMesh class, which now uses 50% less memory.
The average cost of an NC element in a uniformly refined mesh (including the
refinement hierarchy, but excluding the temporary face_list and edge_list)
is now only about 290 bytes. This also makes the class faster.
- Added the ability to integrate delta functions on the right-hand side (by
sampling the test function at the center of the delta coefficient). Currently
this is supported in the DomainLFIntegrator, VectorDomainLFIntegrator and
VectorFEDomainLFIntegrator classes.
- Added five new linear interpolators in fem/bilininteg.cpp to compute products
of scalar and vector fields or products with arbitrary coefficients.
- Added matrix coefficient support to CurlCurlIntegrator.
- Extend the method NodalFiniteElement::Project for VectorCoefficient to work
with arbitrary number of vector components.
Miscellaneous
-------------
- Added a .gitignore file that ignores all files erased by "make distclean",
i.e. the files that can be generated from the source but we don't want to
track in the repository, as well as a few platform-specific files.
- Added Linux, Mac and Windows CI testing on GitHub with Travis CI and Appveyor.
- Added a new macro, MFEM_VERSION, defined as a single integer of the form
(major*100 + minor)*100 + patch. The convention is that an even number
(i.e. even patch number) denotes a "release" version, while an odd number
denotes a "development" version. See config/config.hpp.in.
- Added an option for building in parallel without a METIS dependency. This is
used for example the Laghos miniapp, https://github.com/CEED/Laghos.
- Modified the installation layout: all headers, except the master headers
(mfem.hpp and mfem-performance.hpp), are installed in <PREFIX>/include/mfem;
the master headers are installed in both <PREFIX>/include/mfem and in
<PREFIX>/include. The mfem configuration and testing makefiles (config.mk and
test.mk) are installed in <PREFIX>/share/mfem, instead of <PREFIX>.
- Add three more options for MFEM_TIMER_TYPE.
- Support independent number of digits for cycle and rank in DataCollection.
- Converted Sidre usage from "asctoolkit" to "axom" namespace.
- Various small fixes and styling updates.
API changes
-----------
- The methods GetCoeff of VectorArrayCoefficient and MatrixArrayCoefficient now
return a pointer to Coefficient (instead of reference). Note that NULL pointer
is a valid entry for these two classes - it is treated as the zero function.
- When building with PETSc, the required PETSc version is now 3.8.0. Newer
versions may work too, as long as there are no interface changes in PETSc.
- The class GeometryRefiner now uses the enum in Quadrature1D for its type
specification. In particular, this will affect older versions of GLVis. A
simple upgrade to the latest version of GLVis should resolve this issue.
- Add a block lower triangular preconditioner in using a matrix-free
implementation, see class BlockLowerTriangularPreconditioner.
Version 3.3, released on Jan 28, 2017
@@ -436,7 +173,7 @@ Improved file output
- Added experimental support for an HDF5-based output file format following the
Conduit (https://github.com/LLNL/conduit) mesh blueprint specification for
visualization and/or restart capability. This functionality is aimed primarily
at user of LLNL's axom project (Sidre component) that run problems at extreme
at user of LLNL's ASC Toolkit (Sidre component) that run problems at extreme
scales. Users desiring a small scale binary format may want to look at the
gzstream functionality instead.
+46 -133
View File
@@ -38,14 +38,8 @@ if (NOT CMAKE_CXX_COMPILER)
endif()
endif()
#-------------------------------------------------------------------------------
# Project name and version
#-------------------------------------------------------------------------------
project(mfem NONE)
# Current version of MFEM, see also `makefile`.
# mfem_VERSION = (string)
# MFEM_VERSION = (int) [automatically derived from mfem_VERSION]
set(${PROJECT_NAME}_VERSION 3.3.3)
project(mfem CXX)
set(${PROJECT_NAME}_VERSION 3.3)
# Prohibit in-source build
if (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
@@ -53,65 +47,18 @@ if (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
"MFEM does not support in-source CMake builds at this time.")
endif (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
# Set xSDK defaults.
set(USE_XSDK_DEFAULTS_DEFAULT OFF)
set(XSDK_ENABLE_CXX ON)
set(XSDK_ENABLE_C OFF)
set(XSDK_ENABLE_Fortran OFF)
# Check if we need to enable C or Fortran.
if (CMAKE_VERSION VERSION_LESS 3.2 OR
MFEM_USE_CONDUIT OR
MFEM_USE_SIDRE OR
MFEM_USE_PETSC)
if (CMAKE_VERSION VERSION_LESS 3.2 OR MFEM_USE_SIDRE)
# This seems to be needed by:
# * find_package(BLAS REQUIRED) and
# * find_package(HDF5 REQUIRED) needed, in turn, by:
# - find_package(AXOM REQUIRED)
# * find_package(PETSc REQUIRED)
set(XSDK_ENABLE_C ON)
endif()
if (MFEM_USE_STRUMPACK)
# Just needed to find the MPI_Fortran libraries to link with
set(XSDK_ENABLE_Fortran ON)
endif()
# Include xSDK default CMake file.
include("${CMAKE_CURRENT_SOURCE_DIR}/config/XSDKDefaults.cmake")
# Enable languages.
enable_language(CXX)
if (XSDK_ENABLE_C)
# - find_package(ATK REQUIRED)
enable_language(C)
endif()
if (XSDK_ENABLE_Fortran)
enable_language(Fortran)
endif()
# Suppress warnings about MACOSX_RPATH
set(CMAKE_MACOSX_RPATH OFF CACHE BOOL "")
# CMake needs to know where to find things
set(MFEM_CMAKE_PATH ${PROJECT_SOURCE_DIR}/config)
set(CMAKE_MODULE_PATH ${MFEM_CMAKE_PATH}/cmake/modules)
# Load MFEM CMake utilities.
include(MfemCmakeUtilities)
string(TOUPPER "${PROJECT_NAME}" PROJECT_NAME_UC)
mfem_version_to_int(${${PROJECT_NAME}_VERSION} ${PROJECT_NAME_UC}_VERSION)
set(${PROJECT_NAME_UC}_VERSION_STRING ${${PROJECT_NAME}_VERSION})
if (EXISTS ${PROJECT_SOURCE_DIR}/.git)
execute_process(
COMMAND git describe --all --long --abbrev=40 --dirty --always
WORKING_DIRECTORY "${PROJECT_SOURCE_DIR}"
OUTPUT_VARIABLE ${PROJECT_NAME_UC}_GIT_STRING
ERROR_QUIET OUTPUT_STRIP_TRAILING_WHITESPACE)
endif()
if (NOT ${PROJECT_NAME_UC}_GIT_STRING)
set(${PROJECT_NAME_UC}_GIT_STRING "(unknown)")
endif()
#-------------------------------------------------------------------------------
# Process configuration options
#-------------------------------------------------------------------------------
@@ -123,23 +70,25 @@ else()
set(MFEM_DEBUG OFF)
endif()
# MPI -> hypre; PETSc (optional)
# MPI -> hypre, METIS
if (MFEM_USE_MPI)
find_package(MPI REQUIRED)
set(MPI_CXX_INCLUDE_DIRS ${MPI_CXX_INCLUDE_PATH})
# Parallel MFEM depends on hypre
include_directories(${MPI_CXX_INCLUDE_PATH})
# Parallel MFEM depends on hypre and METIS
find_package(HYPRE REQUIRED)
set(MFEM_HYPRE_VERSION ${HYPRE_VERSION})
include_directories(${HYPRE_INCLUDE_DIRS})
find_package(METIS REQUIRED)
include_directories(${METIS_INCLUDE_DIRS})
if (MFEM_USE_PETSC)
find_package(PETSc REQUIRED)
message(STATUS "Found PETSc version ${PETSC_VERSION}")
if (PETSC_VERSION AND (PETSC_VERSION VERSION_LESS 3.8.0))
message(FATAL_ERROR "PETSc version >= 3.8.0 is required")
if (PETSC_VERSION AND (PETSC_VERSION VERSION_LESS 3.7.5.99))
message(FATAL_ERROR "PETSc version >= 3.7.5.99 is required")
endif()
set(PETSC_INCLUDE_DIRS ${PETSC_INCLUDES})
include_directories(${PETSC_INCLUDES})
endif()
else()
set(PKGS_NEED_MPI SUPERLU PETSC STRUMPACK)
set(PKGS_NEED_MPI SUPERLU PETSC)
foreach(PKG IN LISTS PKGS_NEED_MPI)
if (MFEM_USE_${PKG})
message(STATUS "Disabling package ${PKG} - requires MPI")
@@ -148,19 +97,17 @@ else()
endforeach()
endif()
if (MFEM_USE_METIS)
find_package(METIS REQUIRED)
endif()
# GZSTREAM -> zlib
if (MFEM_USE_GZSTREAM)
find_package(ZLIB REQUIRED)
include_directories(${ZLIB_INCLUDE_DIRS})
endif()
# Backtrace with libunwind
if (MFEM_USE_LIBUNWIND)
set(MFEMBacktrace_REQUIRED_PACKAGES "Libunwind" "LIBDL" "CXXABIDemangle")
find_package(MFEMBacktrace REQUIRED)
include_directories(${LIBUNWIND_INCLUDE_DIRS})
endif()
# BLAS, LAPACK
@@ -182,6 +129,7 @@ endif()
if (MFEM_USE_SUITESPARSE)
find_package(SuiteSparse REQUIRED
UMFPACK KLU AMD BTF CHOLMOD COLAMD CAMD CCOLAMD config)
include_directories(${SuiteSparse_INCLUDE_DIRS})
endif()
# SUNDIALS
@@ -192,66 +140,63 @@ if (MFEM_USE_SUNDIALS)
find_package(SUNDIALS REQUIRED
NVector_Serial NVector_Parallel NVector_ParHyp CVODE ARKODE KINSOL)
endif()
include_directories(${SUNDIALS_INCLUDE_DIRS})
endif()
# Mesquite
if (MFEM_USE_MESQUITE)
find_package(Mesquite REQUIRED)
include_directories(${MESQUITE_INCLUDE_DIRS})
endif()
# SuperLU_DIST can only be enabled in parallel
# SuperLU_DIST can only be enabled if parallel
if (MFEM_USE_SUPERLU)
if (MFEM_USE_MPI)
find_package(SuperLUDist REQUIRED)
include_directories(${SuperLUDist_INCLUDE_DIRS})
else()
message(FATAL_ERROR " *** SuperLU_DIST requires that MPI be enabled.")
endif()
endif()
# STRUMPACK can only be enabled in parallel
if (MFEM_USE_STRUMPACK)
if (MFEM_USE_MPI)
find_package(STRUMPACK REQUIRED)
else()
message(FATAL_ERROR " *** STRUMPACK requires that MPI be enabled.")
endif()
endif()
# Gecko
if (MFEM_USE_GECKO)
find_package(Gecko REQUIRED)
include_directories(${GECKO_INCLUDE_DIRS})
endif()
# GnuTLS
if (MFEM_USE_GNUTLS)
find_package(_GnuTLS REQUIRED)
include_directories(${GNUTLS_INCLUDE_DIRS})
endif()
# NetCDF
if (MFEM_USE_NETCDF)
find_package(NetCDF REQUIRED)
include_directories(${NETCDF_INCLUDE_DIRS})
endif()
# MPFR
if (MFEM_USE_MPFR)
find_package(MPFR REQUIRED)
endif()
if (MFEM_USE_CONDUIT)
find_package(Conduit REQUIRED conduit relay blueprint )
include_directories(${MPFR_INCLUDE_DIRS})
endif()
# Axom/Sidre
if (MFEM_USE_SIDRE)
find_package(Axom REQUIRED Sidre SLIC axom_utils)
if (NOT MFEM_USE_MPI)
find_package(ATK REQUIRED Sidre SLIC common)
else()
find_package(ATK REQUIRED Sidre SPIO SLIC common)
endif()
include_directories(${ATK_INCLUDE_DIRS})
endif()
# MFEM_TIMER_TYPE
if (NOT DEFINED MFEM_TIMER_TYPE)
if (APPLE)
# use std::clock from <ctime> for UserTime and
# use mach_absolute_time from <mach/mach_time.h> for RealTime
set(MFEM_TIMER_TYPE 4)
set(MFEM_TIMER_TYPE 0) # use std::clock from <ctime>
elseif (WIN32)
set(MFEM_TIMER_TYPE 3) # QueryPerformanceCounter from <windows.h>
else()
@@ -265,27 +210,17 @@ if (NOT DEFINED MFEM_TIMER_TYPE)
endif()
# List all possible libraries in order of dependencies.
# [METIS < SuiteSparse]:
# With newer versions of SuiteSparse which include METIS header using 64-bit
# integers, the METIS header (with 32-bit indices, as used by mfem) needs to
# be before SuiteSparse.
set(MFEM_TPLS MPI_CXX OPENMP BLAS LAPACK METIS HYPRE SuiteSparse SUNDIALS PETSC
MESQUITE SuperLUDist STRUMPACK AXOM CONDUIT GECKO GNUTLS NETCDF MPFR POSIXCLOCKS
set(MFEM_TPLS HYPRE OPENMP SUNDIALS MESQUITE SuiteSparse SuperLUDist
ParMETIS METIS LAPACK BLAS GECKO GNUTLS NETCDF PETSC MPFR ATK POSIXCLOCKS
MFEMBacktrace ZLIB)
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
set(TPL_LIBRARIES "")
set(TPL_INCLUDE_DIRS "")
foreach(TPL IN LISTS MFEM_TPLS)
if (${TPL}_FOUND)
message(STATUS "MFEM: using package ${TPL}")
list(APPEND TPL_LIBRARIES ${${TPL}_LIBRARIES})
list(APPEND TPL_INCLUDE_DIRS ${${TPL}_INCLUDE_DIRS})
endif()
endforeach(TPL)
list(REMOVE_DUPLICATES TPL_LIBRARIES)
list(REMOVE_DUPLICATES TPL_INCLUDE_DIRS)
# message(STATUS "TPL_INCLUDE_DIRS = ${TPL_INCLUDE_DIRS}")
include_directories(${TPL_INCLUDE_DIRS})
if (OPENMP_FOUND)
message(STATUS "MFEM: using package OpenMP")
@@ -293,8 +228,6 @@ if (OPENMP_FOUND)
endif()
message(STATUS "MFEM build type: CMAKE_BUILD_TYPE = ${CMAKE_BUILD_TYPE}")
message(STATUS "MFEM version: v${MFEM_VERSION_STRING}")
message(STATUS "MFEM git string: ${MFEM_GIT_STRING}")
# Windows specific
set(_USE_MATH_DEFINES ${WIN32})
@@ -304,6 +237,7 @@ set(_USE_MATH_DEFINES ${WIN32})
#-------------------------------------------------------------------------------
# Headers and sources
include(MfemCmakeUtilities)
set(SOURCES "")
set(HEADERS "")
set(MFEM_SOURCE_DIRS general linalg mesh fem)
@@ -315,30 +249,21 @@ set(MASTER_HEADERS
${PROJECT_SOURCE_DIR}/mfem.hpp
${PROJECT_SOURCE_DIR}/mfem-performance.hpp)
set(_lib_path "${CMAKE_INSTALL_PREFIX}/lib")
set(CMAKE_INSTALL_RPATH_USE_LINK_PATH ON CACHE BOOL "")
set(CMAKE_INSTALL_RPATH "${_lib_path}" CACHE PATH "")
set(CMAKE_INSTALL_NAME_DIR "${_lib_path}" CACHE PATH "")
# Declaring the library
add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
list(REMOVE_DUPLICATES TPL_LIBRARIES)
# message(STATUS " TPL_LIBRARIES = ${TPL_LIBRARIES}")
if (CMAKE_VERSION VERSION_GREATER 2.8.11)
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES})
else()
target_link_libraries(mfem ${TPL_LIBRARIES})
endif()
if (MINGW)
target_link_libraries(mfem ws2_32)
endif()
set_target_properties(mfem PROPERTIES VERSION "${mfem_VERSION}")
set_target_properties(mfem PROPERTIES SOVERSION "${mfem_VERSION}")
# If building out-of-source, define MFEM_BUILD_DIR to point to the build
# directory.
if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
target_compile_definitions(mfem PRIVATE
"MFEM_BUILD_DIR=${PROJECT_BINARY_DIR}")
"-DMFEM_BUILD_DIR=${PROJECT_BINARY_DIR}")
endif()
# Generate configuration file in the build directory: config/_config.hpp.
@@ -356,11 +281,6 @@ if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
"// Auto-generated file.
#define MFEM_BUILD_DIR ${PROJECT_BINARY_DIR}
#include \"${PROJECT_SOURCE_DIR}/${Header}\"
")
# This version will be installed in the top include directory:
file(WRITE "${PROJECT_BINARY_DIR}/InstallHeaders/${Header}"
"// Auto-generated file.
#include \"mfem/${Header}\"
")
endforeach()
endif()
@@ -412,12 +332,12 @@ endif()
# Add 'check' target - quick test
if (NOT MFEM_USE_MPI)
add_custom_target(check
${CMAKE_CTEST_COMMAND} -R '^ex1_ser' -C ${CMAKE_CFG_INTDIR}
${CMAKE_CTEST_COMMAND} -R ex1_ser -E performance -C ${CMAKE_CFG_INTDIR}
USES_TERMINAL)
add_dependencies(check ex1)
else()
add_custom_target(check
${CMAKE_CTEST_COMMAND} -R '^ex1p' -C ${CMAKE_CFG_INTDIR}
${CMAKE_CTEST_COMMAND} -R ex1p -E performance -C ${CMAKE_CFG_INTDIR}
USES_TERMINAL)
add_dependencies(check ex1p)
endif()
@@ -432,6 +352,7 @@ add_subdirectory(doc)
#-------------------------------------------------------------------------------
message(STATUS "CMAKE_INSTALL_PREFIX = ${CMAKE_INSTALL_PREFIX}")
string(TOUPPER "${PROJECT_NAME}" PROJECT_NAME_UC)
set(INSTALL_INCLUDE_DIR include
CACHE PATH "Relative path for installing header files.")
set(INSTALL_LIB_DIR lib
@@ -451,15 +372,11 @@ install(TARGETS ${PROJECT_NAME}
DESTINATION ${INSTALL_LIB_DIR})
# Install the master headers
foreach(Header mfem.hpp mfem-performance.hpp)
install(FILES ${PROJECT_BINARY_DIR}/InstallHeaders/${Header}
DESTINATION ${INSTALL_INCLUDE_DIR})
endforeach()
install(FILES ${MASTER_HEADERS} DESTINATION ${INSTALL_INCLUDE_DIR}/mfem)
install(FILES ${MASTER_HEADERS} DESTINATION ${INSTALL_INCLUDE_DIR})
# Install the headers; currently, the miniapps headers are excluded
install(DIRECTORY ${MFEM_SOURCE_DIRS}
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem
DESTINATION ${INSTALL_INCLUDE_DIR}
FILES_MATCHING PATTERN "*.hpp")
# Install ${HEADERS}
@@ -472,11 +389,11 @@ install(DIRECTORY ${MFEM_SOURCE_DIRS}
# Install the configuration header files
install(FILES ${PROJECT_BINARY_DIR}/config/_config.hpp
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem/config
DESTINATION ${INSTALL_INCLUDE_DIR}/config
RENAME config.hpp)
install(FILES ${PROJECT_SOURCE_DIR}/config/tconfig.hpp
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem/config)
DESTINATION ${INSTALL_INCLUDE_DIR}/config)
# Package the whole thing up nicely
include(CMakePackageConfigHelpers)
@@ -486,15 +403,11 @@ export(TARGETS ${PROJECT_NAME}
FILE "${PROJECT_BINARY_DIR}/MFEMTargets.cmake")
# Export the package for use from the build-tree (this registers the build-tree
# with the CMake user package registry.)
# TODO: How do we register the install-tree? Replacing the build-tree?
# with a global CMake-registry)
export(PACKAGE ${PROJECT_NAME})
# Extract the include directories required to use MFEM
get_target_property(MFEM_TPL_INCLUDE_DIRS mfem INCLUDE_DIRECTORIES)
if (NOT MFEM_TPL_INCLUDE_DIRS)
set(MFEM_TPL_INCLUDE_DIRS "")
endif()
# This is the build-tree version
set(INCLUDE_INSTALL_DIRS ${PROJECT_BINARY_DIR} ${MFEM_TPL_INCLUDE_DIRS})
-437
View File
@@ -1,437 +0,0 @@
# How to Contribute
The MFEM team welcomes contributions at all levels: bugfixes; code
improvements; simplifications; new mesh, discretization or solver
capabilities; improved documentation; new examples and miniapps;
HPC performance improvements; ...
Use a pull request (PR) toward the `mfem:master` branch to propose your
contribution. If you are planning significant code changes, or have any
questions, you can also open an [issue](https://github.com/mfem/mfem/issues)
before issuing a PR. We also welcome your [simulation
images](http://mfem.org/gallery/), which you can submit via a pull request in
[mfem/web](https://github.com/mfem/web).
See the [Quick Summary](#quick-summary) section for the main highlights of our
GitHub workflow. For more details, consult the following sections and refer
back to them before issuing pull requests:
- [GitHub Workflow](#github-workflow)
- [MFEM Organization](#mfem-organization)
- [New Feature Development](#new-feature-development)
- [Developer Guidelines](#developer-guidelines)
- [Pull Requests](#pull-requests)
- [Pull Request Checklist](#pull-request-checklist)
- [Master/Next Workflow](#masternext-workflow)
- [Releases](#releases)
- [Release Checklist](#release-checklist)
- [LLNL Workflow](#llnl-workflow)
- [Automated Testing](#automated-testing)
- [Contact Information](#contact-information)
Contributing to MFEM requires knowledge of Git and, likely, finite elements. If
you are new to Git, see the [GitHub learning
resources](https://help.github.com/articles/git-and-github-learning-resources/).
To learn more about the finite element method, see our [FEM page](http://mfem.org/fem).
*By submitting a pull request, you are affirming the [Developer's Certificate of
Origin](#developers-certificate-of-origin-11) at the end of this file.*
## Quick Summary
- We encourage you to [join the MFEM organization](#mfem-organization) and create
development branches off `mfem:master`.
- Please follow the [developer guidelines](#developer-guidelines), in particular
with regards to documentation and code styling.
- Pull requests should be issued toward `mfem:master`. Make sure
to check the items off the [Pull Request Checklist](#pull-request-checklist).
- After approval, MFEM developers merge the PR manually in the [mfem:next branch](#masternext-workflow).
- After a week of testing in `mfem:next`, the original PR is merged in `mfem:master`.
- We use [milestones](https://github.com/mfem/mfem/milestones) to coordinate the
work on different PRs toward a release.
- Don't hesitate to [contact us](#contact-information) if you have any questions.
## GitHub Workflow
The GitHub organization, https://github.com/mfem, is the main developer hub for
the MFEM project.
If you plan to make contributions or will like to stay up-to-date with changes
in the code, *we strongly encourage you to [join the MFEM organization](#mfem-organization)*.
This will simplify the workflow (by providing you additional permissions), and
will allow us to reach you directly with project announcements.
### MFEM Organization
- Before you can start, you need a GitHub account, here are a few suggestions:
+ Create the account at: github.com/join.
+ For easy identification, please add your name and maybe a picture of you at: https://github.com/settings/profile.
+ To receive notification, set a primary email at: https://github.com/settings/emails.
+ For password-less pull/push over SSH, add your SSH keys at: https://github.com/settings/keys.
- [Contact us](#contact-information) for an invitation to join the MFEM GitHub
organization.
- You should receive an invitation email, which you can directly accept.
Alternatively, *after logging into GitHub*, you can accept the invitation at
the top of https://github.com/mfem.
- Consider making your membership public by going to https://github.com/orgs/mfem/people
and clicking on the organization visibility dropbox next to your name.
- Project discussions and announcements will be posted at
https://github.com/orgs/mfem/teams/everyone.
- The MFEM source code is in the [mfem](https://github.com/mfem/mfem)
repository.
- The website and corresponding documentation are in the
[web](https://github.com/mfem/web) repository.
- The [PyMFEM](https://github.com/mfem/PyMFEM) repository contains a Python
wrapper for MFEM.
- The [data](https://github.com/mfem/data) repository contains additional
(large) datafiles for MFEM.
### New Feature Development
- A new feature should be important enough that at least one person, the
proposer, is willing to work on it and be its champion.
- The proposer creates a branch for the new feature (with suffix `-dev`), off
the `master` branch, or another existing feature branch, for example:
```
# Clone assuming you have setup your ssh keys on GitHub:
git clone git@github.com:mfem/mfem.git
# Alternatively, clone using the "https" protocol:
git clone https://github.com/mfem/mfem.git
# Create a new feature branch starting from "master":
git checkout master
git pull
git checkout -b feature-dev
# Work on "feature-dev", add local commits
# ...
# One time only) push the branch to github and setup your local
# branch to track the github branch (for "git pull"):
git push -u origin feature-dev
```
- **We prefer that you create the new feature branch inside the MFEM organization
as opposed to in a fork.** This allows everyone in the community to collaborate
in one central place.
- If you prefer to work in your fork, please [enable upstream edits](https://help.github.com/articles/allowing-changes-to-a-pull-request-branch-created-from-a-fork/).
- Never use the `next` branch to start a new feature branch!
- The typical feature branch name is `new-feature-dev`, e.g. `pumi-dev`. While
not frequent in MFEM, other suffixes are possible, e.g. `-fix`, `-doc`, etc.
### Developer Guidelines
- *Keep the code lean and as simple as possible*
- Well-designed simple code is frequently more general and powerful.
- Lean code base is easier to understand by new collaborators.
- New features should be added only if they are necessary or generally useful.
- Introduction of language constructions not currently used in MFEM should be
justified and generally avoided (so we can build on cutting-edge systems).
- We prefer basic C++ and the C++03 standard, to keep the code readable by
a large audience and to make sure it compiles anywhere.
- *Keep the code general and reasonably efficient*
- Main goal is fast prototyping for research.
- When in doubt, generality wins over efficiency.
- Respect the needs of different users (current and/or future).
- *Keep things separate and logically organized*
- General usage features go in MFEM (implemented in as much generality as
possible), non-general features go into external apps.
- Inside MFEM, compartmentalize between linalg, fem, mesh, GLVis, etc.
- Contributions that are project-specific or have external dependencies are
allowed (if they are of broader interest), but should be `#ifdef`-ed and not
change the code by default.
- Code specifics
- All significant new classes, methods and functions have Doxygen-style
documentation in source comments.
- Consistent code styling is enforced with `make style` in the top-level
directory. This requires [Artistic Style](http://astyle.sourceforge.net) (we
specifically use version 2.05.1). See also the file `config/mfem.astylerc`.
- Use `mfem::out` and `mfem::err` instead of `std::cout` and `std::cerr` in
internal library code. (You can use `std` in examples and miniapps.)
- When manually resolving conflicts during a merge, make sure to mention the
conflicted files in the commit message.
### Pull Requests
- When your branch is ready for other developers to review / comment on
the code, create a pull request towards `mfem:master`.
- Pull request typically have titles like:
`Description [new-feature-dev]`
for example:
`Parallel Unstructured Mesh Infrastructure (PUMI) integration [pumi-dev]`
Note the branch name suffix (in square brackets).
- Titles may contain a prefix in square brackets to emphasize the type of PR.
Common choices are: `[DON'T MERGE]`, `[WIP]` and `[DISCUSS]`, for example:
`[DISCUSS] Hybridized DG [hdg-dev]`
- Add a description, appropriate labels and assign yourself to the PR. The MFEM
team will add reviewers as appropriate.
- List outstanding TODO items in the description, see PR #222 for an example.
- Track the Travis CI and Appveyor [continuous integration](#automated-testing)
builds at the end of the PR. These should run clean, so address any errors as
soon as possible.
### Pull Request Checklist
Before a PR can be merged, it should satisfy the following:
- [ ] Code builds.
- [ ] Code passes `make style`.
- [ ] Update `CHANGELOG`:
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
- [ ] Update `INSTALL`:
- [ ] Has a new optional library been added? (*Make sure the external library is licensed under LGPL, not GPL!*)
- [ ] Does `make` or `cmake` have a new target?
- [ ] Did the requirements or the installation process change? *(rare)*.
- [ ] Update `.gitignore`:
- [ ] Check if `make distclean; git status` shows any files that are generated from the source but we don't want to track in the repository.
- [ ] Add new patterns (just for the new files above) and re-run the above test.
- [ ] New examples:
- [ ] All sample runs at the top of the example work.
- [ ] Update `examples/makefile`:
- [ ] Add the example code to the appropriate `SEQ_EXAMPLES` and `PAR_EXAMPLES` variables.
- [ ] Add any files generated by it to the `clean` target.
- [ ] Add the example binary and any files generated by it to the top-level `.gitignore` file.
- [ ] Update `examples/CMakeLists.txt`:
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
- [ ] List the new example in `doc/CodeDocumentation.dox`.
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
- [ ] In `examples.md`, list the example under the appropriate categories, add new categories if necessary.
- [ ] Add a short description of the example in the "Extensive Examples" section of `features.md`.
- [ ] New miniapps:
- [ ] All sample runs at the top of the miniapp work.
- [ ] Update top-level `makefile` and `makefile` in corresponding miniapp directory.
- [ ] Add the miniapp binary and any files generated by it to the top-level `.gitignore` file.
- [ ] Update CMake build system:
- [ ] Update the `CMakeLists.txt` file in the `miniapps` directory, if the new miniapp is in a new directory.
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
- [ ] Consider adding a new test for the new miniapp.
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
- [ ] New capability:
- [ ] All significant new classes, methods and functions have Doxygen-style documentation in source comments.
- [ ] Consider adding new sample runs in existing examples to highlight the new capability.
- [ ] Consider saving cool simulation pictures with the new capability in the Confluence gallery (LLNL only) or submitting them, via pull request, to the gallery section of the `mfem/web` repo.
- [ ] If this is a major new feature, consider mentioning in the short summary inside `README` *(rare)*.
- [ ] List major new classes in `doc/CodeDocumentation.dox` *(rare)*.
- [ ] Update this checklist, if the new pull request affects it.
- [ ] (LLNL only) Clone the `tests` repository and run the following tests, see `mfem/tests/README.md`:
- [ ] `compilers`
- [ ] `memcheck`
- [ ] `unit-test`
- [ ] `documentation`
- [ ] (LLNL only) After merging:
- [ ] Regenerate `README.html` files from companion documentation pull requests.
- [ ] Update the `baseline` and `compiler` tests, add new tests if necessary.
- [ ] Consider updating the script `mfem/tests/sample-runs` (`sample-runs-serial` and `sample-runs-parallel`).
### Master/Next Workflow
MFEM uses a `master`/`next`-branch workflow as described below:
- The `master` branch should always be of release quality and changes should not
be merged until they have been fully tested. This branch is protected, and
changes can only be made through pull requests.
- After approval, a pull request is merged manually (by MFEM developers) in the
`next` branch for testing and the `in-next` label is added to the PR.
This can be done as follows:
```
# Pull the latest version of the "feature-dev" branch
git checkout feature-dev
git pull
# Pull the latest version of the "next" branch
git checkout next
git pull
# Merge "feature-dev" into "next", resolving conflicts, if necessary.
# Use the "--no-ff" flag to create a new commit with merge message.
git merge --no-ff feature-dev
# Push the "next" branch to the server
git push
```
- After a week of testing in `next` (excluding bugfixes), both on GitHub, as
well as [internally](#tests-at-llnl) at LLNL, the original PR is merged into
`master` (provided there are no issues).
- After the merge, the feature branch is deleted (unless it is a long-term
project with periodic PRs).
- The `next` branch is used just for integrated testing of all PRs approved for
merging into `master` to verify that each works individually and that all of
them work as a group. This branch can be discarded at any time, though we
typically do that only at the end of a [release cycle](#releases).
### Releases
- Releases are just tags in the `master` branch, e.g. https://github.com/mfem/mfem/releases/tag/v3.3.2,
and have a version that ends in an even "patch" number, e.g. `v3.2.2` or
`v3.4` (by convention `v3.4` is the same as `v3.4.0`.) Between releases, the
version ends in an odd "patch" number, e.g. `v3.3.3`.
- We use [milestones](https://github.com/mfem/mfem/milestones) to coordinate the
work on different PRs toward a release, see for example the
[v3.3.2 release](https://github.com/mfem/mfem/milestone/1?closed=1).
- After a release is complete, the `next` branch is recreated, e.g. as follows
(replace `3.3.2` with current release):
- Rename the current `next` branch to `next-pre-v3.3.2`.
- Create a new `next` branch starting from the `v3.3.2` release.
- Local copies of `next` can then be updated with `git checkout -B next origin/next`.
### Release Checklist
- [ ] Update the MFEM version in the following files:
- [ ] `CHANGELOG`
- [ ] `makefile`
- [ ] `CMakeLists.txt`
- [ ] `doc/CodeDocumentation.conf`
- [ ] (LLNL only) Make sure all `README.html` files in the source repo are up to date.
- [ ] Tag the repository:
```
git tag -a v3.1 -m "Official release v3.1"
git push origin v3.1
```
- [ ] Create the release tarball and push to `mfem/releases`.
- [ ] Recreate the `next` branch as described in previous section.
- [ ] Update and push documentation to `mfem/doxygen`.
- [ ] Update URL shorlinks:
- [ ] Create a shortlink at [https://goo.gl/](https://goo.gl/) for the release tarball, e.g. http://mfem.github.io/releases/mfem-3.1.tgz.
- [ ] (LLNL only) Add and commit the new shorlink in the `links` and `links-mfem` files of the internal `mfem/downloads` repo.
- [ ] Add the new shortlinks to the MFEM packages in `spack`, `homebrew/science`, `VisIt`, etc.
- [ ] Update website in `mfem/web` repo:
- Update version and shortlinks in `src/index.md` and `src/download.md`.
- Use [cloc-1.62.pl](http://cloc.sourceforge.net/) and `ls -lh` to estimate the SLOC and the tarball size in `src/download.md`.
## LLNL Workflow
- The GitHub `master` and `next` branches are mirrored to the LLNL institutional
Bitbucket repository as `gh-master` and `gh-next`.
- `gh-master` is merged into LLNL's internal `master` through pull requests; write
permissions to `master` are restricted to ensure this is the only way in which it
gets updated.
- We never push directly from LLNL to GitHub.
- Versions of the code on LLNL's internal server, from most to least stable:
- MFEM official release on mfem.org -- Most stable, tested in many apps.
- `mfem:master` -- Recent development version, guaranteed to work.
- `mfem:gh-master` -- Stable development version, passed testing, you can use
it to build your code between releases.
- `mfem:gh-next` -- Bleeding-edge development version, may be broken, use at
your own risk.
## Automated Testing
MFEM has several levels of automated testing running on GitHub, as well as on
local Mac and Linux workstations, and Livermore Computing clusters at LLNL.
### Linux and Mac smoke tests
We use Travis CI to drive the default tests on the `master` and `next`
branches. See the `.travis` file and the logs at
[https://travis-ci.org/mfem/mfem](https://travis-ci.org/mfem/mfem).
Testing using Travis CI should be kept lightweight, as there is a 50 minute time
constraint on jobs. Two virtual machines are configured - Mac (OS X) and Linux.
- Tests on the `master` branch are triggered whenever a PR is issued on this branch.
- Tests on the `next` branch are currently scheduled to run each night.
### Windows smoke test
We use Appveyor to test building with the MS Visual C++ compiler in a Windows
environment, as well as to test the CMake build. See the `.appveyor` file and the
build logs at
[https://ci.appveyor.com/project/mfem/mfem](https://ci.appveyor.com/project/mfem/mfem).
CMake is used to generate the MSVC Project files and drive the build. A release
and debug build is performed with a simple run of `ex1` to verify the executable.
### Tests at LLNL
At LLNL, we mirror the `master` and `next` branches internally (to `gh-master`
and `gh-next`) and run longer nightly tests via cron. On the weekends, a more
extensive test is run which extracts and executes all the different sample runs
from each example.
## Contact Information
- Contact the MFEM team by posting to the [GitHub issue tracker](https://github.com/mfem/mfem).
Please perform a search to make sure your question has not been answered already.
- Email communications should be sent to the MFEM developers mailing list,
mfem-dev@llnl.gov.
## [Developer's Certificate of Origin 1.1](https://developercertificate.org/)
By making a contribution to this project, I certify that:
(a) The contribution was created in whole or in part by me and I have the right
to submit it under the open source license indicated in the file; or
(b) The contribution is based upon previous work that, to the best of my
knowledge, is covered under an appropriate open source license and I have
the right under that license to submit that work with modifications, whether
created in whole or in part by me, under the same open source license
(unless I am permitted to submit under a different license), as indicated in
the file; or
(c) The contribution was provided directly to me by some other person who
certified (a), (b) or (c) and I have not modified it.
(d) I understand and agree that this project and the contribution are public and
that a record of the contribution (including all personal information I
submit with it, including my sign-off) is maintained indefinitely and may be
redistributed consistent with this project or the open source license(s)
involved.
+17 -112
View File
@@ -18,19 +18,13 @@ requires an MPI C++ compiler, as well as the following external libraries:
- METIS (a family of multilevel partitioning algorithms)
http://glaros.dtc.umn.edu/gkhome/metis/metis/overview
The METIS dependency can be disabled but that is not generally recommended, see
the option MFEM_USE_METIS.
The library supports two build systems: one based on GNU make, and a second one
based on CMake. Both build systems are described below. Some hints for building
without GNU make or CMake can be found at the end of this file.
In addition to the native build systems, MFEM packages are also available in the
following package managers:
- Spack, https://github.com/spack/spack
- OpenHPC, http://openhpc.community
- Homebrew/Science, https://github.com/Homebrew/homebrew-science
In addition to the native build systems, MFEM packages are also available in
the Homebrew/Science, https://github.com/Homebrew/homebrew-science, and the
Spack, https://github.com/LLNL/spack, package managers.
Quick start with GNU make
@@ -144,7 +138,7 @@ check the results from all the serial/parallel MFEM examples and miniapps use:
Note that by default MFEM uses "mpirun -np" in its test runs (this is also what
is used in the sample runs of its examples and miniapps). The MPI launcher can
be changed by the user as described in the "Specifying an MPI job launcher"
be changed by the user as described in the "Specifying a MPI job launcher"
section at the end of this file.
Running all the tests may take a while. Implementation details about the check
@@ -156,8 +150,8 @@ An optional installation of the library and the headers can be performed with
make install [PREFIX=<dir>]
The library will be installed in $(PREFIX)/lib, the headers in
$(PREFIX)/include, and the configuration makefile (config.mk) in
$(PREFIX)/share/mfem. The PREFIX option can also be set during configuration.
$(PREFIX)/include, and the configuration makefile (config.mk) in $(PREFIX).
The PREFIX option can also be set during configuration.
Information about the current build configuration can be viewed using
@@ -187,8 +181,6 @@ examples/ directory.
Configuration options (GNU make)
================================
See the configuration file config/defaults.mk for the default settings.
Compilers:
CXX - C++ compiler, serial build
MPICXX - MPI C++ compiler, parallel build
@@ -199,14 +191,10 @@ Compiler options:
CXXFLAGS - If not set, defined based on the above optimized/debug flags
CPPFLAGS - Additional compiler options
Build options:
STATIC - Build a static version of the library (YES/NO), default = YES
SHARED - Build a shared version of the library (YES/NO), default = NO
Installation options:
PREFIX - Specify the installation directory. The library (libmfem.a) will be
installed in $(PREFIX)/lib, the headers in $(PREFIX)/include, and
the configuration makefile (config.mk) in $(PREFIX)/share/mfem.
the configuration makefile (config.mk) in $(PREFIX).
INSTALL - Specify the install program, e.g /usr/bin/install
MFEM library features/options (GNU make)
@@ -215,21 +203,10 @@ MFEM_USE_MPI = YES/NO
Choose parallel/serial build. The parallel build requires proper setup of the
HYPRE_* and METIS_* library options, see below.
MFEM_USE_METIS = YES/NO
Enable/disable the use of the METIS library. By default, this option is set
to the value of MFEM_USE_MPI. If this option is explicitly disabled in a
parallel build, then the only parallel partitioning (domain decomposition)
option in the library will be Cartesian partitioning with box meshes, and
thus most of the parallel examples and miniapps will fail.
MFEM_DEBUG = YES/NO
Choose debug/optimized build. The debug build enables a number of messages
and consistency checks that may simplify bug-hunting.
MFEM_USE_EXCEPTIONS = YES/NO
Enable the use of exceptions. In particular, modifies the default bahavior
when errors are encountered: throw an exception, instead of aborting.
MFEM_USE_LIBUNWIND = YES/NO
Use libunwind to print a stacktrace whenever mfem_error is raised. The
information printed is enough to determine the line numbers where the
@@ -254,16 +231,13 @@ MFEM_USE_MEMALLOC = YES/NO
Internal MFEM option: enable batch allocation for some small objects.
Recommended value is YES.
MFEM_TIMER_TYPE = 0/1/2/3/4/5/6/NO
MFEM_TIMER_TYPE = 0/1/2/3/NO
Specify which library functions to use in the class StopWatch used for
measuring time. The available options are:
0 - use std::clock from <ctime>, standard C++
1 - use times from <sys/times.h>
2 - use high-resolution POSIX clocks (see option POSIX_CLOCKS_LIB)
3 - use QueryPerformanceCounter from <windows.h>
4 - use mach_absolute_time from <mach/mach_time.h> + std::clock (Mac)
5 - use gettimeofday from <sys/time.h>
6 - use MPI_Wtime from <mpi.h>
NO - use option 3 if the compiler macro _WIN32 is defined, 0 otherwise
MFEM_USE_SUNDIALS = YES/NO
@@ -287,12 +261,6 @@ MFEM_USE_SUPERLU = YES/NO
SuperLURowLocMatrix a distributed CSR matrix class needed by SuperLU. When
enabled, this option uses the SUPERLU_* library options, see below.
MFEM_USE_STRUMPACK = YES/NO
Enable MFEM functionality based on the STRUMPACK sparse direct solver and
preconditioner through the STRUMPACKSolver and STRUMPACKRowLocMatrix
classes. When enabled, this option uses the STRUMPACK_* library options, see
below.
MFEM_USE_GNUTLS = YES/NO
Enable secure socket support in class socketstream, using the auxiliary
GnuTLS_* classes, based on the GnuTLS library. This option may be useful in
@@ -304,9 +272,6 @@ MFEM_USE_GNUTLS = YES/NO
the script 'glvis-keygen.sh' in the main GLVis directory can be used to do
that:
bash glvis-keygen.sh ["Your Name"] ["Your Email"]
In MFEM v3.3.2 and earlier, the secure authentication is based on OpenPGP
keys, while later versions use X.509 certificates. The latest version of the
script 'glvis-keygen.sh' can be used to generate both types of keys.
When MFEM_USE_GNUTLS is enabled, the additional build options, GNUTLS_*, are
also used, see below.
@@ -333,13 +298,6 @@ MFEM_USE_SIDRE = YES/NO
specification. When enabled, this option requires installation of HDF5 (see
also MFEM_USE_NETCDF), Conduit and LLNL's axom project.
MFEM_USE_CONDUIT = YES/NO
Enables support for converting MFEM Mesh and Grid Function objects to and
from Conduit Mesh Blueprint Descriptions (https://github.com/LLNL/conduit/)
and support for JSON and Binary I/O via Conduit Relay. This option requires
an installation of Conduit. If Conduit was built with HDF5 support, it also
requires an installation of HDF5 (see also MFEM_USE_NETCDF).
MFEM_USE_GZSTREAM = YES/NO
Enables use of on-the-fly gzip compressed streams. With this feature enabled
(YES), MFEM can compress its output files on-the-fly. In addition, it can
@@ -350,7 +308,6 @@ MFEM_USE_GZSTREAM = YES/NO
able to properly read an input file if it is gzip compressed. In that case,
the solution is to uncompress the file with an external tool (such as gunzip)
before attempting to use it with MFEM.
When enabled, this option uses the ZLIB_* library options, see below.
MFEM_BUILD_TAG = (any value)
An optional tag to characterize the build. Exported to config/config.mk.
@@ -376,8 +333,8 @@ The specific libraries and their options are:
URL: http://www.llnl.gov/CASC/hypre
Options: HYPRE_OPT, HYPRE_LIB.
- METIS, used when MFEM_USE_METIS = YES. If using METIS 5, set
MFEM_USE_METIS_5 = YES (default is to use METIS 4).
- METIS, required for the parallel build, i.e. when MFEM_USE_MPI = YES. If using
METIS 5, set MFEM_USE_METIS_5 = YES (default is to use METIS 4).
URL: http://glaros.dtc.umn.edu/gkhome/metis/metis/overview
Options: METIS_OPT, METIS_LIB.
@@ -395,10 +352,7 @@ The specific libraries and their options are:
Option: POSIX_CLOCKS_LIB (default = -lrt).
- SUNDIALS (optional), used when MFEM_USE_SUNDIALS = YES.
Beginning with MFEM v3.3, SUNDIALS v2.7.0 is supported.
Beginning with MFEM v3.3.2, SUNDIALS v3.0.0 is also supported.
If MFEM_USE_MPI is enabled, we expect that SUNDIALS is built with support for
both MPI and hypre.
In parallel we expect that SUNDIALS is built with support for MPI and hypre.
URL: http://computation.llnl.gov/projects/sundials/sundials-software
Options: SUNDIALS_OPT, SUNDIALS_LIB.
@@ -417,14 +371,6 @@ The specific libraries and their options are:
URL: http://crd-legacy.lbl.gov/~xiaoye/SuperLU
Options: SUPERLU_OPT, SUPERLU_LIB.
- STRUMPACK (optional), used when MFEM_USE_STRUMPACK = YES. Note that STRUMPACK
requires the PT-Scotch and Scalapack libraries as well as ParMETIS, which
includes METIS 5 in its distribution.
The support for STRUMPACK was added in MFEM v3.3.2 and it requires STRUMPACK
2.0.0 or later.
URL: http://portal.nersc.gov/project/sparse/strumpack
Options: STRUMPACK_OPT, STRUMPACK_LIB.
- GnuTLS (optional), used when MFEM_USE_GNUTLS = YES. On most Linux systems,
GnuTLS is available as a development package, e.g. gnutls-devel. On Mac OS X,
one can get the library through the Homebrew package manager (http://brew.sh).
@@ -438,7 +384,7 @@ The specific libraries and their options are:
URL: www.unidata.ucar.edu/software/netcdf
Options: NETCDF_OPT, NETCDF_LIB.
- PETSc (optional), used when MFEM_USE_PETSC = YES. Version 3.8 or higher of
- PETSc (optional), used when MFEM_USE_PETSC = YES. Version 3.8 or higher of
the PETSC dev branch is required. The MFEM and PETSc builds can share common
libraries, e.g., hypre and SUNDIALS. Here's an example configuration, assuming
PETSc has been cloned on the same level as mfem and hypre:
@@ -455,12 +401,6 @@ The specific libraries and their options are:
https://support.hdfgroup.org/HDF5 (HDF5)
Options: SIDRE_OPT, SIDRE_LIB.
- Conduit, used when MFEM_USE_CONDUIT = YES. Direct Conduit Mesh Blueprint
support requires Conduit >= v0.3.1 and VisIt >= v2.13.1 to read the output.
URL: https://github.com/LLNL/conduit (Conduit)
https://support.hdfgroup.org/HDF5 (HDF5)
Options: CONDUIT_OPT, CONDUIT_LIB.
- MPFR (optional), used when MFEM_USE_MPFR = YES.
URL: http://mpfr.org, it depends on the GMP library: https://gmplib.org
Options: MPFR_OPT, MPFR_LIB.
@@ -471,11 +411,6 @@ The specific libraries and their options are:
URL: http://www.nongnu.org/libunwind
Options: LIBUNWIND_OPT, LIBUNWIND_LIB.
- ZLIB (optional), used when MFEM_USE_GZSTREAM = YES, or when MFEM_USE_NETCDF =
YES (in the default settings for NETCDF_OPT and NETCDF_LIB).
URL: https://zlib.net
Options: ZLIB_OPT, ZLIB_LIB.
Building with CMake
===================
@@ -507,12 +442,6 @@ Debug and optimization options are controlled through the CMake variable
CMAKE_BUILD_TYPE which can be set to standard values like "Debug", and "Release"
(default).
To use a specific generator use the "-G <generator>" option of cmake:
cmake <mfem-source-dir> -G "Xcode"
cmake <mfem-source-dir> -G "Visual Studio 12 2013"
cmake <mfem-source-dir> -G "MinGW Makefiles"
With CMake it is possible to build MFEM as a shared library using the standard
CMake option -DBUILD_SHARED_LIBS=1.
@@ -520,30 +449,15 @@ Once configured, the library can be built simply with (assuming a UNIX type
system, where the default is to generate "UNIX Makefiles")
make -j 4
or
cmake --build .
or
cmake --build . --config Release [Visual Studio, Xcode]
The build can be quick-tested by running
make check
or
cmake --build . --target check
or
cmake --build . --config Release --target check [Visual Studio, Xcode]
which will simply compile and run Example 1/1p. For more extensive tests that
check the results from all the serial/parallel MFEM examples and miniapps use:
make exec -j 4
make test
or
cmake --build . --target exec
cmake --build . --target test
or
cmake --build . --config Release --target exec [Visual Studio, Xcode]
cmake --build . --config Release --target RUN_TESTS [Visual Studio, Xcode]
Note that running all the tests may take a while.
@@ -551,11 +465,6 @@ Installation prefix can be configured by setting the standard CMake variable
CMAKE_INSTALL_PREFIX. To install the library, use
make install
or
cmake --build . --target install
or
cmake --build . --config Release --target install [Xcode]
cmake --build . --config Release --target INSTALL [Visual Studio]
The library will be installed in <PREFIX>/lib, the headers in <PREFIX>/include,
and the configuration CMake files in <PREFIX>/lib/cmake/mfem.
@@ -563,8 +472,6 @@ and the configuration CMake files in <PREFIX>/lib/cmake/mfem.
Configuration variables (CMake)
===============================
See the configuration file config/defaults.cmake for the default settings.
Non-standard CMake variables for compilers:
CXX - If set, overwrite the auto-detected C++ compiler, serial build
MPICXX - If set, overwrite the auto-detected MPI C++ compiler, parallel build
@@ -578,7 +485,6 @@ The following options are equivalent to the GNU make options with the same name:
[see "MFEM library features/options (GNU make)" above]
MFEM_USE_MPI
MFEM_USE_METIS - Set to ${MFEM_USE_MPI}, can be overwritten.
MFEM_USE_LIBUNWIND
MFEM_USE_LAPACK
MFEM_THREAD_SAFE
@@ -588,7 +494,6 @@ MFEM_TIMER_TYPE - Set automatically, can be overwritten.
MFEM_USE_MESQUITE
MFEM_USE_SUITESPARSE
MFEM_USE_SUPERLU
MFEM_USE_STRUMPACK
MFEM_USE_GNUTLS
MFEM_USE_NETCDF
MFEM_USE_MPFR
@@ -631,7 +536,7 @@ The CMake build system adds auto-detection for the following packages/libraries:
- METIS - The option MFEM_USE_METIS_5 is auto-detected.
- MESQUITE
- SuiteSparse
- SuperLUDist, STRUMPACK
- SuperLUDist
- ParMETIS
- GNUTLS - Extends the built-in CMake support, to search GNUTLS_DIR as well.
- NETCDF
@@ -653,15 +558,15 @@ Before using another build system (e.g. Visual Studio) it is necessary to create
a proper configuration header file, config/config.hpp, using the template from
config/config.hpp.in:
cp config/config.hpp.in config/_config.hpp
cp config/config.hpp.in config/config.hpp
The file config/_config.hpp can then be edited to enable desired options. The
The file config/config.hpp can then be edited to enable desired options. The
MFEM library is simply a combination of all object files obtained by compiling
the .cpp source files in the source directories: general, linalg, mesh, and fem.
Specifying an MPI job launcher
==============================
Specifying a MPI job launcher
=============================
By default, MFEM will use 'mpirun -np #' to launch any of its parallel tests or
miniapps, where # is the number of MPI tasks. An alternate MPI launcher can be
provided by setting the MFEM_MPIEXEC and MFEM_MPIEXEC_NP config variables.
-1
View File
@@ -60,4 +60,3 @@ This project is released under the LGPL v2.1 license. See LICENSE file for full
details.
LLNL Release Number: LLNL-CODE-443211
DOI: 10.11578/dc.20171025.1248
-31
View File
@@ -1,31 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_ALL_HPP
#define MFEM_BACKENDS_ALL_HPP
#include "../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "base/backend.hpp"
#ifdef MFEM_USE_OCCA
#include "occa/backend.hpp"
#endif
#ifdef MFEM_USE_OMP
#include "omp/backend.hpp"
#endif
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_ALL_HPP
-215
View File
@@ -1,215 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_BASE_ARRAY_HPP
#define MFEM_BACKENDS_BASE_ARRAY_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "layout.hpp"
#include "utils.hpp"
namespace mfem
{
/// Extension to the template class Array<T>
class PArray : public RefCounted
{
protected:
/// Layout with shared ownership (smart pointer)
DLayout layout;
/**
@name Virtual interface
*/
///@{
virtual void *DoGetData() const = 0;
/** @brief Create and return a new array (in @a *clone) of the same dynamic
type as this array using the same layout and ItemSize().
Set @a *clone to NULL if allocation fails.
If @a copy_data is true, the contents of this array is copied to the new
array; otherwise, the new array remains uninitialized.
If @a buffer is not NULL, return the array data of the newly created
object (in @a *buffer) , if it is stored as a contiguous array on the
host; otherwise, set @a *buffer to NULL. */
virtual PArray *DoClone(bool copy_data, void **buffer,
std::size_t item_size) const = 0;
/// Resize the array, reallocating its data if necessary.
/** If @a buffer is not NULL, return the array data (in @a *buffer), if it
is stored as a contiguous array on the host; otherwise, set @a *buffer to
NULL. Returns 0 on success and non-zero otherwise, e.g. if memory
allocation fails.
If the @a new_layout is not supported, a non-zero error code will be
returned.
The @a new_layout has to be valid, i.e. new_layout != NULL and
new_layout->HasEngine() == true.
@note If reallocation is performed, the previous content of the array is
NOT copied to the new location. */
virtual int DoResize(PLayout &new_layout, void **buffer,
std::size_t item_size) = 0;
/** @brief Get access to the contents of the array in host memory, as a
contiguous array. */
/** If the array data is stored as a contiguous array in host memory, return
a pointer to it. Otherwise, copy the data to @a buffer (if @a buffer is
not NULL) and return @a buffer.
@note If not NULL, @a buffer is assumed to be of size greater than or
equal to Size(). */
virtual void *DoPullData(void *buffer, std::size_t item_size) = 0;
/** @brief Set all entries of the array to the (single) value pointed to by
@a value_ptr. */
virtual void DoFill(const void *value_ptr, std::size_t item_size) = 0;
/** @brief Set all Size() entries of the array from the given contiguous
array, @a src_buffer, on the host. */
virtual void DoPushData(const void *src_buffer, std::size_t item_size) = 0;
/// Copy the data from @a src to @a *this.
/** Both arrays must have the same dynamic type, layout, and item_size. */
virtual void DoAssign(const PArray &src, std::size_t item_size) = 0;
///@}
// End: Virtual interface
public:
/** @brief The @a layout parameter will be reference counted and therefore it
should be dynamically allocated. */
/** The @a layout must be valid in the sense that layout != NULL and
layout->HasEngine() == true. */
PArray(PLayout &p_layout)
: layout(&p_layout)
{
MFEM_ASSERT(layout && layout->HasEngine(), "invalid layout");
}
virtual ~PArray() { }
/// Get the current size of the array.
std::size_t Size() const { return layout->Size(); }
/// Get the current layout of the array.
PLayout &GetLayout() const { return *layout; }
/// TODO
template <typename derived_t>
derived_t &As() { return dynamic_cast<derived_t&>(*this); }
/// TODO
template <typename derived_t>
const derived_t &As() const { return dynamic_cast<const derived_t&>(*this); }
// TODO: Error handling ... handle errors at the Engine level, at the class
// level, or at the method level?
// TODO: Asynchronous execution interface ...
/**
@name Public virtual interface
*/
///@{
template <typename T=void>
T* GetData() const { return (T*) DoGetData(); }
/** @brief Create and return a new array (in @a *clone) of the same dynamic
type as this array using the same layout and ItemSize().
Set @a *clone to NULL if allocation fails.
If @a copy_data is true, the contents of this array is copied to the new
array; otherwise, the new array remains uninitialized.
If @a buffer is not NULL, return the array data of the newly created
object (in @a *buffer) , if it is stored as a contiguous array on the
host; otherwise, set @a *buffer to NULL. */
template <typename T>
DArray Clone(bool copy_data, T **buffer) const
{ return DArray(DoClone(copy_data, (void**)buffer, sizeof(T))); }
/// Resize the array, reallocating its data if necessary.
/** If @a buffer is not NULL, return the array data (in @a *buffer), if it
is stored as a contiguous array on the host; otherwise, set @a *buffer to
NULL. Returns 0 on success and non-zero otherwise, e.g. if memory
allocation fails.
If the @a new_layout is not supported, a non-zero error code will be
returned.
The @a new_layout has to be valid, i.e. new_layout != NULL and
new_layout->HasEngine() == true.
@note If reallocation is performed, the previous content of the array is
NOT copied to the new location. */
template <typename T>
int Resize(PLayout &new_layout, T **buffer)
{ return DoResize(new_layout, (void**)buffer, sizeof(T)); }
/// Shortcut for Resize(*layout, buffer).
/** This method is useful for updating the array after its layout is changed
externally. */
template <typename T>
int Update(T **buffer)
{ return DoResize(*layout, (void**)buffer, sizeof(T)); }
/// Shortcut for layout->Resize(new_size) followed by Update()
template <typename T>
int Resize(std::size_t new_size, T **buffer)
{ layout->Resize(new_size); return Update(buffer); }
/** @brief Get access to the contents of the array in host memory, as a
contiguous array. */
/** If the array data is stored as a contiguous array in host memory, return
a pointer to it. Otherwise, copy the data to @a buffer (if @a buffer is
not NULL) and return @a buffer.
@note If not NULL, @a buffer is assumed to be of size greater than or
equal to Size(). */
template <typename T>
T *PullData(T *buffer)
{ return Size() ? (T*)DoPullData((void*)buffer, sizeof(T)) : NULL; }
/** @brief Set all entries of the array to the (single) value pointed to by
@a value_ptr. */
template <typename T>
void Fill(const T &value) { if (Size()) { DoFill(&value, sizeof(T)); } }
/** @brief Set all Size() entries of the array from the given contiguous
array, @a src_buffer, on the host. */
template <typename T>
void PushData(const T *src_buffer)
{ if (Size()) { DoPushData(src_buffer, sizeof(T)); } }
/// Copy the data from @a src to @a *this.
/** Both arrays must have the same dynamic type, layout, and entry type. */
template <typename T>
void Assign(const PArray &src) { if (Size()) { DoAssign(src, sizeof(T)); } }
///@}
// End: Virtual interface
};
} // namespace mfem
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_BASE_ARRAY_HPP
-57
View File
@@ -1,57 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_BASE_BACKEND_HPP
#define MFEM_BACKENDS_BASE_BACKEND_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "memory_resource.hpp"
#include "engine.hpp"
#include "array.hpp"
#include "vector.hpp"
#include "fespace.hpp"
#include "bilinearform.hpp"
#include <string>
#ifdef MFEM_USE_MPI
#include <mpi.h>
#endif
namespace mfem
{
/// TODO
class Backend
{
public:
/// TODO
virtual ~Backend() { }
/// TODO
virtual bool Supports(const std::string &engine_spec) const = 0;
/// TODO
virtual Engine *Create(const std::string &engine_spec) = 0;
#ifdef MFEM_USE_MPI
/// TODO
virtual Engine *Create(MPI_Comm comm, const std::string &engine_spec) = 0;
#endif
};
} // namespace mfem
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_BASE_BACKEND_HPP
-72
View File
@@ -1,72 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_BASE_BILINEARFORM_HPP
#define MFEM_BACKENDS_BASE_BILINEARFORM_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "engine.hpp"
namespace mfem
{
class Vector;
class OperatorHandle;
class BilinearForm;
/// TODO: doxygen
class PBilinearForm : public RefCounted
{
protected:
/// Engine with shared ownership
SharedPtr<const Engine> engine;
/// Not owned.
BilinearForm *bform;
public:
/// TODO: doxygen
PBilinearForm(const Engine &e, BilinearForm &bf)
: engine(&e), bform(&bf) { }
/// Virtual destructor
virtual ~PBilinearForm() { }
/// Get the associated Engine
const Engine &GetEngine() const { return *engine; }
/// Assemble the PBilinearForm.
/** This method is called from the method BilinearForm::Assemble() of the
associated BilinearForm #bform.
@returns True, if the host assembly should be skipped. */
virtual bool Assemble() = 0;
/// TODO: doxygen
virtual void FormSystemMatrix(const Array<int> &ess_tdof_list,
OperatorHandle &A) = 0;
/// TODO: doxygen
virtual void FormLinearSystem(const Array<int> &ess_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B,
int copy_interior) = 0;
/// TODO: doxygen
virtual void RecoverFEMSolution(const Vector &X, const Vector &b,
Vector &x) = 0;
};
} // namespace mfem
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_BASE_BILINEARFORM_HPP
-29
View File
@@ -1,29 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "engine.hpp"
#include "fespace.hpp"
#include "bilinearform.hpp"
namespace mfem
{
DFiniteElementSpace Engine::MakeFESpace(FiniteElementSpace &fes) const
{
return DFiniteElementSpace(new PFiniteElementSpace(*this, fes));
}
} // namespace mfem
#endif // MFEM_USE_BACKENDS
-210
View File
@@ -1,210 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_BASE_ENGINE_HPP
#define MFEM_BACKENDS_BASE_ENGINE_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "../../general/scalars.hpp"
#include "memory_resource.hpp"
#include "smart_pointers.hpp"
#include "utils.hpp"
#ifdef MFEM_USE_MPI
#include <mpi.h>
#endif
namespace mfem
{
// Forward declarations.
class Backend;
template <typename T> class Array;
class Vector;
class Operator;
class FiniteElementSpace;
class LinearForm;
class BilinearForm;
class MixedBilinearForm;
class NonlinearForm;
/// In parallel, each MPI rank will usually create a single engine.
class Engine : public RefCounted
{
protected:
Backend *backend; ///< Backend that created the engine. Not owned.
#ifdef MFEM_USE_MPI
MPI_Comm comm; ///< Associated MPI communicator (may be MPI_COMM_NULL).
#endif
/// Number of memory resources used by the Engine.
int num_mem_res;
/// Number of workers used by the Engine.
int num_workers;
/// Memory resources used by the engine - array of pointers.
/** Both the array and the entries are owned. */
MemoryResource **memory_resources;
/// Relative computational speed of the workers. Owned.
double *workers_weights;
/// For each worker, which memory resource it uses.
int *workers_mem_res;
public:
/// TODO: doxygen
Engine(Backend *b, int n_mem, int n_workers)
: backend(b),
#ifdef MFEM_USE_MPI
comm(MPI_COMM_NULL),
#endif
num_mem_res(n_mem),
num_workers(n_workers),
memory_resources(new MemoryResource*[num_mem_res]()),
workers_weights(new double[num_workers]()),
workers_mem_res(new int[num_workers]())
{ /* Note: all arrays are value-initialized with zeros. */ }
/// TODO: doxygen
virtual ~Engine()
{
delete [] workers_mem_res;
delete [] workers_weights;
for (int i = 0; i < num_mem_res; i++)
{
delete memory_resources[i];
}
delete [] memory_resources;
}
/**
@name Machine resources interface
*/
///@{
#ifdef MFEM_USE_MPI
/// Get the associated MPI_Comm
MPI_Comm GetComm() const { return comm; }
#endif
/// TODO
int GetNumMemRes() const { return num_mem_res; }
/// TODO
MemoryResource &GetMemRes(int idx) const { return *memory_resources[idx]; }
/// TODO
int GetNumWorkers() const { return num_workers; }
/// TODO
const double *GetWorkersWeights() const { return workers_weights; }
/// TODO
const int *GetWorkersMemRes() const { return workers_mem_res; }
///@}
// End: Machine resources interface
/// TODO
template <typename derived_t>
derived_t &As() { *util::As<derived_t>(this); }
/// TODO
template <typename derived_t>
const derived_t &As() const { *util::As<const derived_t>(this); }
// TODO: Error handling ... handle errors at the Engine level, at the class
// level, or at the method level?
/**
@name Virtual interface: finite element data structures and algorithms
*/
///@{
// TODO: Asynchronous execution in this class ...
/// Allocate and return a new layout for the given @a size.
/** The layout decomposition is determined automatically by the Engine using
a deterministic algorithm: calls to this method with the same @a size
will produce the same result, as long as the Engine remains unmodified
between the calls.
The returned object is allocated with operator new and must be
deallocated by the caller.
TODO: Returns NULL if memory allocation fails?
*/
virtual DLayout MakeLayout(std::size_t size) const = 0;
/// Allocate and return a new layout for the given worker decomposition.
/** The returned object is allocated with operator new and must be
deallocated by the caller.
TODO: Returns NULL if memory allocation fails?
The @a offsets should satisfy: offsets.Size() == number of workers + 1,
offsets[0] == 0, and offsets[i] <= offsets[i+1], for i: 0 <= i < number
of workers. */
virtual DLayout MakeLayout(const Array<std::size_t> &offsets) const = 0;
// Note: There may be other ways to construct layouts in the future, e.g.
// block-vector layouts, or multi-vector layouts.
/// TODO
virtual DArray MakeArray(PLayout &layout, std::size_t item_size) const = 0;
/// Allocate and return a new vector using the given @a layout.
/** The returned object is a smart pointer that will automatically deallocate
the vector.
TODO: Produce an error if memory allocation fails?
Only layouts returned by this Engine are guaranteed to be supported.
Using a type that is not supported will produce an error. */
virtual DVector MakeVector(PLayout &layout,
int type_id = ScalarId<double>::value) const = 0;
/// TODO: doxygen
virtual DFiniteElementSpace MakeFESpace(FiniteElementSpace &fes) const;
/// TODO: doxygen
virtual DBilinearForm MakeBilinearForm(BilinearForm &bf) const = 0;
// Question: How do we construct coefficients?
/// FIXME - What will the actual parameters be?
virtual void AssembleLinearForm(LinearForm &l_form) const = 0;
/// FIXME - What will the actual parameters be?
virtual Operator *MakeOperator(const MixedBilinearForm &mbl_form) const = 0;
/// FIXME - What will the actual parameters be?
virtual Operator *MakeOperator(const NonlinearForm &nl_form) const = 0;
///@}
// End: Virtual interface
};
} // namespace mfem
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_BASE_ENGINE_HPP
-61
View File
@@ -1,61 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_BASE_FE_SPACE_HPP
#define MFEM_BACKENDS_BASE_FE_SPACE_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "engine.hpp"
#include "utils.hpp"
namespace mfem
{
class FiniteElementSpace;
/// TODO: doxygen
class PFiniteElementSpace : public RefCounted
{
protected:
/// Engine with shared ownership
SharedPtr<const Engine> engine;
/// Not owned.
FiniteElementSpace *fes;
public:
/// TODO: doxygen
PFiniteElementSpace(const Engine &e, FiniteElementSpace &fespace)
: engine(&e), fes(&fespace) { }
/// Virtual destructor
virtual ~PFiniteElementSpace() { }
/// Get the associated engine
const Engine &GetEngine() const { return *engine; }
mfem::FiniteElementSpace* GetFESpace() const { return fes; }
/// TODO
template <typename derived_t>
derived_t &As() { return *util::As<derived_t>(this); }
/// TODO
template <typename derived_t>
const derived_t &As() const { return *util::As<const derived_t>(this); }
};
} // namespace mfem
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_BASE_FE_SPACE_HPP
-104
View File
@@ -1,104 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_BASE_LAYOUT_HPP
#define MFEM_BACKENDS_BASE_LAYOUT_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "smart_pointers.hpp"
#include "engine.hpp"
namespace mfem
{
/// Polymorphic layout (array/vector layout descriptor)
class PLayout : public RefCounted
{
protected:
/// Engine with shared ownership
SharedPtr<const Engine> engine;
std::size_t size;
template <typename DObject>
struct Maker
{
template <typename entry_t>
static DObject MakeNew(PLayout &layout);
};
public:
explicit PLayout(std::size_t s = 0) : engine(NULL), size(s) { }
explicit PLayout(const Engine &e, std::size_t s = 0)
: engine(&e), size(s) { }
virtual ~PLayout() { }
/**
@name Virtual interface
*/
///@{
/// Resize the layout
virtual void Resize(std::size_t new_size) { size = new_size; }
/// Resize the layout based on the given worker offsets
virtual void Resize(const Array<std::size_t> &offsets)
{ MFEM_ABORT("method not supported"); }
///@}
// End: Virtual interface
/// Layouts without engine cannot create DArray, DVector, etc.
bool HasEngine() const { return engine != NULL; }
/// TODO: doxygen
const Engine &GetEngine() const { return *engine; }
/// TODO: doxygen
std::size_t Size() const { return size; }
/// TODO
template <typename derived_t>
derived_t &As() { return *util::As<derived_t>(this); }
/// TODO
template <typename derived_t>
const derived_t &As() const { return *util::As<const derived_t>(this); }
/// TODO: doxygen
template <typename DObject, typename entry_t>
DObject Make()
{
MFEM_ASSERT(HasEngine(), "this method requires an Engine");
return Maker<DObject>::template MakeNew<entry_t>(*this);
}
};
template <> struct PLayout::Maker<DArray>
{
template <typename entry_t> static DArray MakeNew(PLayout &layout)
{ return layout.GetEngine().MakeArray(layout, sizeof(entry_t)); }
};
template <> struct PLayout::Maker<DVector>
{
template <typename entry_t> static DVector MakeNew(PLayout &layout)
{ return layout.GetEngine().MakeVector(layout, ScalarId<entry_t>::value); }
};
} // namespace mfem
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_BASE_LAYOUT_HPP
-59
View File
@@ -1,59 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "memory_resource.hpp"
#include "../../general/error.hpp"
#include <cstdlib>
#include <cstring>
#include <cerrno>
namespace mfem
{
void *NewDeleteMemoryResource::DoAllocate(std::size_t bytes,
std::size_t alignment)
{
void *p = ::operator new[](bytes);
MFEM_VERIFY(!alignment || (std::size_t)(p) % alignment == 0,
"invalid alignment");
return p;
}
void NewDeleteMemoryResource::DoDeallocate(void *p, std::size_t bytes,
std::size_t alignment)
{
::operator delete[](p);
}
void *AlignedMemoryResource::DoAllocate(std::size_t bytes,
std::size_t alignment)
{
void *p;
if (!alignment) { alignment = sizeof(long double); }
MFEM_VERIFY(posix_memalign(&p, alignment, bytes) == 0,
"error in posix_memalign(): " << strerror(errno));
return p;
}
void AlignedMemoryResource::DoDeallocate(void *p, std::size_t bytes,
std::size_t alignment)
{
free(p);
}
} // namespace mfem
#endif // MFEM_USE_BACKENDS
-70
View File
@@ -1,70 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
#define MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include <cstddef>
namespace mfem
{
/// Polymorphic memory resource. Similar to C++17's std::pmr::memory_resource.
class MemoryResource
{
protected:
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment) = 0;
virtual void DoDeallocate(void* p, std::size_t bytes,
std::size_t alignment) = 0;
public:
// Implicitly defined default & copy constructors
/// Virtual destructor.
virtual ~MemoryResource() { }
/// If alignment == 0, use default alignment.
void *Allocate(std::size_t bytes, std::size_t alignment = 0)
{ return DoAllocate(bytes, alignment); }
/// If alignment == 0, use default alignment.
void Deallocate(void *p, std::size_t bytes, std::size_t alignment = 0)
{ DoDeallocate(p, bytes, alignment); }
};
/** @brief Dynamic host memory resource using operator new[](std::size_t) for
allocation and operator delete[](void*) for deallocation. */
class NewDeleteMemoryResource : public MemoryResource
{
protected:
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment);
virtual void DoDeallocate(void *p, std::size_t bytes, std::size_t alignment);
};
/** @brief Dynamic host memory resource using posix_memalign() for aligned
allocation and free() for deallocation. */
class AlignedMemoryResource : public MemoryResource
{
protected:
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment);
virtual void DoDeallocate(void *p, std::size_t bytes, std::size_t alignment);
};
} // namespace mfem
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
-233
View File
@@ -1,233 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
#define MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "utils.hpp"
#include "../../general/error.hpp"
#include <cstddef>
// #define MFEM_TRACE_SHARED_PTR
#ifdef MFEM_TRACE_SHARED_PTR
#include "../../general/globals.hpp"
#endif
namespace mfem
{
/// Base class for classes with simple reference counting.
/** Reference counting is performed by the class SharedPtr. */
class RefCounted
{
private:
mutable unsigned ref_count;
/// Only class SharedPtr can access ref_count.
template <typename T> friend class SharedPtr;
public:
RefCounted() : ref_count(0) { }
/** @brief Prevent SharedPtr objects from deleting this object by
incrementing the reference counter by one. */
void DontDelete() const { ++ref_count; }
};
/** @brief Smart pointer class that manages objects of type T derived from class
RefCounted. */
/** This class is generally meant to work with dynamically allocated object,
specifically objects allocated with operator new(). It will invoke operator
delete() to destroy the managed object when its reference counter reaches
zero. This behavior can be overriden by calling RefCounted::DontDelete() to
ensure that an object will not be deleted by a SharedPtr that holds a
pointer to it.
@note This class is NOT thread-safe and does not support circular ownership.
*/
template <typename T>
class SharedPtr
{
public:
typedef T stored_type;
private:
T *ptr;
void Init(T *new_ptr)
{
ptr = new_ptr;
if (ptr) { ++ptr->RefCounted::ref_count; }
#ifdef MFEM_TRACE_SHARED_PTR
#elif 0
mfem::out << " [" << _MFEM_FUNC_NAME << "]: ptr = " << ptr;
if (ptr)
{
mfem::out << ", new ref_count = " << ptr->RefCounted::ref_count;
}
mfem::out << '\n';
#endif
}
void Destroy()
{
MFEM_ASSERT(!ptr || ptr->RefCounted::ref_count >= 1, "invalid use");
if (ptr && --ptr->RefCounted::ref_count == 0) { delete ptr; }
#ifdef MFEM_TRACE_SHARED_PTR
#elif 0
mfem::out << " [" << _MFEM_FUNC_NAME << "]: ptr = " << ptr;
if (ptr)
{
mfem::out << ", new ref_count = " << ptr->RefCounted::ref_count;
}
mfem::out << '\n';
#endif
}
public:
SharedPtr() : ptr(NULL)
{
#ifdef MFEM_TRACE_SHARED_PTR
mfem::out << '[' << _MFEM_FUNC_NAME << "]: ptr = " << ptr << '\n';
#endif
}
SharedPtr(const SharedPtr &other)
{
#ifdef MFEM_TRACE_SHARED_PTR
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
#endif
Init(other.ptr);
}
template <typename U>
SharedPtr(const SharedPtr<U> &other)
{
#ifdef MFEM_TRACE_SHARED_PTR
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
#endif
Init(other.Get());
}
explicit SharedPtr(T *p)
{
#ifdef MFEM_TRACE_SHARED_PTR
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
#endif
Init(p);
}
~SharedPtr()
{
#ifdef MFEM_TRACE_SHARED_PTR
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
#endif
Destroy();
}
SharedPtr &operator=(const SharedPtr &other)
{
#ifdef MFEM_TRACE_SHARED_PTR
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
#endif
Reset(other.ptr); return *this;
}
template <typename U>
SharedPtr &operator=(const SharedPtr<U> &other)
{
#ifdef MFEM_TRACE_SHARED_PTR
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
#endif
Reset(other.Get()); return *this;
}
T &operator*() const { return *ptr; }
T *operator->() const { return ptr; }
operator bool() const { return ptr; }
bool operator!() const { return !ptr; }
template <typename U>
bool operator==(const SharedPtr<U> &other) const
{ return ptr == other.Ptr(); }
template <typename U>
bool operator!=(const SharedPtr<U> &other) const
{ return ptr != other.Ptr(); }
template <typename U>
bool operator==(const U &p) const { return ptr == (void*) p; }
template <typename U>
bool operator!=(const U &p) const { return ptr != (void*) p; }
T *Get() const { return ptr; }
/// TODO
template <typename derived_t>
derived_t *As() const { return util::As<derived_t>(ptr); }
unsigned UseCount() const { return ptr ? ptr->RefCounted::ref_count : 0; }
void Reset()
{
#ifdef MFEM_TRACE_SHARED_PTR
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
#endif
Destroy();
ptr = NULL;
}
/// The type U* needs to be implicitly convertible to T*
template <typename U>
void Reset(U *new_ptr)
{
#ifdef MFEM_TRACE_SHARED_PTR
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
#endif
if (ptr != new_ptr) { Destroy(); Init(new_ptr); }
}
void Swap(SharedPtr &other)
{
#ifdef MFEM_TRACE_SHARED_PTR
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
#endif
std::swap(ptr, other.ptr);
}
};
template <class T>
inline void Swap(SharedPtr<T> &a, SharedPtr<T> &b) { a.Swap(b); }
class PLayout;
typedef SharedPtr<PLayout> DLayout;
class PArray;
typedef SharedPtr<PArray> DArray;
class PVector;
typedef SharedPtr<PVector> DVector;
class PFiniteElementSpace;
typedef SharedPtr<PFiniteElementSpace> DFiniteElementSpace;
class PBilinearForm;
typedef SharedPtr<PBilinearForm> DBilinearForm;
} // namespace mfem
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
-52
View File
@@ -1,52 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_BASE_UTILS_HPP
#define MFEM_BACKENDS_BASE_UTILS_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "../../general/error.hpp"
namespace mfem
{
namespace util
{
//
// Inline methods
//
/// TODO: doxygen
template <typename derived_t, typename base_t>
inline derived_t *As(base_t *base_obj)
{
MFEM_ASSERT(dynamic_cast<derived_t*>(base_obj) != NULL,
"invalid object type");
return static_cast<derived_t*>(base_obj);
}
/// TODO: doxygen
template <typename derived_t, typename base_t>
inline derived_t *Is(base_t *base_obj)
{
return dynamic_cast<derived_t*>(base_obj);
}
} // namespace mfem::util
} // namespace mfem
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_BASE_UTILS_HPP
-153
View File
@@ -1,153 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_BASE_VECTOR_HPP
#define MFEM_BACKENDS_BASE_VECTOR_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_BACKENDS
#include "../../general/scalars.hpp"
#include "array.hpp"
namespace mfem
{
/// Polymorphic vector - array of scalars.
class PVector : virtual public PArray
{
protected:
/**
@name Virtual interface
*/
///@{
/** @brief Create and return a new vector of the same dynamic type as this
vector using the same layout with entries specified by @a buffer_type_id
which should be a constant defined by the `value` field in a
specialization of the template class mfem::ScalarId.
Returns NULL if allocation fails.
If @a copy_data is true, the contents of this vector is copied to the new
vector; otherwise, the new vector remains uninitialized.
If @a buffer is not NULL, return the vector data of the newly created
object (in @a *buffer), if it is stored as a contiguous array on the
host; otherwise, set @a *buffer to NULL. */
virtual PVector *DoVectorClone(bool copy_data, void **buffer,
int buffer_type_id) const = 0;
/** @brief Compute and return the dot product of @a *this and @a x. In the
case of an MPI-parallel vector, the result must be the MPI-global dot
product. */
/** Both vectors must have the same dynamic type and layout. */
virtual void DoDotProduct(const PVector &x, void *result,
int result_type_id) const = 0;
// TODO: add reduction operations: min, max, sum
/// Perform the operation @a *this = @a a @a x + @a b @a y.
/** Rules:
- the dynamic type of both @a x and @a y is the same as that of @a *this
- if @a a == 0, neither @a x nor its data are accessed
- if @a b == 0, neither @a y nor its data are accessed
- @a x's data is never the same as @a y's data, unless @a a == 0, or
@a b == 0
- @a x's data or @a y's data may be the same as the data of @a *this
- all accessed vectors, @a x, @a y, and @a *this have the same layout. */
virtual void DoAxpby(const void *a, const PVector &x,
const void *b, const PVector &y,
int ab_type_id) = 0;
///@}
// End: Virtual interface
public:
/** @brief Create a PVector. */
/** The @a layout must be valid in the sense that layout != NULL and
layout->HasEngine() == true. */
PVector(PLayout &p_layout)
: PArray(p_layout) { }
template <typename derived_t>
derived_t &As() { return *util::As<derived_t>(this); }
template <typename derived_t>
const derived_t &As() const { return *util::As<const derived_t>(this); }
// TODO: Error handling ... handle errors at the Engine level, at the class
// level, or at the method level?
// TODO: Asynchronous execution interface ...
// TODO: Multi-vector interface ...
/**
@name Public virtual interface
*/
///@{
/** @brief Create and return a new vector of the same dynamic type as this
vector using the same layout with entries of type @a scalar_t.
If @a copy_data is true, the contents of this vector is copied to the new
vector; otherwise, the new vector remains uninitialized.
If @a buffer is not NULL, return the vector data of the newly created
object (in @a *buffer) , if it is stored as a contiguous array on the
host; otherwise, set @a *buffer to NULL. */
template <typename scalar_t>
DVector Clone(bool copy_data, scalar_t **buffer) const
{
return DVector(DoVectorClone(copy_data, (void**)buffer,
ScalarId<scalar_t>::value));
}
/** @brief Compute and return the dot product of @a *this and @a x. In the
case of an MPI-parallel vector, the result must be the MPI-global dot
product. */
/** Both vectors must have the same dynamic type and layout. */
template <typename scalar_t>
scalar_t DotProduct(const PVector &x) const
{
scalar_t result;
DoDotProduct(x, &result, ScalarId<scalar_t>::value);
return result;
}
// TODO: add reduction operations: min, max, sum
/// Perform the operation @a *this = @a a @a x + @a b @a y.
/** Rules:
- the dynamic type of both @a x and @a y is the same as that of @a *this
- if @a a == 0, neither @a x nor its data are accessed
- if @a b == 0, neither @a y nor its data are accessed
- @a x's data is never the same as @a y's data, unless @a a == 0, or
@a b == 0
- @a x's data or @a y's data may be the same as the data of @a *this
- all accessed vectors, @a x, @a y, and @a *this have the same layout. */
template <typename scalar_t>
void Axpby(const scalar_t &a, const PVector &x,
const scalar_t &b, const PVector &y)
{ if (Size()) { DoAxpby(&a, x, &b, y, ScalarId<scalar_t>::value); } }
///@}
// End: Virtual interface
};
} // namespace mfem
#endif // MFEM_USE_BACKENDS
#endif // MFEM_BACKENDS_BASE_VECTOR_HPP
-66
View File
@@ -1,66 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
/*
---[ Defines Known At Compile-Time ]------------
ELEMENT_BATCH : How many elements are in each
. computation batch
NUM_DOFS_1D : Dofs in the 1D segments
NUM_DOFS_2D : Dofs in the 2D faces
NUM_DOFS_3D : Dofs in the 3D domain
NUM_QUAD_1D : Dofs in the 1D segments
NUM_QUAD_2D : Dofs in the 2D faces
NUM_QUAD_3D : Dofs in the 3D domain
NUM_MAX_1D : max(NUM_QUAD_1D, NUM_DOFS_1D)
NUM_QUAD_DOFS_1D: NUM_QUAD_1D * NUM_DOFS_1D
COEFF_ARGS : Code that passes required arguments to the kernel
COEFF : Code that computes the coefficient
================================================
[MISSING]
- Add support to auto-pick @dim and use @idxOrder on stack arrays
| double a[2][2];
| a[0][1]; <-- regular index
| a(0,1); <-- uses @idxOrder a[0][1] or a[1][0]
- Add support for @idxOrder to change indexing order after allocation
| double a[2][2] @idxOrder(0,1);
| a(0,1) -> a[1][0]
| @set(a, idxOrder(1,0));
| a(0,1) -> a[0][1]
- Add support to iterate over loop depending on mode
| for(i; @inner) {
| for(0 < j < N) {} <-- ++j or j += block?
| }
*/
#include "mfem-occa://defines.okl"
#if USING_TENSOR_OPS
# ifdef OCCA_USING_GPU
# if USING_LOW_ORDER
# include "mfem-occa://diffusion/tensor/gpuHighOrder.okl"
# else
# include "mfem-occa://diffusion/tensor/gpuHighOrder.okl"
# endif
# else
# include "mfem-occa://diffusion/tensor/cpu.okl"
# endif
#else
# ifdef OCCA_USING_GPU
# if USING_LOW_ORDER
# include "mfem-occa://diffusion/simplex/gpuHighOrder.okl"
# else
# include "mfem-occa://diffusion/simplex/gpuHighOrder.okl"
# endif
# else
# include "mfem-occa://diffusion/simplex/cpu.okl"
# endif
#endif
-123
View File
@@ -1,123 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "array.hpp"
namespace mfem
{
namespace occa
{
PArray *Array::DoClone(bool copy_data, void **buffer,
std::size_t item_size) const
{
Array *new_array = new Array(OccaLayout(), item_size);
if (copy_data)
{
new_array->slice.copyFrom(slice);
}
if (buffer)
{
*buffer = new_array->GetBuffer();
}
return new_array;
}
int Array::DoResize(PLayout &new_layout, void **buffer,
std::size_t item_size)
{
MFEM_ASSERT(dynamic_cast<Layout *>(&new_layout) != NULL,
"new_layout is not an OCCA Layout");
Layout *lt = static_cast<Layout *>(&new_layout);
layout.Reset(lt); // Reset() checks if the pointer is the same
int err = ResizeData(lt, item_size);
if (!err && buffer)
{
*buffer = GetBuffer();
}
return err;
}
void *Array::DoPullData(void *buffer, std::size_t item_size)
{
// called only when Size() != 0
if (!slice.getDevice().hasSeparateMemorySpace())
{
return slice.ptr();
}
if (buffer)
{
slice.copyTo(buffer);
}
return buffer;
}
void Array::DoFill(const void *value_ptr, std::size_t item_size)
{
// called only when Size() != 0
switch (item_size)
{
case sizeof(int8_t):
OccaFill((const int8_t *)value_ptr);
break;
case sizeof(int16_t):
OccaFill((const int16_t *)value_ptr);
break;
case sizeof(int32_t):
OccaFill((const int32_t *)value_ptr);
break;
// case sizeof(int64_t):
// OccaFill((const int64_t *)value_ptr);
// break;
case sizeof(double):
OccaFill((const double *)value_ptr);
break;
// case sizeof(::occa::double2):
// OccaFill((const ::occa::double2 *)value_ptr);
// break;
default:
MFEM_ABORT("item_size = " << item_size << " is not supported");
}
}
void Array::DoPushData(const void *src_buffer, std::size_t item_size)
{
// called only when Size() != 0
if (slice.getDevice().hasSeparateMemorySpace() || slice.ptr() != src_buffer)
{
slice.copyFrom(src_buffer);
}
}
void Array::DoAssign(const PArray &src, std::size_t item_size)
{
// called only when Size() != 0
// Note: static_cast can not be used here since PArray is a virtual base
// class.
const Array *source = dynamic_cast<const Array *>(&src);
MFEM_ASSERT(source != NULL, "invalid source Array type");
MFEM_ASSERT(Size() == source->Size(), "");
slice.copyFrom(source->slice);
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-133
View File
@@ -1,133 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_ARRAY_HPP
#define MFEM_BACKENDS_OCCA_ARRAY_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include <occa.hpp>
#include "layout.hpp"
#include "../base/array.hpp"
namespace mfem
{
namespace occa
{
class Array : public virtual PArray
{
protected:
//
// Inherited fields
//
// DLayout layout;
// Always true: Size()*item_size == slice.size() <= data.size()
mutable ::occa::memory data, slice;
//
// Virtual interface
//
virtual void *DoGetData() const { return GetBuffer(); }
virtual PArray *DoClone(bool copy_data, void **buffer,
std::size_t item_size) const;
virtual int DoResize(PLayout &new_layout, void **buffer,
std::size_t item_size);
virtual void *DoPullData(void *buffer, std::size_t item_size);
virtual void DoFill(const void *value_ptr, std::size_t item_size);
virtual void DoPushData(const void *src_buffer, std::size_t item_size);
virtual void DoAssign(const PArray &src, std::size_t item_size);
//
// Auxiliary methods
//
inline void *GetBuffer() const;
inline int ResizeData(const Layout *lt, std::size_t item_size);
template <typename T>
inline void OccaFill(const T *val_ptr)
{ ::occa::linalg::operator_eq<T>(slice, *val_ptr); }
public:
Array(Layout &lt, std::size_t item_size)
: PArray(lt),
data(lt.Alloc(lt.Size()*item_size)),
slice(data)
{ }
virtual ~Array() { }
inline void MakeRef(Array &master);
Layout &OccaLayout() const
{ return *static_cast<Layout *>(layout.Get()); }
::occa::memory &OccaMem() { return slice; }
const ::occa::memory &OccaMem() const { return slice; }
};
//
// Inline methods
//
inline void *Array::GetBuffer() const
{
if (!slice.getDevice().hasSeparateMemorySpace())
{
return slice.ptr();
}
return NULL;
}
inline int Array::ResizeData(const Layout *lt, std::size_t item_size)
{
const std::size_t new_bytes = lt->Size()*item_size;
if (data.size() < new_bytes ||
data.getDHandle() != lt->OccaEngine().GetDevice().getDHandle())
{
data = lt->Alloc(new_bytes);
slice = data;
// If memory allocation fails - an exception is thrown.
}
else if (slice.size() != new_bytes)
{
slice = data.slice(0, new_bytes);
}
return 0;
}
inline void Array::MakeRef(Array &master)
{
layout = master.layout;
data = master.data;
slice = master.slice;
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_ARRAY_HPP
-47
View File
@@ -1,47 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "backend.hpp"
#include "engine.hpp"
namespace mfem
{
namespace occa
{
bool Backend::Supports(const std::string &engine_spec) const
{
// TODO: check if 'engine_spec' is valid OCCA string.
return true;
}
mfem::Engine *Create(const std::string &engine_spec)
{
return new Engine(engine_spec);
}
#ifdef MFEM_USE_MPI
mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec)
{
return new Engine(comm, engine_spec);
}
#endif
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-49
View File
@@ -1,49 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_BACKEND_HPP
#define MFEM_BACKENDS_OCCA_BACKEND_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
// Only the Backend and Engine classes should be exposed through "backend.hpp"
#include "../base/backend.hpp"
#include "engine.hpp"
#include <occa.hpp>
namespace mfem
{
namespace occa
{
class Backend : public mfem::Backend
{
public:
virtual ~Backend();
virtual bool Supports(const std::string &engine_spec) const;
virtual mfem::Engine *Create(const std::string &engine_spec);
#ifdef MFEM_USE_MPI
virtual mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec);
#endif
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_BACKEND_HPP
-514
View File
@@ -1,514 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "backend.hpp"
#include "bilininteg.hpp"
#include "../../fem/bilinearform.hpp"
namespace mfem
{
namespace occa
{
OccaBilinearForm::OccaBilinearForm(FiniteElementSpace *ofespace_) :
Operator(ofespace_->OccaVLayout()),
localX((ofespace_->OccaEVLayout().DontDelete(), ofespace_->OccaEVLayout())),
localY((ofespace_->OccaEVLayout().DontDelete(), ofespace_->OccaEVLayout()))
{
Init(ofespace_->OccaEngine(), ofespace_, ofespace_);
}
OccaBilinearForm::OccaBilinearForm(FiniteElementSpace *otrialFESpace_,
FiniteElementSpace *otestFESpace_) :
Operator(otrialFESpace_->OccaVLayout(),
otestFESpace_->OccaVLayout()),
localX((otrialFESpace_->OccaEVLayout().DontDelete(), otrialFESpace_->OccaEVLayout())),
localY((otestFESpace_->OccaEVLayout().DontDelete(), otestFESpace_->OccaEVLayout()))
{
Init(otrialFESpace_->OccaEngine(), otrialFESpace_, otestFESpace_);
}
void OccaBilinearForm::Init(const Engine &e,
FiniteElementSpace *otrialFESpace_,
FiniteElementSpace *otestFESpace_)
{
engine.Reset(&e);
otrialFESpace = otrialFESpace_;
trialFESpace = otrialFESpace_->GetFESpace();
otestFESpace = otestFESpace_;
testFESpace = otestFESpace_->GetFESpace();
mesh = trialFESpace->GetMesh();
const int elements = GetNE();
const int trialVDim = trialFESpace->GetVDim();
const int trialLocalDofs = otrialFESpace->GetLocalDofs();
const int testLocalDofs = otestFESpace->GetLocalDofs();
// First-touch policy when running with OpenMP
if (GetDevice().mode() == "OpenMP")
{
const std::string &okl_path = OccaEngine().GetOklPath();
const std::string &okl_defines = OccaEngine().GetOklDefines();
::occa::kernel initLocalKernel =
GetDevice().buildKernel(okl_path + "utils.okl",
"InitLocalVector",
okl_defines);
const std::size_t sd = sizeof(double);
const uint64_t trialEntries = sd * (elements * trialLocalDofs);
const uint64_t testEntries = sd * (elements * testLocalDofs);
for (int v = 0; v < trialVDim; ++v)
{
const uint64_t trialOffset = v * trialEntries;
const uint64_t testOffset = v * testEntries;
initLocalKernel(elements, trialLocalDofs,
localX.OccaMem().slice(trialOffset, trialEntries));
initLocalKernel(elements, testLocalDofs,
localY.OccaMem().slice(testOffset, testEntries));
}
}
}
int OccaBilinearForm::BaseGeom() const
{
return mesh->GetElementBaseGeometry();
}
int OccaBilinearForm::GetDim() const
{
return mesh->Dimension();
}
int64_t OccaBilinearForm::GetNE() const
{
return mesh->GetNE();
}
Mesh& OccaBilinearForm::GetMesh() const
{
return *mesh;
}
FiniteElementSpace& OccaBilinearForm::GetTrialOccaFESpace() const
{
return *otrialFESpace;
}
FiniteElementSpace& OccaBilinearForm::GetTestOccaFESpace() const
{
return *otestFESpace;
}
mfem::FiniteElementSpace& OccaBilinearForm::GetTrialFESpace() const
{
return *trialFESpace;
}
mfem::FiniteElementSpace& OccaBilinearForm::GetTestFESpace() const
{
return *testFESpace;
}
int64_t OccaBilinearForm::GetTrialNDofs() const
{
return trialFESpace->GetNDofs();
}
int64_t OccaBilinearForm::GetTestNDofs() const
{
return testFESpace->GetNDofs();
}
int64_t OccaBilinearForm::GetTrialVDim() const
{
return trialFESpace->GetVDim();
}
int64_t OccaBilinearForm::GetTestVDim() const
{
return testFESpace->GetVDim();
}
const FiniteElement& OccaBilinearForm::GetTrialFE(const int i) const
{
return *(trialFESpace->GetFE(i));
}
const FiniteElement& OccaBilinearForm::GetTestFE(const int i) const
{
return *(testFESpace->GetFE(i));
}
// Adds new Domain Integrator.
void OccaBilinearForm::AddDomainIntegrator(OccaIntegrator *integrator,
const ::occa::properties &props)
{
AddIntegrator(integrator, props, DomainIntegrator);
}
// Adds new Boundary Integrator.
void OccaBilinearForm::AddBoundaryIntegrator(OccaIntegrator *integrator,
const ::occa::properties &props)
{
AddIntegrator(integrator, props, BoundaryIntegrator);
}
// Adds new interior Face Integrator.
void OccaBilinearForm::AddInteriorFaceIntegrator(OccaIntegrator *integrator,
const ::occa::properties &props)
{
AddIntegrator(integrator, props, InteriorFaceIntegrator);
}
// Adds new boundary Face Integrator.
void OccaBilinearForm::AddBoundaryFaceIntegrator(OccaIntegrator *integrator,
const ::occa::properties &props)
{
AddIntegrator(integrator, props, BoundaryFaceIntegrator);
}
// Adds Integrator based on OccaIntegratorType
void OccaBilinearForm::AddIntegrator(OccaIntegrator *integrator,
const ::occa::properties &props,
const OccaIntegratorType itype)
{
if (integrator == NULL)
{
std::stringstream error_ss;
error_ss << "OccaBilinearForm::";
switch (itype)
{
case DomainIntegrator : error_ss << "AddDomainIntegrator"; break;
case BoundaryIntegrator : error_ss << "AddBoundaryIntegrator"; break;
case InteriorFaceIntegrator: error_ss << "AddInteriorFaceIntegrator"; break;
case BoundaryFaceIntegrator: error_ss << "AddBoundaryFaceIntegrator"; break;
}
error_ss << " (...):\n"
<< " Integrator is NULL";
const std::string error = error_ss.str();
mfem_error(error.c_str());
}
integrator->SetupIntegrator(*this, baseKernelProps + props, itype);
integrators.push_back(integrator);
}
const mfem::Operator* OccaBilinearForm::GetTrialProlongation() const
{
return otrialFESpace->GetProlongationOperator();
}
const mfem::Operator* OccaBilinearForm::GetTestProlongation() const
{
return otestFESpace->GetProlongationOperator();
}
const mfem::Operator* OccaBilinearForm::GetTrialRestriction() const
{
return otrialFESpace->GetRestrictionOperator();
}
const mfem::Operator* OccaBilinearForm::GetTestRestriction() const
{
return otestFESpace->GetRestrictionOperator();
}
void OccaBilinearForm::Assemble()
{
// [MISSING] Find geometric information that is needed by intergrators
// to share between integrators.
const int integratorCount = (int) integrators.size();
for (int i = 0; i < integratorCount; ++i)
{
integrators[i]->Assemble();
}
}
void OccaBilinearForm::FormLinearSystem(const mfem::Array<int> &constraintList,
mfem::Vector &x, mfem::Vector &b,
mfem::Operator *&Aout,
mfem::Vector &X, mfem::Vector &B,
int copy_interior)
{
FormOperator(constraintList, Aout);
InitRHS(constraintList, x, b, Aout, X, B, copy_interior);
}
void OccaBilinearForm::FormOperator(const mfem::Array<int> &constraintList,
mfem::Operator *&Aout)
{
const mfem::Operator *trialP = GetTrialProlongation();
const mfem::Operator *testP = GetTestProlongation();
mfem::Operator *rap = this;
if (trialP)
{
rap = new RAPOperator(*testP, *this, *trialP);
}
Aout = new OccaConstrainedOperator(rap, constraintList,
rap != this);
}
void OccaBilinearForm::InitRHS(const mfem::Array<int> &constraintList,
mfem::Vector &x, mfem::Vector &b,
mfem::Operator *A,
mfem::Vector &X, mfem::Vector &B,
int copy_interior)
{
const std::string okl_defines = OccaEngine().GetOklDefines();
// FIXME: move these kernels to the Backend?
static ::occa::kernelBuilder get_subvector_builder =
::occa::linalg::customLinearMethod(
"vector_get_subvector",
"const int dof_i = v2[i];"
"v0[i] = dof_i >= 0 ? v1[dof_i] : -v1[-dof_i - 1];",
"defines: {"
" VTYPE0: 'double',"
" VTYPE1: 'double',"
" VTYPE2: 'int',"
" TILESIZE: 128,"
"}" + okl_defines);
static ::occa::kernelBuilder set_subvector_builder =
::occa::linalg::customLinearMethod(
"vector_set_subvector",
"const int dof_i = v2[i];"
"if (dof_i >= 0) { v0[dof_i] = v1[i]; }"
"else { v0[-dof_i - 1] = -v1[i]; }",
"defines: {"
" VTYPE0: 'double',"
" VTYPE1: 'double',"
" VTYPE2: 'int',"
" TILESIZE: 128,"
"}" + okl_defines);
const mfem::Operator *P = GetTrialProlongation();
const mfem::Operator *R = GetTrialRestriction();
if (P)
{
// Variational restriction with P
B.Resize(P->InLayout());
P->MultTranspose(b, B);
X.Resize(R->OutLayout());
R->Mult(x, X);
}
else
{
// rap, X and B point to the same data as this, x and b
X.MakeRef(x);
B.MakeRef(b);
}
if (!copy_interior && constraintList.Size() > 0)
{
::occa::kernel get_subvector_kernel =
get_subvector_builder.build(GetDevice());
::occa::kernel set_subvector_kernel =
set_subvector_builder.build(GetDevice());
const Array &constrList = constraintList.Get_PArray()->As<Array>();
Vector subvec(constrList.OccaLayout());
get_subvector_kernel(constraintList.Size(),
subvec.OccaMem(),
X.Get_PVector()->As<Vector>().OccaMem(),
constrList.OccaMem());
X.Fill(0.0);
set_subvector_kernel(constraintList.Size(),
X.Get_PVector()->As<Vector>().OccaMem(),
subvec.OccaMem(),
constrList.OccaMem());
}
OccaConstrainedOperator *cA = dynamic_cast<OccaConstrainedOperator*>(A);
if (cA)
{
cA->EliminateRHS(X.Get_PVector()->As<Vector>(),
B.Get_PVector()->As<Vector>());
}
else
{
mfem_error("OccaBilinearForm::InitRHS expects an OccaConstrainedOperator");
}
}
// Matrix vector multiplication.
void OccaBilinearForm::Mult_(const Vector &x, Vector &y) const
{
otrialFESpace->GlobalToLocal(x, localX);
localY.Fill<double>(0.0);
const int integratorCount = (int) integrators.size();
for (int i = 0; i < integratorCount; ++i)
{
integrators[i]->MultAdd(localX, localY);
}
otestFESpace->LocalToGlobal(localY, y);
}
// Matrix transpose vector multiplication.
void OccaBilinearForm::MultTranspose_(const Vector &x, Vector &y) const
{
otestFESpace->GlobalToLocal(x, localX);
localY.Fill<double>(0.0);
const int integratorCount = (int) integrators.size();
for (int i = 0; i < integratorCount; ++i)
{
integrators[i]->MultTransposeAdd(localX, localY);
}
otrialFESpace->LocalToGlobal(localY, y);
}
void OccaBilinearForm::OccaRecoverFEMSolution(const mfem::Vector &X,
const mfem::Vector &b,
mfem::Vector &x)
{
const mfem::Operator *P = this->GetTrialProlongation();
if (P)
{
// Apply conforming prolongation
x.Resize(P->OutLayout());
P->Mult(X, x);
}
// Otherwise X and x point to the same data
}
// Frees memory bilinear form.
OccaBilinearForm::~OccaBilinearForm()
{
// Make sure all integrators free their data
IntegratorVector::iterator it = integrators.begin();
while (it != integrators.end())
{
delete *it;
++it;
}
}
void BilinearForm::InitOccaBilinearForm()
{
// Init 'obform' using 'bform'
MFEM_ASSERT(bform != NULL, "");
MFEM_ASSERT(obform == NULL, "");
FiniteElementSpace &ofes =
bform->FESpace()->Get_PFESpace()->As<FiniteElementSpace>();
obform = new OccaBilinearForm(&ofes);
// Transfer domain integrators
mfem::Array<mfem::BilinearFormIntegrator*> &dbfi = *bform->GetDBFI();
for (int i = 0; i < dbfi.Size(); i++)
{
std::string integ_name(dbfi[i]->Name());
Coefficient *scal_coeff = dbfi[i]->GetScalarCoefficient();
ConstantCoefficient *const_coeff =
dynamic_cast<ConstantCoefficient*>(scal_coeff);
// TODO: other types of coefficients ...
double val = const_coeff ? const_coeff->constant : 1.0;
OccaCoefficient ocoeff(obform->OccaEngine(), val);
OccaIntegrator *ointeg = NULL;
if (integ_name == "(undefined)")
{
MFEM_ABORT("BilinearFormIntegrator does not define Name()");
}
else if (integ_name == "diffusion")
{
ointeg = new OccaDiffusionIntegrator(ocoeff);
}
else
{
MFEM_ABORT("BilinearFormIntegrator [Name() = " << integ_name
<< "] is not supported");
}
const mfem::IntegrationRule *ir = dbfi[i]->GetIntRule();
if (ir) { ointeg->SetIntegrationRule(*ir); }
obform->AddDomainIntegrator(ointeg);
}
// TODO: other types of integrators ...
}
bool BilinearForm::Assemble()
{
if (obform == NULL) { InitOccaBilinearForm(); }
obform->Assemble();
return true; // --> host assembly is not needed
}
void BilinearForm::FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
mfem::OperatorHandle &A)
{
if (A.Type() == mfem::Operator::ANY_TYPE)
{
mfem::Operator *Aout = NULL;
obform->FormOperator(ess_tdof_list, Aout);
A.Reset(Aout);
}
else
{
MFEM_ABORT("Operator::Type is not supported, type = " << A.Type());
}
}
void BilinearForm::FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
mfem::Vector &x, mfem::Vector &b,
mfem::OperatorHandle &A,
mfem::Vector &X, mfem::Vector &B,
int copy_interior)
{
FormSystemMatrix(ess_tdof_list, A);
obform->InitRHS(ess_tdof_list, x, b, A.Ptr(), X, B, copy_interior);
}
void BilinearForm::RecoverFEMSolution(const mfem::Vector &X,
const mfem::Vector &b,
mfem::Vector &x)
{
obform->OccaRecoverFEMSolution(X, b, x);
}
BilinearForm::~BilinearForm()
{
delete obform;
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-213
View File
@@ -1,213 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
#define MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "fespace.hpp"
namespace mfem
{
namespace occa
{
enum OccaIntegratorType
{
DomainIntegrator = 0,
BoundaryIntegrator = 1,
InteriorFaceIntegrator = 2,
BoundaryFaceIntegrator = 3
};
class OccaIntegrator;
/** Class for bilinear form - "Matrix" with associated FE space and
BLFIntegrators. */
class OccaBilinearForm : public Operator
{
friend class OccaIntegrator;
protected:
typedef std::vector<OccaIntegrator*> IntegratorVector;
SharedPtr<const Engine> engine;
// State information
mutable mfem::Mesh *mesh;
mutable FiniteElementSpace *otrialFESpace;
mutable mfem::FiniteElementSpace *trialFESpace;
mutable FiniteElementSpace *otestFESpace;
mutable mfem::FiniteElementSpace *testFESpace;
IntegratorVector integrators;
// Device data
::occa::properties baseKernelProps;
// The input and output vectors are mapped to local nodes for efficient
// operations. In other words, they are E-vectors.
// The size is: (number of elements) * (nodes in element) * (vector dim)
mutable Vector localX, localY;
public:
OccaBilinearForm(FiniteElementSpace *ofespace_);
OccaBilinearForm(FiniteElementSpace *otrialFESpace_,
FiniteElementSpace *otestFESpace_);
void Init(const Engine &e,
FiniteElementSpace *otrialFESpace_,
FiniteElementSpace *otestFESpace_);
const Engine &OccaEngine() const { return *engine; }
::occa::device GetDevice(int idx = 0) const
{ return engine->GetDevice(idx); }
// Useful mesh Information
int BaseGeom() const;
int GetDim() const;
int64_t GetNE() const;
mfem::Mesh& GetMesh() const;
FiniteElementSpace& GetTrialOccaFESpace() const;
FiniteElementSpace& GetTestOccaFESpace() const;
mfem::FiniteElementSpace& GetTrialFESpace() const;
mfem::FiniteElementSpace& GetTestFESpace() const;
// Useful FE information
int64_t GetTrialNDofs() const;
int64_t GetTestNDofs() const;
int64_t GetTrialVDim() const;
int64_t GetTestVDim() const;
const mfem::FiniteElement& GetTrialFE(const int i) const;
const mfem::FiniteElement& GetTestFE(const int i) const;
// Adds new Domain Integrator.
void AddDomainIntegrator(OccaIntegrator *integrator,
const ::occa::properties &props =
::occa::properties());
// Adds new Boundary Integrator.
void AddBoundaryIntegrator(OccaIntegrator *integrator,
const ::occa::properties &props =
::occa::properties());
// Adds new interior Face Integrator.
void AddInteriorFaceIntegrator(OccaIntegrator *integrator,
const ::occa::properties &props =
::occa::properties());
// Adds new boundary Face Integrator.
void AddBoundaryFaceIntegrator(OccaIntegrator *integrator,
const ::occa::properties &props =
::occa::properties());
// Adds Integrator based on OccaIntegratorType
void AddIntegrator(OccaIntegrator *integrator,
const ::occa::properties &props,
const OccaIntegratorType itype);
virtual const mfem::Operator *GetTrialProlongation() const;
virtual const mfem::Operator *GetTestProlongation() const;
virtual const mfem::Operator *GetTrialRestriction() const;
virtual const mfem::Operator *GetTestRestriction() const;
// Assembles the form i.e. sums over all domain/bdr integrators.
virtual void Assemble();
void FormLinearSystem(const mfem::Array<int> &constraintList,
mfem::Vector &x, mfem::Vector &b,
mfem::Operator *&Aout,
mfem::Vector &X, mfem::Vector &B,
int copy_interior = 0);
void FormOperator(const mfem::Array<int> &constraintList,
mfem::Operator *&Aout);
void InitRHS(const mfem::Array<int> &constraintList,
mfem::Vector &x, mfem::Vector &b,
mfem::Operator *Aout,
mfem::Vector &X, mfem::Vector &B,
int copy_interior = 0);
// overrides
virtual void Mult_(const Vector &x, Vector &y) const;
virtual void MultTranspose_(const Vector &x, Vector &y) const;
void OccaRecoverFEMSolution(const mfem::Vector &X, const mfem::Vector &b,
mfem::Vector &x);
// Destroys bilinear form.
~OccaBilinearForm();
};
/// TODO: doxygen
class BilinearForm : public mfem::PBilinearForm
{
protected:
//
// Inherited fields
//
// SharedPtr<const mfem::Engine> engine;
// mfem::BilinearForm *bform;
OccaBilinearForm *obform;
// Called from Assemble() if obform is NULL to initialize obform.
void InitOccaBilinearForm();
public:
/// TODO: doxygen
BilinearForm(const Engine &e, mfem::BilinearForm &bf)
: mfem::PBilinearForm(e, bf), obform(NULL) { }
/// Virtual destructor
virtual ~BilinearForm();
/// Assemble the PBilinearForm.
/** This method is called from the method mfem::BilinearForm::Assemble() of
the associated mfem::BilinearForm, #bform.
@returns True, if the host assembly should NOT be performed. */
virtual bool Assemble();
virtual void FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
mfem::OperatorHandle &A);
virtual void FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
mfem::Vector &x, mfem::Vector &b,
mfem::OperatorHandle &A,
mfem::Vector &X, mfem::Vector &B,
int copy_interior);
virtual void RecoverFEMSolution(const mfem::Vector &X, const mfem::Vector &b,
mfem::Vector &x);
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
-956
View File
@@ -1,956 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "bilininteg.hpp"
#include "../../fem/fem.hpp"
namespace mfem
{
namespace occa
{
std::map<std::string, OccaDofQuadMaps> OccaDofQuadMaps::AllDofQuadMaps;
OccaGeometry OccaGeometry::Get(::occa::device device,
FiniteElementSpace &ofespace,
const mfem::IntegrationRule &ir,
const int flags)
{
OccaGeometry geom;
mfem::Mesh &mesh = *(ofespace.GetMesh());
if (!mesh.GetNodes())
{
mesh.SetCurvature(1, false, -1, mfem::Ordering::byVDIM);
}
mfem::GridFunction &nodes = *(mesh.GetNodes());
const mfem::FiniteElementSpace &fespace = *(nodes.FESpace());
const mfem::FiniteElement &fe = *(fespace.GetFE(0));
const int dims = fe.GetDim();
const int elements = fespace.GetNE();
const int numDofs = fe.GetDof();
const int numQuad = ir.GetNPoints();
MFEM_ASSERT(dims == mesh.SpaceDimension(), "");
geom.meshNodes.allocate(device,
dims, numDofs, elements);
const mfem::Table &e2dTable = fespace.GetElementToDofTable();
const int *elementMap = e2dTable.GetJ();
nodes.Pull();
for (int e = 0; e < elements; ++e)
{
for (int dof = 0; dof < numDofs; ++dof)
{
const int gid = elementMap[dof + numDofs*e];
for (int dim = 0; dim < dims; ++dim)
{
geom.meshNodes(dim, dof, e) = nodes[fespace.DofToVDof(gid,dim)];
}
}
}
geom.meshNodes.keepInDevice();
if (flags & Jacobian)
{
geom.J.allocate(device,
dims, dims, numQuad, elements);
}
else
{
geom.J.allocate(device, 1);
}
if (flags & JacobianInv)
{
geom.invJ.allocate(device,
dims, dims, numQuad, elements);
}
else
{
geom.invJ.allocate(device, 1);
}
if (flags & JacobianDet)
{
geom.detJ.allocate(device,
numQuad, elements);
}
else
{
geom.detJ.allocate(device, 1);
}
geom.J.stopManaging();
geom.invJ.stopManaging();
geom.detJ.stopManaging();
OccaDofQuadMaps &maps = OccaDofQuadMaps::GetSimplexMaps(device, fe, ir);
::occa::properties props;
props["defines/NUM_DOFS"] = numDofs;
props["defines/NUM_QUAD"] = numQuad;
props["defines/STORE_JACOBIAN"] = (flags & Jacobian);
props["defines/STORE_JACOBIAN_INV"] = (flags & JacobianInv);
props["defines/STORE_JACOBIAN_DET"] = (flags & JacobianDet);
const std::string &okl_path = ofespace.OccaEngine().GetOklPath();
const std::string &okl_defines = ofespace.OccaEngine().GetOklDefines();
::occa::kernel init = device.buildKernel(okl_path + "geometry.okl",
stringWithDim("InitGeometryInfo",
fe.GetDim()),
props + okl_defines);
init(elements,
maps.dofToQuadD,
geom.meshNodes,
geom.J, geom.invJ, geom.detJ);
return geom;
}
OccaDofQuadMaps::OccaDofQuadMaps() :
hash() {}
OccaDofQuadMaps::OccaDofQuadMaps(const OccaDofQuadMaps &maps)
{
*this = maps;
}
OccaDofQuadMaps& OccaDofQuadMaps::operator = (const OccaDofQuadMaps &maps)
{
hash = maps.hash;
dofToQuad = maps.dofToQuad;
dofToQuadD = maps.dofToQuadD;
quadToDof = maps.quadToDof;
quadToDofD = maps.quadToDofD;
quadWeights = maps.quadWeights;
return *this;
}
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
const FiniteElementSpace &fespace,
const mfem::IntegrationRule &ir,
const bool transpose)
{
return Get(device,
*fespace.GetFE(0),
*fespace.GetFE(0),
ir,
transpose);
}
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
const mfem::FiniteElement &fe,
const mfem::IntegrationRule &ir,
const bool transpose)
{
return Get(device, fe, fe, ir, transpose);
}
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
const FiniteElementSpace &trialFESpace,
const FiniteElementSpace &testFESpace,
const mfem::IntegrationRule &ir,
const bool transpose)
{
return Get(device,
*trialFESpace.GetFE(0),
*testFESpace.GetFE(0),
ir,
transpose);
}
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
const mfem::FiniteElement &trialFE,
const mfem::FiniteElement &testFE,
const mfem::IntegrationRule &ir,
const bool transpose)
{
return (dynamic_cast<const mfem::TensorBasisElement*>(&trialFE)
? GetTensorMaps(device, trialFE, testFE, ir, transpose)
: GetSimplexMaps(device, trialFE, testFE, ir, transpose));
}
OccaDofQuadMaps& OccaDofQuadMaps::GetTensorMaps(::occa::device device,
const mfem::FiniteElement &fe,
const mfem::IntegrationRule &ir,
const bool transpose)
{
return GetTensorMaps(device,
fe, fe,
ir, transpose);
}
OccaDofQuadMaps& OccaDofQuadMaps::GetTensorMaps(::occa::device device,
const mfem::FiniteElement &trialFE,
const mfem::FiniteElement &testFE,
const mfem::IntegrationRule &ir,
const bool transpose)
{
const mfem::TensorBasisElement &trialTFE =
dynamic_cast<const mfem::TensorBasisElement&>(trialFE);
const mfem::TensorBasisElement &testTFE =
dynamic_cast<const mfem::TensorBasisElement&>(testFE);
std::stringstream ss;
ss << ::occa::hash(device)
<< "Tensor"
<< "O1:" << trialFE.GetOrder()
<< "O2:" << testFE.GetOrder()
<< "BT1:" << trialTFE.GetBasisType()
<< "BT2:" << testTFE.GetBasisType()
<< "Q:" << ir.GetNPoints();
std::string hash = ss.str();
// If we've already made the dof-quad maps, reuse them
OccaDofQuadMaps &maps = AllDofQuadMaps[hash];
if (!maps.hash.size())
{
// Create the dof-quad maps
maps.hash = hash;
OccaDofQuadMaps trialMaps = GetD2QTensorMaps(device, trialFE, ir);
OccaDofQuadMaps testMaps = GetD2QTensorMaps(device, testFE , ir, true);
maps.dofToQuad = trialMaps.dofToQuad;
maps.dofToQuadD = trialMaps.dofToQuadD;
maps.quadToDof = testMaps.dofToQuad;
maps.quadToDofD = testMaps.dofToQuadD;
maps.quadWeights = testMaps.quadWeights;
}
return maps;
}
OccaDofQuadMaps OccaDofQuadMaps::GetD2QTensorMaps(::occa::device device,
const mfem::FiniteElement &fe,
const mfem::IntegrationRule &ir,
const bool transpose)
{
const mfem::TensorBasisElement &tfe =
dynamic_cast<const mfem::TensorBasisElement&>(fe);
const mfem::Poly_1D::Basis &basis = tfe.GetBasis1D();
const int order = fe.GetOrder();
// [MISSING] Get 1D dofs
const int dofs = order + 1;
const int dims = fe.GetDim();
// Create the dof -> quadrature point map
const mfem::IntegrationRule &ir1D =
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir.GetOrder());
const int quadPoints = ir1D.GetNPoints();
const int quadPoints2D = quadPoints*quadPoints;
const int quadPoints3D = quadPoints2D*quadPoints;
const int quadPointsND = ((dims == 1) ? quadPoints :
((dims == 2) ? quadPoints2D : quadPoints3D));
OccaDofQuadMaps maps;
// Initialize the dof -> quad mapping
maps.dofToQuad.allocate(device,
quadPoints, dofs);
maps.dofToQuadD.allocate(device,
quadPoints, dofs);
double *quadWeights1DData = NULL;
if (transpose)
{
maps.dofToQuad.reindex(1,0);
maps.dofToQuadD.reindex(1,0);
// Initialize quad weights only for transpose
maps.quadWeights.allocate(device,
quadPointsND);
quadWeights1DData = new double[quadPoints];
}
mfem::Vector d2q(dofs);
mfem::Vector d2qD(dofs);
for (int q = 0; q < quadPoints; ++q)
{
const mfem::IntegrationPoint &ip = ir1D.IntPoint(q);
basis.Eval(ip.x, d2q, d2qD);
if (transpose)
{
quadWeights1DData[q] = ip.weight;
}
for (int d = 0; d < dofs; ++d)
{
maps.dofToQuad(q, d) = d2q[d];
maps.dofToQuadD(q, d) = d2qD[d];
}
}
maps.dofToQuad.keepInDevice();
maps.dofToQuadD.keepInDevice();
if (transpose)
{
for (int q = 0; q < quadPointsND; ++q)
{
const int qx = q % quadPoints;
const int qz = q / quadPoints2D;
const int qy = (q - qz*quadPoints2D) / quadPoints;
double w = quadWeights1DData[qx];
if (dims > 1)
{
w *= quadWeights1DData[qy];
}
if (dims > 2)
{
w *= quadWeights1DData[qz];
}
maps.quadWeights[q] = w;
}
maps.quadWeights.keepInDevice();
delete [] quadWeights1DData;
}
return maps;
}
OccaDofQuadMaps& OccaDofQuadMaps::GetSimplexMaps(::occa::device device,
const mfem::FiniteElement &fe,
const mfem::IntegrationRule &ir,
const bool transpose)
{
return GetSimplexMaps(device,
fe, fe,
ir, transpose);
}
OccaDofQuadMaps& OccaDofQuadMaps::GetSimplexMaps(::occa::device device,
const mfem::FiniteElement &trialFE,
const mfem::FiniteElement &testFE,
const mfem::IntegrationRule &ir,
const bool transpose)
{
std::stringstream ss;
ss << ::occa::hash(device)
<< "Simplex"
<< "O1:" << trialFE.GetOrder()
<< "O2:" << testFE.GetOrder()
<< "Q:" << ir.GetNPoints();
std::string hash = ss.str();
// If we've already made the dof-quad maps, reuse them
OccaDofQuadMaps &maps = AllDofQuadMaps[hash];
if (!maps.hash.size())
{
// Create the dof-quad maps
maps.hash = hash;
OccaDofQuadMaps trialMaps = GetD2QSimplexMaps(device, trialFE, ir);
OccaDofQuadMaps testMaps = GetD2QSimplexMaps(device, testFE , ir, true);
maps.dofToQuad = trialMaps.dofToQuad;
maps.dofToQuadD = trialMaps.dofToQuadD;
maps.quadToDof = testMaps.dofToQuad;
maps.quadToDofD = testMaps.dofToQuadD;
maps.quadWeights = testMaps.quadWeights;
}
return maps;
}
OccaDofQuadMaps OccaDofQuadMaps::GetD2QSimplexMaps(::occa::device device,
const mfem::FiniteElement &fe,
const mfem::IntegrationRule &ir,
const bool transpose)
{
const int dims = fe.GetDim();
const int numDofs = fe.GetDof();
const int numQuad = ir.GetNPoints();
OccaDofQuadMaps maps;
// Initialize the dof -> quad mapping
maps.dofToQuad.allocate(device,
numQuad, numDofs);
maps.dofToQuadD.allocate(device,
dims, numQuad, numDofs);
if (transpose)
{
maps.dofToQuad.reindex(1,0);
maps.dofToQuadD.reindex(1,0);
// Initialize quad weights only for transpose
maps.quadWeights.allocate(device,
numQuad);
}
mfem::Vector d2q(numDofs);
mfem::DenseMatrix d2qD(numDofs, dims);
for (int q = 0; q < numQuad; ++q)
{
const mfem::IntegrationPoint &ip = ir.IntPoint(q);
if (transpose)
{
maps.quadWeights[q] = ip.weight;
}
fe.CalcShape(ip, d2q);
fe.CalcDShape(ip, d2qD);
for (int d = 0; d < numDofs; ++d)
{
const double w = d2q[d];
maps.dofToQuad(q, d) = w;
for (int dim = 0; dim < dims; ++dim)
{
const double wD = d2qD(d, dim);
maps.dofToQuadD(dim, q, d) = wD;
}
}
}
maps.dofToQuad.keepInDevice();
maps.dofToQuadD.keepInDevice();
if (transpose)
{
maps.quadWeights.keepInDevice();
}
return maps;
}
//---[ Integrator Defines ]-----------
std::string stringWithDim(const std::string &s, const int dim)
{
std::string ret = s;
ret += ('0' + (char) dim);
ret += 'D';
return ret;
}
int closestWarpBatchTo(const int value)
{
return ((value + 31) / 32) * 32;
}
int closestMultipleWarpBatch(const int multiple, const int maxSize)
{
if (multiple > maxSize)
{
return maxSize;
}
int batch = (32 / multiple);
int minDiff = 32 - (multiple * batch);
for (int i = 64; i <= maxSize; i += 32)
{
const int newDiff = i - (multiple * (i / multiple));
if (newDiff < minDiff)
{
batch = (i / multiple);
minDiff = newDiff;
}
}
return batch;
}
void SetProperties(FiniteElementSpace &fespace,
const mfem::IntegrationRule &ir,
::occa::properties &props)
{
SetProperties(fespace, fespace, ir, props);
}
void SetProperties(FiniteElementSpace &trialFESpace,
FiniteElementSpace &testFESpace,
const mfem::IntegrationRule &ir,
::occa::properties &props)
{
props["defines/TRIAL_VDIM"] = trialFESpace.GetVDim();
props["defines/TEST_VDIM"] = testFESpace.GetVDim();
props["defines/NUM_DIM"] = trialFESpace.GetDim();
if (trialFESpace.hasTensorBasis())
{
SetTensorProperties(trialFESpace, testFESpace, ir, props);
}
else
{
SetSimplexProperties(trialFESpace, testFESpace, ir, props);
}
}
void SetTensorProperties(FiniteElementSpace &fespace,
const mfem::IntegrationRule &ir,
::occa::properties &props)
{
SetTensorProperties(fespace, fespace, ir, props);
}
void SetTensorProperties(FiniteElementSpace &trialFESpace,
FiniteElementSpace &testFESpace,
const mfem::IntegrationRule &ir,
::occa::properties &props)
{
const mfem::FiniteElement &trialFE = *(trialFESpace.GetFE(0));
const mfem::FiniteElement &testFE = *(testFESpace.GetFE(0));
const mfem::IntegrationRule &ir1D =
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir.GetOrder());
const int trialDofs = trialFE.GetDof();
const int testDofs = testFE.GetDof();
const int numQuad = ir.GetNPoints();
const int trialDofs1D = trialFE.GetOrder() + 1;
const int testDofs1D = testFE.GetOrder() + 1;
const int quad1D = ir1D.GetNPoints();
int trialDofsND = trialDofs1D;
int testDofsND = testDofs1D;
int quadND = quad1D;
const bool trialByVDIM = (trialFESpace.GetOrdering() == mfem::Ordering::byVDIM);
const bool testByVDIM = (testFESpace.GetOrdering() == mfem::Ordering::byVDIM);
props["defines/ORDERING_BY_NODES"] = 0;
props["defines/ORDERING_BY_VDIM"] = 1;
props["defines/VDIM_ORDERING"] = (int) trialByVDIM;
props["defines/TRIAL_ORDERING"] = (int) trialByVDIM;
props["defines/TEST_ORDERING"] = (int) testByVDIM;
props["defines/USING_TENSOR_OPS"] = 1;
props["defines/NUM_DOFS"] = trialDofs;
props["defines/NUM_QUAD"] = numQuad;
props["defines/TRIAL_DOFS"] = trialDofs;
props["defines/TEST_DOFS"] = testDofs;
for (int d = 1; d <= 3; ++d)
{
if (d > 1)
{
trialDofsND *= trialDofs1D;
testDofsND *= testDofs1D;
quadND *= quad1D;
}
props["defines"][stringWithDim("NUM_DOFS_", d)] = trialDofsND;
props["defines"][stringWithDim("NUM_QUAD_", d)] = quadND;
props["defines"][stringWithDim("TRIAL_DOFS_", d)] = trialDofsND;
props["defines"][stringWithDim("TEST_DOFS_" , d)] = testDofsND;
}
// 1D Defines
const int m1InnerBatch = 32 * ((quad1D + 31) / 32);
props["defines/A1_ELEMENT_BATCH"] = closestMultipleWarpBatch(quad1D, 512);
props["defines/M1_OUTER_ELEMENT_BATCH"] = closestMultipleWarpBatch(m1InnerBatch,
512);
props["defines/M1_INNER_ELEMENT_BATCH"] = m1InnerBatch;
// 2D Defines
props["defines/A2_ELEMENT_BATCH"] = 1;
props["defines/A2_QUAD_BATCH"] = 1;
props["defines/M2_ELEMENT_BATCH"] = 32;
// 3D Defines
const int a3QuadBatch = closestMultipleWarpBatch(quadND, 512);
props["defines/A3_ELEMENT_BATCH"] = closestMultipleWarpBatch(a3QuadBatch, 512);
props["defines/A3_QUAD_BATCH"] = a3QuadBatch;
}
void SetSimplexProperties(FiniteElementSpace &fespace,
const mfem::IntegrationRule &ir,
::occa::properties &props)
{
SetSimplexProperties(fespace, fespace, ir, props);
}
void SetSimplexProperties(FiniteElementSpace &trialFESpace,
FiniteElementSpace &testFESpace,
const mfem::IntegrationRule &ir,
::occa::properties &props)
{
const mfem::FiniteElement &trialFE = *(trialFESpace.GetFE(0));
const mfem::FiniteElement &testFE = *(testFESpace.GetFE(0));
const int trialDofs = trialFE.GetDof();
const int testDofs = testFE.GetDof();
const int numQuad = ir.GetNPoints();
const int maxDQ = std::max(std::max(trialDofs, testDofs), numQuad);
const bool trialByVDIM = (trialFESpace.GetOrdering() == mfem::Ordering::byVDIM);
const bool testByVDIM = (testFESpace.GetOrdering() == mfem::Ordering::byVDIM);
props["defines/ORDERING_BY_NODES"] = 0;
props["defines/ORDERING_BY_VDIM"] = 1;
props["defines/VDIM_ORDERING"] = (int) trialByVDIM;
props["defines/TRIAL_ORDERING"] = (int) trialByVDIM;
props["defines/TEST_ORDERING"] = (int) testByVDIM;
props["defines/USING_TENSOR_OPS"] = 0;
props["defines/NUM_DOFS"] = trialDofs;
props["defines/NUM_QUAD"] = numQuad;
props["defines/TRIAL_DOFS"] = trialDofs;
props["defines/TEST_DOFS"] = testDofs;
// 2D Defines
const int quadBatch = closestWarpBatchTo(numQuad);
props["defines/A2_ELEMENT_BATCH"] = closestMultipleWarpBatch(quadBatch, 2048);
props["defines/A2_QUAD_BATCH"] = quadBatch;
props["defines/M2_INNER_BATCH"] = closestWarpBatchTo(maxDQ);
// 3D Defines
props["defines/A3_ELEMENT_BATCH"] = closestMultipleWarpBatch(quadBatch, 2048);
props["defines/A3_QUAD_BATCH"] = quadBatch;
props["defines/M3_INNER_BATCH"] = closestWarpBatchTo(maxDQ);
}
//---[ Base Integrator ]--------------
OccaIntegrator::OccaIntegrator(const Engine &e)
: engine(&e),
bform(),
mesh(),
otrialFESpace(),
otestFESpace(),
trialFESpace(),
testFESpace(),
itype(DomainIntegrator),
ir(NULL),
hasTensorBasis(false) { }
OccaIntegrator::~OccaIntegrator() {}
void OccaIntegrator::SetupMaps()
{
maps = OccaDofQuadMaps::Get(GetDevice(),
*otrialFESpace,
*otestFESpace,
*ir);
mapsTranspose = OccaDofQuadMaps::Get(GetDevice(),
*otestFESpace,
*otrialFESpace,
*ir);
}
FiniteElementSpace& OccaIntegrator::GetTrialOccaFESpace() const
{
return *otrialFESpace;
}
FiniteElementSpace& OccaIntegrator::GetTestOccaFESpace() const
{
return *otestFESpace;
}
mfem::FiniteElementSpace& OccaIntegrator::GetTrialFESpace() const
{
return *trialFESpace;
}
mfem::FiniteElementSpace& OccaIntegrator::GetTestFESpace() const
{
return *testFESpace;
}
void OccaIntegrator::SetIntegrationRule(const mfem::IntegrationRule &ir_)
{
ir = &ir_;
}
const mfem::IntegrationRule& OccaIntegrator::GetIntegrationRule() const
{
return *ir;
}
OccaDofQuadMaps& OccaIntegrator::GetDofQuadMaps()
{
return maps;
}
void OccaIntegrator::SetupIntegrator(OccaBilinearForm &bform_,
const ::occa::properties &props_,
const OccaIntegratorType itype_)
{
MFEM_ASSERT(engine == &bform_.OccaEngine(), "");
bform = &bform_;
mesh = &(bform_.GetMesh());
otrialFESpace = &(bform_.GetTrialOccaFESpace());
otestFESpace = &(bform_.GetTestOccaFESpace());
trialFESpace = &(bform_.GetTrialFESpace());
testFESpace = &(bform_.GetTestFESpace());
hasTensorBasis = otrialFESpace->hasTensorBasis();
props = props_;
itype = itype_;
if (ir == NULL)
{
SetupIntegrationRule();
}
SetupMaps();
SetProperties(*otrialFESpace,
*otestFESpace,
*ir,
props);
Setup();
}
OccaGeometry OccaIntegrator::GetGeometry(const int flags)
{
return OccaGeometry::Get(GetDevice(), *otrialFESpace, *ir, flags);
}
::occa::kernel OccaIntegrator::GetAssembleKernel(const ::occa::properties
&props)
{
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
return GetKernel(stringWithDim("Assemble", fe.GetDim()),
props);
}
::occa::kernel OccaIntegrator::GetMultAddKernel(const ::occa::properties &props)
{
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
return GetKernel(stringWithDim("MultAdd", fe.GetDim()),
props);
}
::occa::kernel OccaIntegrator::GetKernel(const std::string &kernelName,
const ::occa::properties &props)
{
const std::string filename = GetName() + ".okl";
const std::string &okl_path = OccaEngine().GetOklPath();
const std::string &okl_defines = OccaEngine().GetOklDefines();
return GetDevice().buildKernel(okl_path + filename,
kernelName,
props + okl_defines);
}
//====================================
//---[ Diffusion Integrator ]---------
OccaDiffusionIntegrator::OccaDiffusionIntegrator(const OccaCoefficient &coeff_)
:
OccaIntegrator(coeff_.OccaEngine()),
coeff(coeff_),
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
{
coeff.SetName("COEFF");
}
OccaDiffusionIntegrator::~OccaDiffusionIntegrator() {}
std::string OccaDiffusionIntegrator::GetName()
{
return "DiffusionIntegrator";
}
void OccaDiffusionIntegrator::SetupIntegrationRule()
{
const FiniteElement &trialFE = *(trialFESpace->GetFE(0));
const FiniteElement &testFE = *(testFESpace->GetFE(0));
ir = &mfem::DiffusionIntegrator::GetRule(trialFE, testFE);
}
void OccaDiffusionIntegrator::Setup()
{
::occa::properties kernelProps = props;
coeff.Setup(*this, kernelProps);
// Setup assemble and mult kernels
assembleKernel = GetAssembleKernel(kernelProps);
multKernel = GetMultAddKernel(kernelProps);
}
void OccaDiffusionIntegrator::Assemble()
{
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
const int dims = fe.GetDim();
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
const int elements = trialFESpace->GetNE();
const int quadraturePoints = ir->GetNPoints();
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
assembledOperator.Resize<double>(symmDims * quadraturePoints * elements,
NULL);
assembleKernel((int) mesh->GetNE(),
maps.quadWeights,
geom.J,
coeff,
assembledOperator.OccaMem());
}
void OccaDiffusionIntegrator::MultAdd(Vector &x, Vector &y)
{
// Note: x and y are E-vectors
multKernel((int) mesh->GetNE(),
maps.dofToQuad,
maps.dofToQuadD,
maps.quadToDof,
maps.quadToDofD,
assembledOperator.OccaMem(),
x.OccaMem(), y.OccaMem());
}
//====================================
//---[ Mass Integrator ]--------------
OccaMassIntegrator::OccaMassIntegrator(const OccaCoefficient &coeff_) :
OccaIntegrator(coeff_.OccaEngine()),
coeff(coeff_),
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
{
coeff.SetName("COEFF");
}
OccaMassIntegrator::~OccaMassIntegrator() {}
std::string OccaMassIntegrator::GetName()
{
return "MassIntegrator";
}
void OccaMassIntegrator::SetupIntegrationRule()
{
const mfem::FiniteElement &trialFE = *(trialFESpace->GetFE(0));
const mfem::FiniteElement &testFE = *(testFESpace->GetFE(0));
mfem::ElementTransformation &T = *trialFESpace->GetElementTransformation(0);
ir = &mfem::MassIntegrator::GetRule(trialFE, testFE, T);
}
void OccaMassIntegrator::Setup()
{
::occa::properties kernelProps = props;
coeff.Setup(*this, kernelProps);
// Setup assemble and mult kernels
assembleKernel = GetAssembleKernel(kernelProps);
multKernel = GetMultAddKernel(kernelProps);
}
void OccaMassIntegrator::Assemble()
{
if (assembledOperator.Size())
{
return;
}
const int elements = trialFESpace->GetNE();
const int quadraturePoints = ir->GetNPoints();
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
assembledOperator.Resize<double>(quadraturePoints * elements, NULL);
assembleKernel((int) mesh->GetNE(),
maps.quadWeights,
geom.J,
coeff,
assembledOperator.OccaMem());
}
void OccaMassIntegrator::SetOperator(Vector &v)
{
assembledOperator = v;
}
void OccaMassIntegrator::MultAdd(Vector &x, Vector &y)
{
multKernel((int) mesh->GetNE(),
maps.dofToQuad,
maps.dofToQuadD,
maps.quadToDof,
maps.quadToDofD,
assembledOperator.OccaMem(),
x.OccaMem(), y.OccaMem());
}
//====================================
//---[ Vector Mass Integrator ]--------------
OccaVectorMassIntegrator::OccaVectorMassIntegrator(const OccaCoefficient &
coeff_)
:
OccaIntegrator(coeff_.OccaEngine()),
coeff(coeff_),
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
{
coeff.SetName("COEFF");
}
OccaVectorMassIntegrator::~OccaVectorMassIntegrator() {}
std::string OccaVectorMassIntegrator::GetName()
{
return "VectorMassIntegrator";
}
void OccaVectorMassIntegrator::SetupIntegrationRule()
{
const mfem::FiniteElement &trialFE = *(trialFESpace->GetFE(0));
const mfem::FiniteElement &testFE = *(testFESpace->GetFE(0));
mfem::ElementTransformation &T = *trialFESpace->GetElementTransformation(0);
ir = &mfem::MassIntegrator::GetRule(trialFE, testFE, T);
}
void OccaVectorMassIntegrator::Setup()
{
::occa::properties kernelProps = props;
coeff.Setup(*this, kernelProps);
// Setup assemble and mult kernels
assembleKernel = GetAssembleKernel(kernelProps);
multKernel = GetMultAddKernel(kernelProps);
}
void OccaVectorMassIntegrator::Assemble()
{
const int elements = trialFESpace->GetNE();
const int quadraturePoints = ir->GetNPoints();
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
assembledOperator.Resize<double>(quadraturePoints * elements, NULL);
assembleKernel((int) mesh->GetNE(),
maps.quadWeights,
geom.J,
coeff,
assembledOperator.OccaMem());
}
void OccaVectorMassIntegrator::MultAdd(Vector &x, Vector &y)
{
multKernel((int) mesh->GetNE(),
maps.dofToQuad,
maps.dofToQuadD,
maps.quadToDof,
maps.quadToDofD,
assembledOperator.OccaMem(),
x.OccaMem(), y.OccaMem());
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-323
View File
@@ -1,323 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
#define MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "fespace.hpp"
#include "bilinearform.hpp"
#include "coefficient.hpp"
namespace mfem
{
namespace occa
{
class OccaGeometry
{
public:
::occa::array<double> meshNodes;
::occa::array<double> J, invJ, detJ;
// byVDIM -> [x y z x y z x y z]
// byNodes -> [x x x y y y z z z]
static const int Jacobian = (1 << 0);
static const int JacobianInv = (1 << 1);
static const int JacobianDet = (1 << 2);
static OccaGeometry Get(::occa::device device,
FiniteElementSpace &ofespace,
const IntegrationRule &ir,
const int flags = (Jacobian |
JacobianInv |
JacobianDet));
};
class OccaDofQuadMaps
{
private:
// Reuse dof-quad maps
static std::map<std::string, OccaDofQuadMaps> AllDofQuadMaps;
std::string hash;
public:
// Local stiffness matrices (B and B^T operators)
::occa::array<double, ::occa::dynamic> dofToQuad, dofToQuadD; // B
::occa::array<double, ::occa::dynamic> quadToDof, quadToDofD; // B^T
::occa::array<double> quadWeights;
OccaDofQuadMaps();
OccaDofQuadMaps(const OccaDofQuadMaps &maps);
OccaDofQuadMaps& operator = (const OccaDofQuadMaps &maps);
// [[x y] [x y] [x y]]
// [[x y z] [x y z] [x y z]]
// mfem::GridFunction* mfem::Mesh::GetNodes() { return Nodes; }
// mfem::FiniteElementSpace *Nodes->FESpace()
// 25
// 1D [x x x x x x]
// 2D [x y x y x y]
// GetVdim()
// 3D ordering == byVDIM -> [x y z x y z x y z x y z x y z x y z]
// ordering == byNODES -> [x x x x x x y y y y y y z z z z z z]
static OccaDofQuadMaps& Get(::occa::device device,
const FiniteElementSpace &fespace,
const mfem::IntegrationRule &ir,
const bool transpose = false);
static OccaDofQuadMaps& Get(::occa::device device,
const mfem::FiniteElement &fe,
const mfem::IntegrationRule &ir,
const bool transpose = false);
static OccaDofQuadMaps& Get(::occa::device device,
const FiniteElementSpace &trialFESpace,
const FiniteElementSpace &testFESpace,
const mfem::IntegrationRule &ir,
const bool transpose = false);
static OccaDofQuadMaps& Get(::occa::device device,
const mfem::FiniteElement &trialFE,
const mfem::FiniteElement &testFE,
const mfem::IntegrationRule &ir,
const bool transpose = false);
static OccaDofQuadMaps& GetTensorMaps(::occa::device device,
const mfem::FiniteElement &fe,
const mfem::IntegrationRule &ir,
const bool transpose = false);
static OccaDofQuadMaps& GetTensorMaps(::occa::device device,
const mfem::FiniteElement &trialFE,
const mfem::FiniteElement &testFE,
const mfem::IntegrationRule &ir,
const bool transpose = false);
static OccaDofQuadMaps GetD2QTensorMaps(::occa::device device,
const mfem::FiniteElement &fe,
const mfem::IntegrationRule &ir,
const bool transpose = false);
static OccaDofQuadMaps& GetSimplexMaps(::occa::device device,
const mfem::FiniteElement &fe,
const mfem::IntegrationRule &ir,
const bool transpose = false);
static OccaDofQuadMaps& GetSimplexMaps(::occa::device device,
const mfem::FiniteElement &trialFE,
const mfem::FiniteElement &testFE,
const mfem::IntegrationRule &ir,
const bool transpose = false);
static OccaDofQuadMaps GetD2QSimplexMaps(::occa::device device,
const mfem::FiniteElement &fe,
const mfem::IntegrationRule &ir,
const bool transpose = false);
};
//---[ Define Methods ]---------------
std::string stringWithDim(const std::string &s, const int dim);
int closestWarpBatch(const int multiple, const int maxSize);
void SetProperties(FiniteElementSpace &fespace,
const mfem::IntegrationRule &ir,
::occa::properties &props);
void SetProperties(FiniteElementSpace &trialFESpace,
FiniteElementSpace &testFESpace,
const mfem::IntegrationRule &ir,
::occa::properties &props);
void SetTensorProperties(FiniteElementSpace &fespace,
const mfem::IntegrationRule &ir,
::occa::properties &props);
void SetTensorProperties(FiniteElementSpace &trialFESpace,
FiniteElementSpace &testFESpace,
const IntegrationRule &ir,
::occa::properties &props);
void SetSimplexProperties(FiniteElementSpace &fespace,
const IntegrationRule &ir,
::occa::properties &props);
void SetSimplexProperties(FiniteElementSpace &trialFESpace,
FiniteElementSpace &testFESpace,
const IntegrationRule &ir,
::occa::properties &props);
//---[ Base Integrator ]--------------
class OccaIntegrator
{
protected:
SharedPtr<const Engine> engine;
OccaBilinearForm *bform;
mfem::Mesh *mesh;
FiniteElementSpace *otrialFESpace;
FiniteElementSpace *otestFESpace;
mfem::FiniteElementSpace *trialFESpace;
mfem::FiniteElementSpace *testFESpace;
::occa::properties props;
OccaIntegratorType itype;
const IntegrationRule *ir;
bool hasTensorBasis;
OccaDofQuadMaps maps;
OccaDofQuadMaps mapsTranspose;
public:
OccaIntegrator(const Engine &e);
virtual ~OccaIntegrator();
const Engine &OccaEngine() const { return *engine; }
::occa::device GetDevice(int idx = 0) const
{ return engine->GetDevice(idx); }
virtual std::string GetName() = 0;
FiniteElementSpace& GetTrialOccaFESpace() const;
FiniteElementSpace& GetTestOccaFESpace() const;
mfem::FiniteElementSpace& GetTrialFESpace() const;
mfem::FiniteElementSpace& GetTestFESpace() const;
void SetIntegrationRule(const mfem::IntegrationRule &ir_);
const mfem::IntegrationRule& GetIntegrationRule() const;
OccaDofQuadMaps& GetDofQuadMaps();
void SetupMaps();
virtual void SetupIntegrationRule() = 0;
virtual void SetupIntegrator(OccaBilinearForm &bform_,
const ::occa::properties &props_,
const OccaIntegratorType itype_);
virtual void Setup() = 0;
virtual void Assemble() = 0;
/// This method works on E-vectors!
virtual void MultAdd(Vector &x, Vector &y) = 0;
virtual void MultTransposeAdd(Vector &x, Vector &y)
{
mfem_error("OccaIntegrator::MultTransposeAdd() is not overloaded!");
}
OccaGeometry GetGeometry(const int flags = (OccaGeometry::Jacobian |
OccaGeometry::JacobianInv |
OccaGeometry::JacobianDet));
::occa::kernel GetAssembleKernel(const ::occa::properties &props);
::occa::kernel GetMultAddKernel(const ::occa::properties &props);
::occa::kernel GetKernel(const std::string &kernelName,
const ::occa::properties &props);
};
//====================================
//---[ Diffusion Integrator ]---------
class OccaDiffusionIntegrator : public OccaIntegrator
{
private:
OccaCoefficient coeff;
::occa::kernel assembleKernel, multKernel;
Vector assembledOperator;
public:
OccaDiffusionIntegrator(const OccaCoefficient &coeff_);
virtual ~OccaDiffusionIntegrator();
virtual std::string GetName();
virtual void SetupIntegrationRule();
virtual void Setup();
virtual void Assemble();
virtual void MultAdd(Vector &x, Vector &y);
};
//====================================
//---[ Mass Integrator ]--------------
class OccaMassIntegrator : public OccaIntegrator
{
private:
OccaCoefficient coeff;
::occa::kernel assembleKernel, multKernel;
Vector assembledOperator;
public:
OccaMassIntegrator(const OccaCoefficient &coeff_);
virtual ~OccaMassIntegrator();
virtual std::string GetName();
virtual void SetupIntegrationRule();
virtual void Setup();
virtual void Assemble();
void SetOperator(Vector &v);
virtual void MultAdd(Vector &x, Vector &y);
};
//====================================
//---[ Vector Mass Integrator ]--------------
class OccaVectorMassIntegrator : public OccaIntegrator
{
private:
OccaCoefficient coeff;
::occa::kernel assembleKernel, multKernel;
Vector assembledOperator;
public:
OccaVectorMassIntegrator(const OccaCoefficient &coeff_);
virtual ~OccaVectorMassIntegrator();
virtual std::string GetName();
virtual void SetupIntegrationRule();
virtual void Setup();
virtual void Assemble();
virtual void MultAdd(Vector &x, Vector &y);
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
-344
View File
@@ -1,344 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "coefficient.hpp"
#include "bilininteg.hpp"
namespace mfem
{
namespace occa
{
//---[ Parameter ]------------
OccaParameter::~OccaParameter() {}
void OccaParameter::Setup(OccaIntegrator &integ,
::occa::properties &props) {}
::occa::kernelArg OccaParameter::KernelArgs()
{
return ::occa::kernelArg();
}
//====================================
//---[ Include Parameter ]------------
OccaIncludeParameter::OccaIncludeParameter(const std::string &filename_) :
filename(filename_) {}
OccaParameter* OccaIncludeParameter::Clone()
{
return new OccaIncludeParameter(filename);
}
void OccaIncludeParameter::Setup(OccaIntegrator &integ,
::occa::properties &props)
{
props["headers"].asArray() += "#include " + filename;
}
//====================================
//---[ Source Parameter ]------------
OccaSourceParameter::OccaSourceParameter(const std::string &source_) :
source(source_) {}
OccaParameter* OccaSourceParameter::Clone()
{
return new OccaSourceParameter(source);
}
void OccaSourceParameter::Setup(OccaIntegrator &integ,
::occa::properties &props)
{
props["headers"].asArray() += source;
}
//====================================
//---[ Vector Parameter ]-------
OccaVectorParameter::OccaVectorParameter(const std::string &name_,
Vector &v_,
const bool useRestrict_) :
name(name_),
v(v_),
useRestrict(useRestrict_),
attr("") {}
OccaVectorParameter::OccaVectorParameter(const std::string &name_,
Vector &v_,
const std::string &attr_,
const bool useRestrict_) :
name(name_),
v(v_),
useRestrict(useRestrict_),
attr(attr_) {}
OccaParameter* OccaVectorParameter::Clone()
{
return new OccaVectorParameter(name, v, attr, useRestrict);
}
void OccaVectorParameter::Setup(OccaIntegrator &integ,
::occa::properties &props)
{
std::string &args = (props["defines/COEFF_ARGS"]
.asString()
.string());
args += "const double *";
if (useRestrict)
{
args += " restrict ";
}
args += name;
if (attr.size())
{
args += ' ';
args += attr;
}
args += ",\n";
}
::occa::kernelArg OccaVectorParameter::KernelArgs()
{
return ::occa::kernelArg(v.OccaMem());
}
//====================================
//---[ GridFunction Parameter ]-------
OccaGridFunctionParameter::OccaGridFunctionParameter(const std::string &name_,
OccaGridFunction &gf_,
const bool useRestrict_)
: name(name_),
gf(gf_),
gfQuad(*(new Layout(gf_.OccaLayout().OccaEngine(), 0))),
useRestrict(useRestrict_) {}
OccaParameter* OccaGridFunctionParameter::Clone()
{
OccaGridFunctionParameter *param =
new OccaGridFunctionParameter(name, gf, useRestrict);
param->gfQuad.MakeRef(gfQuad);
return param;
}
void OccaGridFunctionParameter::Setup(OccaIntegrator &integ,
::occa::properties &props)
{
std::string &args = (props["defines/COEFF_ARGS"]
.asString()
.string());
args += "const double *";
if (useRestrict)
{
args += " restrict ";
}
args += name;
args += " @dim(NUM_QUAD, numElements),\n";
gf.ToQuad(integ.GetIntegrationRule(), gfQuad);
}
::occa::kernelArg OccaGridFunctionParameter::KernelArgs()
{
return gfQuad.OccaMem();
}
//====================================
//---[ Coefficient ]------------------
OccaCoefficient::OccaCoefficient(const Engine &e, const double value) :
engine(&e),
integ(NULL),
name("COEFF")
{
coeffValue = value;
}
OccaCoefficient::OccaCoefficient(const Engine &e, const std::string &source) :
engine(&e),
integ(NULL),
name("COEFF")
{
coeffValue = source;
}
OccaCoefficient::OccaCoefficient(const Engine &e, const char *source) :
engine(&e),
integ(NULL),
name("COEFF")
{
coeffValue = source;
}
OccaCoefficient::OccaCoefficient(const OccaCoefficient &coeff) :
engine(coeff.engine),
integ(NULL),
name(coeff.name),
coeffValue(coeff.coeffValue)
{
const int paramCount = (int) coeff.params.size();
for (int i = 0; i < paramCount; ++i)
{
params.push_back(coeff.params[i]->Clone());
}
}
OccaCoefficient::~OccaCoefficient()
{
const int paramCount = (int) params.size();
for (int i = 0; i < paramCount; ++i)
{
delete params[i];
}
}
OccaCoefficient& OccaCoefficient::SetName(const std::string &name_)
{
name = name_;
return *this;
}
void OccaCoefficient::Setup(OccaIntegrator &integ_,
::occa::properties &props_)
{
integ = &integ_;
const int paramCount = (int) params.size();
props_["defines"][name + "_ARGS"] = "";
for (int i = 0; i < paramCount; ++i)
{
params[i]->Setup(integ_, props_);
}
props_["defines"][name] = coeffValue;
props = props_;
}
OccaCoefficient& OccaCoefficient::Add(OccaParameter *param)
{
params.push_back(param);
return *this;
}
OccaCoefficient& OccaCoefficient::IncludeHeader(const std::string &filename)
{
return Add(new OccaIncludeParameter(filename));
}
OccaCoefficient& OccaCoefficient::IncludeSource(const std::string &source)
{
return Add(new OccaSourceParameter(source));
}
OccaCoefficient& OccaCoefficient::AddVector(const std::string &name_,
Vector &v,
const bool useRestrict)
{
return Add(new OccaVectorParameter(name_, v, useRestrict));
}
OccaCoefficient& OccaCoefficient::AddVector(const std::string &name_,
Vector &v,
const std::string &attr,
const bool useRestrict)
{
return Add(new OccaVectorParameter(name_, v, attr, useRestrict));
}
OccaCoefficient& OccaCoefficient::AddGridFunction(const std::string &name_,
OccaGridFunction &gf,
const bool useRestrict)
{
return Add(new OccaGridFunctionParameter(name_, gf, useRestrict));
}
bool OccaCoefficient::IsConstant()
{
return coeffValue.isNumber();
}
double OccaCoefficient::GetConstantValue()
{
if (!IsConstant())
{
mfem_error("OccaCoefficient is not constant");
}
return coeffValue.number();
}
Vector OccaCoefficient::Eval()
{
if (integ == NULL)
{
mfem_error("OccaCoefficient requires a Setup() call before Eval()");
}
mfem::FiniteElementSpace &fespace = integ->GetTrialFESpace();
const mfem::IntegrationRule &ir = integ->GetIntegrationRule();
const int elements = fespace.GetNE();
const int numQuad = ir.GetNPoints();
Vector quadCoeff(*(new Layout(OccaEngine(), numQuad * elements)));
Eval(quadCoeff);
return quadCoeff;
}
void OccaCoefficient::Eval(Vector &quadCoeff)
{
const std::string &okl_path = OccaEngine().GetOklPath();
const std::string &okl_defines = OccaEngine().GetOklDefines();
static ::occa::kernelBuilder builder =
::occa::kernelBuilder::fromFile(okl_path + "coefficient.okl",
"CoefficientEval", okl_defines);
if (integ == NULL)
{
mfem_error("OccaCoefficient requires a Setup() call before Eval()");
}
const int elements = integ->GetTrialFESpace().GetNE();
::occa::properties kernelProps = props;
if (name != "COEFF")
{
kernelProps["defines/COEFF"] = name;
kernelProps["defines/COEFF_ARGS"] = name + "_ARGS";
}
kernelProps += okl_defines;
::occa::kernel evalKernel = builder.build(GetDevice(), kernelProps);
evalKernel(elements, *this, quadCoeff.OccaMem());
}
OccaCoefficient::operator ::occa::kernelArg ()
{
::occa::kernelArg kArg;
const int paramCount = (int) params.size();
for (int i = 0; i < paramCount; ++i)
{
kArg.add(params[i]->KernelArgs());
}
return kArg;
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-284
View File
@@ -1,284 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
#define MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "vector.hpp"
#include "gridfunc.hpp"
namespace mfem
{
namespace occa
{
class OccaIntegrator;
class OccaParameter
{
public:
virtual ~OccaParameter();
virtual OccaParameter* Clone() = 0;
virtual void Setup(OccaIntegrator &integ,
::occa::properties &props);
virtual ::occa::kernelArg KernelArgs();
};
//---[ Include Parameter ]------------
class OccaIncludeParameter : public OccaParameter
{
private:
std::string filename;
public:
OccaIncludeParameter(const std::string &filename_);
virtual OccaParameter* Clone();
virtual void Setup(OccaIntegrator &integ,
::occa::properties &props);
};
//====================================
//---[ Source Parameter ]------------
class OccaSourceParameter : public OccaParameter
{
private:
std::string source;
public:
OccaSourceParameter(const std::string &filename_);
virtual OccaParameter* Clone();
virtual void Setup(OccaIntegrator &integ,
::occa::properties &props);
};
//====================================
//---[ Define Parameter ]------------
template <class TM>
class OccaDefineParameter : public OccaParameter
{
private:
const std::string name;
TM value;
public:
OccaDefineParameter(const std::string &name_,
const TM &value_) :
name(name_),
value(value_) {}
virtual OccaParameter* Clone()
{
return new OccaDefineParameter(name, value);
}
virtual void Setup(OccaIntegrator &integ,
::occa::properties &props)
{
props["defines"][name] = value;
}
};
//====================================
//---[ Variable Parameter ]-----------
template <class TM>
class OccaVariableParameter : public OccaParameter
{
private:
const std::string name;
const TM &value;
public:
OccaVariableParameter(const std::string &name_,
const TM &value_) :
name(name_),
value(value_) {}
virtual OccaParameter* Clone()
{
return new OccaVariableParameter(name, value);
}
virtual void Setup(OccaIntegrator &integ,
::occa::properties &props)
{
std::string &args = (props["defines/COEFF_ARGS"]
.asString()
.string());
// const TM name,\n"
args += "const ";
args += ::occa::primitiveinfo<TM>::name;
args += ' ';
args += name;
args += ",\n";
}
virtual ::occa::kernelArg KernelArgs()
{
return ::occa::kernelArg(value);
}
};
//====================================
//---[ Vector Parameter ]-------
class OccaVectorParameter : public OccaParameter
{
private:
const std::string name;
Vector v;
bool useRestrict;
std::string attr;
public:
OccaVectorParameter(const std::string &name_,
Vector &v_,
const bool useRestrict_ = false);
OccaVectorParameter(const std::string &name_,
Vector &v_,
const std::string &attr_,
const bool useRestrict_ = false);
virtual OccaParameter* Clone();
virtual void Setup(OccaIntegrator &integ,
::occa::properties &props);
virtual ::occa::kernelArg KernelArgs();
};
//====================================
//---[ GridFunction Parameter ]-------
class OccaGridFunctionParameter : public OccaParameter
{
private:
const std::string name;
OccaGridFunction &gf;
Vector gfQuad;
bool useRestrict;
public:
OccaGridFunctionParameter(const std::string &name_,
OccaGridFunction &gf_,
const bool useRestrict_ = false);
virtual OccaParameter* Clone();
virtual void Setup(OccaIntegrator &integ,
::occa::properties &props);
virtual ::occa::kernelArg KernelArgs();
};
//====================================
//---[ Coefficient ]------------------
// [MISSING]
// Needs to know about the integrator's
// - fespace
// - ir
// Step where parameters that need the ir get called for setup
// For example, GridFunction (d, e) -> (q, e)
class OccaCoefficient
{
private:
SharedPtr<const Engine> engine;
OccaIntegrator *integ;
std::string name;
::occa::json coeffValue;
::occa::properties props;
std::vector<OccaParameter*> params;
public:
OccaCoefficient(const Engine &e, const double value = 1.0);
OccaCoefficient(const Engine &e, const std::string &source);
OccaCoefficient(const Engine &e, const char *source);
~OccaCoefficient();
OccaCoefficient(const OccaCoefficient &coeff);
const Engine &OccaEngine() const { return *engine; }
::occa::device GetDevice(int idx = 0) const
{ return engine->GetDevice(idx); }
OccaCoefficient& SetName(const std::string &name_);
void Setup(OccaIntegrator &integ_,
::occa::properties &props_);
OccaCoefficient& Add(OccaParameter *param);
OccaCoefficient& IncludeHeader(const std::string &filename);
OccaCoefficient& IncludeSource(const std::string &source);
template <class TM>
OccaCoefficient& AddDefine(const std::string &name_, const TM &value)
{
return Add(new OccaDefineParameter<TM>(name_, value));
}
template <class TM>
OccaCoefficient& AddVariable(const std::string &name_, const TM &value)
{
return Add(new OccaVariableParameter<TM>(name_, value));
}
OccaCoefficient& AddVector(const std::string &name_,
Vector &v,
const bool useRestrict = false);
OccaCoefficient& AddVector(const std::string &name_,
Vector &v,
const std::string &attr,
const bool useRestrict = false);
OccaCoefficient& AddGridFunction(const std::string &name_,
OccaGridFunction &gf,
const bool useRestrict = false);
bool IsConstant();
double GetConstantValue();
Vector Eval();
void Eval(Vector &quadCoeff);
operator ::occa::kernelArg ();
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
-40
View File
@@ -1,40 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_OCCA_DEFINES
#define MFEM_OCCA_DEFINES
#ifndef USING_TENSOR_OPS
# define USING_TENSOR_OPS 0
#endif
#ifdef OCCA_USING_GPU
# define GPU_ORDER_2(I0, I1) @dimOrder(I0, I1)
# define GPU_ORDER_3(I0, I1, I2) @dimOrder(I0, I1, I2)
# define GPU_ORDER_4(I0, I1, I2, I3) @dimOrder(I0, I1, I2, I3)
#else
# define GPU_ORDER_2(I0, I1) @dimOrder(0, 1)
# define GPU_ORDER_3(I0, I1, I2) @dimOrder(0, 1, 2)
# define GPU_ORDER_4(I0, I1, I2, I3) @dimOrder(0, 1, 2, 3)
#endif
#ifndef COEFF
# define COEFF 1.0
# define COEFF_ARGS
#endif
#if USING_TENSOR_OPS
# include "mfem-occa://defines/tensor.okl"
#else
# include "mfem-occa://defines/simplex.okl"
#endif
#endif
-40
View File
@@ -1,40 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#define USING_LOW_ORDER 1
#define USING_HI_ORDER 0
typedef double* DofToQuad_t @dim(NUM_QUAD, NUM_DOFS);
typedef double* DofToQuadD2D_t @dim(2, NUM_QUAD, NUM_DOFS);
typedef double* DofToQuadD3D_t @dim(3, NUM_QUAD, NUM_DOFS);
typedef double* QuadToDof_t @dim(NUM_DOFS, NUM_QUAD);
typedef double* QuadToDofD2D_t @dim(2, NUM_DOFS, NUM_QUAD);
typedef double* QuadToDofD3D_t @dim(3, NUM_DOFS, NUM_QUAD);
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD, numElements);
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD, numElements);
typedef double* SymmOperator2D_t @dim(3, NUM_QUAD, numElements);
typedef double* SymmOperator3D_t @dim(6, NUM_QUAD, numElements);
typedef double* DLocal_t @dim(NUM_DOFS, numElements);
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
#if VDIM_ORDERING == ORDERING_BY_VDIM
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements);
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements);
#else
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements) @dimOrder(2,0,1);
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements) @dimOrder(2,0,1);
#endif
typedef int* DLocalMap_t @dim(NUM_DOFS, numElements);
-85
View File
@@ -1,85 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#if NUM_QUAD_1D < NUM_DOFS_1D
# define NUM_MAX_1D NUM_DOFS_1D
#else
# define NUM_MAX_1D NUM_QUAD_1D
#endif
#define NUM_MAX_2D (NUM_MAX_1D * NUM_MAX_1D)
#define NUM_QUAD_DOFS_1D (NUM_QUAD_1D * NUM_DOFS_1D)
#define QUAD_2D_ID(X, Y) (X + ((Y) * NUM_QUAD_1D))
#define DOFS_2D_ID(X, Y) (X + ((Y) * NUM_DOFS_1D))
#define QUAD_3D_ID(X, Y, Z) (X + ((Y) * NUM_QUAD_1D) + ((Z) * NUM_QUAD_2D))
#define DOFS_3D_ID(X, Y, Z) (X + ((Y) * NUM_DOFS_1D) + ((Z) * NUM_DOFS_2D))
#if NUM_MAX_1D < 8
# define USING_LOW_ORDER 1
# define USING_HI_ORDER 0
#else
# define USING_LOW_ORDER 0
# define USING_HI_ORDER 1
#endif
#define M1_ELEMENT_BATCHES (M1_OUTER_ELEMENT_BATCH * M1_INNER_ELEMENT_BATCH)
typedef double* DofToQuad_t @dim(NUM_QUAD_1D, NUM_DOFS_1D);
typedef double* QuadToDof_t @dim(NUM_DOFS_1D, NUM_QUAD_1D);
typedef double* Jacobian_t @dim(NUM_DIM, NUM_DIM, numElements);
typedef double* Jacobian1D_t @dim(NUM_QUAD_1D, numElements);
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD_2D, numElements);
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD_3D, numElements);
typedef double* SymmOperator1D_t @dim(NUM_QUAD_1D, numElements);
typedef double* SymmOperator2D_t @dim(3, NUM_QUAD_2D, numElements);
typedef double* SymmOperator3D_t @dim(6, NUM_QUAD_3D, numElements);
typedef double* DLocal_t @dim(NUM_DOFS, numElements);
typedef double* DLocal1D_t @dim(NUM_DOFS_1D, numElements);
typedef double* DLocal2D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, numElements);
typedef double* DLocal3D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
typedef double* QLocal1D_t @dim(NUM_QUAD_1D, numElements);
typedef double* QLocal2D_t @dim(NUM_QUAD_1D, NUM_QUAD_1D, numElements);
typedef double* QLocal3D_t @dim(NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
#if VDIM_ORDERING == ORDERING_BY_VDIM
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements);
typedef double* DVLocal1D_t @dim(NUM_VDIM, NUM_DOFS_1D, numElements);
typedef double* DVLocal2D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
typedef double* DVLocal3D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements);
typedef double* QVLocal1D_t @dim(NUM_VDIM, NUM_QUAD_1D, numElements);
typedef double* QVLocal2D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
typedef double* QVLocal3D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
#else
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements) @dimOrder(2,0,1);
typedef double* DVLocal1D_t @dim(NUM_VDIM, NUM_DOFS_1D, numElements) @dimOrder(2,0,1);
typedef double* DVLocal2D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, numElements) @dimOrder(3,0,1,2);
typedef double* DVLocal3D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements) @dimOrder(4,0,1,2,3);
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements) @dimOrder(2,0,1);
typedef double* QVLocal1D_t @dim(NUM_VDIM, NUM_QUAD_1D, numElements) @dimOrder(2,0,1);
typedef double* QVLocal2D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, numElements) @dimOrder(3,0,1,2);
typedef double* QVLocal3D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements) @dimOrder(4,0,1,2,3);
#endif
typedef int* DLocalMap_t @dim(NUM_DOFS, numElements);
typedef int* DLocalMap1D_t @dim(NUM_DOFS_1D, numElements);
typedef int* DLocalMap2D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, numElements);
typedef int* DLocalMap3D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
-168
View File
@@ -1,168 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mfem-occa://defines.okl"
//---[ 2D ]-----------------------------
@kernel void Assemble2D(const int numElements,
const double * restrict quadWeights,
const Jacobian2D_t restrict J,
COEFF_ARGS
SymmOperator2D_t restrict oper) {
for (int e = 0; e < numElements; ++e; @outer) {
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
oper(0, q, e) = c_detJ * (J12*J12 + J22*J22); // (1,1)
oper(1, q, e) = -c_detJ * (J12*J11 + J22*J21); // (1,2) + (2,1)
oper(2, q, e) = c_detJ * (J11*J11 + J21*J21); // (2,2)
}
}
}
@kernel void MultAdd2D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DofToQuadD2D_t restrict dofToQuadD,
const QuadToDof_t restrict quadToDof,
const QuadToDofD2D_t restrict quadToDofD,
const SymmOperator2D_t restrict oper,
const DLocal_t restrict solIn,
DLocal_t restrict solOut) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
double r_sol[NUM_DOFS];
for (int d = 0; d < NUM_DOFS; ++d) {
r_sol[d] = 0;
}
for (int q = 0; q < NUM_QUAD; ++q) {
double gradX = 0, gradY = 0;
for (int d = 0; d < NUM_DOFS; ++d) {
const double s = solIn(d, e);
gradX += s * quadToDofD(0, d, q);
gradY += s * quadToDofD(1, d, q);
}
const double O11 = oper(0, q, e);
const double O12 = oper(1, q, e);
const double O22 = oper(2, q, e);
const double gradX2 = (O11 * gradX) + (O12 * gradY);
const double gradY2 = (O12 * gradX) + (O22 * gradY);
for (int d = 0; d < NUM_DOFS; ++d) {
r_sol[d] += ((gradX2 * quadToDofD(0, d, q)) +
(gradY2 * quadToDofD(1, d, q)));
}
}
for (int d = 0; d < NUM_DOFS; ++d) {
solOut(d, e) += r_sol[d];
}
}
}
}
//======================================
//---[ 3D ]-----------------------------
@kernel void Assemble3D(const int numElements,
const double * restrict quadWeights,
const Jacobian3D_t restrict J,
COEFF_ARGS
SymmOperator3D_t restrict oper) {
for (int e = 0; e < numElements; ++e; @outer) {
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
const double c_detJ = quadWeights[q] * COEFF / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J23 * J31) - (J21 * J33);
const double A13 = (J21 * J32) - (J22 * J31);
const double A21 = (J13 * J32) - (J12 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J12 * J31) - (J11 * J32);
const double A31 = (J12 * J23) - (J13 * J22);
const double A32 = (J13 * J21) - (J11 * J23);
const double A33 = (J11 * J22) - (J12 * J21);
// adj(J)^Tadj(J)
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2) + (2,1)
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3) + (3,1)
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3) + (3,2)
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
}
}
}
@kernel void MultAdd3D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DofToQuadD3D_t restrict dofToQuadD,
const QuadToDof_t restrict quadToDof,
const QuadToDofD3D_t restrict quadToDofD,
const SymmOperator3D_t restrict oper,
const DLocal_t restrict solIn,
DLocal_t restrict solOut) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
double r_sol[NUM_DOFS];
for (int d = 0; d < NUM_DOFS; ++d) {
r_sol[d] = 0;
}
for (int q = 0; q < NUM_QUAD; ++q) {
double gradX = 0, gradY = 0, gradZ = 0;
for (int d = 0; d < NUM_DOFS; ++d) {
const double s = solIn(d, e);
gradX += s * quadToDofD(0, d, q);
gradY += s * quadToDofD(1, d, q);
gradZ += s * quadToDofD(2, d, q);
}
const double O11 = oper(0, q, e);
const double O12 = oper(1, q, e);
const double O13 = oper(2, q, e);
const double O22 = oper(3, q, e);
const double O23 = oper(4, q, e);
const double O33 = oper(5, q, e);
const double gradX2 = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
const double gradY2 = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
const double gradZ2 = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
for (int d = 0; d < NUM_DOFS; ++d) {
r_sol[d] += ((gradX2 * quadToDofD(0, d, q)) +
(gradY2 * quadToDofD(1, d, q)) +
(gradZ2 * quadToDofD(2, d, q)));
}
}
for (int d = 0; d < NUM_DOFS; ++d) {
solOut(d, e) += r_sol[d];
}
}
}
}
//======================================
@@ -1,182 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mfem-occa://defines.okl"
//---[ 2D ]-----------------------------
@kernel void Assemble2D(const int numElements,
const double *quadWeights,
const Jacobian2D_t J,
COEFF_ARGS
SymmOperator2D_t oper) {
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
if (e < numElements) {
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
for (int q = qOff; q < NUM_QUAD; q += A2_QUAD_BATCH) {
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
oper(0, q, e) = c_detJ * (J12*J12 + J22*J22); // (1,1)
oper(1, q, e) = -c_detJ * (J12*J11 + J22*J21); // (1,2) + (2,1)
oper(2, q, e) = c_detJ * (J11*J11 + J21*J21); // (2,2)
}
}
}
}
}
}
@kernel void MultAdd2D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DofToQuadD2D_t restrict dofToQuadD,
const QuadToDof_t restrict quadToDof,
const QuadToDofD2D_t restrict quadToDofD,
const SymmOperator2D_t restrict oper,
const DLocal_t restrict solIn,
DLocal_t restrict solOut) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
@shared double s_gradX[NUM_QUAD];
@shared double s_gradY[NUM_QUAD];
for (int qOff = 0; qOff < M2_INNER_BATCH; ++qOff; @inner) {
for (int q = qOff; q < NUM_QUAD; q += M2_INNER_BATCH) {
double gradX = 0, gradY = 0;
for (int d = 0; d < NUM_DOFS; ++d) {
const double s = solIn(d, e);
gradX += s * quadToDofD(0, d, q);
gradY += s * quadToDofD(1, d, q);
}
const double O11 = oper(0, q, e);
const double O12 = oper(1, q, e);
const double O22 = oper(2, q, e);
s_gradX[q] = (O11 * gradX) + (O12 * gradY);
s_gradY[q] = (O12 * gradX) + (O22 * gradY);
}
}
for (int dOff = 0; dOff < M2_INNER_BATCH; ++dOff) {
for (int d = dOff; d < NUM_DOFS; d += M2_INNER_BATCH) {
double r_sol = 0;
for (int q = 0; q < NUM_QUAD; ++q) {
// FIXME: s_gradX and s_gradY are @shared used outside of @inner
r_sol += ((s_gradX[q] * quadToDofD(0, d, q)) +
(s_gradY[q] * quadToDofD(1, d, q)));
}
solOut(d, e) += r_sol;
}
}
}
}
//======================================
//---[ 3D ]-----------------------------
@kernel void Assemble3D(const int numElements,
const double *quadWeights,
const Jacobian3D_t J,
COEFF_ARGS
SymmOperator3D_t oper) {
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
if (e < numElements) {
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
for (int q = qOff; q < NUM_QUAD; q += A3_QUAD_BATCH) {
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
const double c_detJ = quadWeights[q] * COEFF / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J23 * J31) - (J21 * J33);
const double A13 = (J21 * J32) - (J22 * J31);
const double A21 = (J13 * J32) - (J12 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J12 * J31) - (J11 * J32);
const double A31 = (J12 * J23) - (J13 * J22);
const double A32 = (J13 * J21) - (J11 * J23);
const double A33 = (J11 * J22) - (J12 * J21);
// adj(J)^Tadj(J)
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2) + (2,1)
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3) + (3,1)
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3) + (3,2)
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
}
}
}
}
}
}
@kernel void MultAdd3D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DofToQuadD3D_t restrict dofToQuadD,
const QuadToDof_t restrict quadToDof,
const QuadToDofD3D_t restrict quadToDofD,
const SymmOperator3D_t restrict oper,
const DLocal_t restrict solIn,
DLocal_t restrict solOut) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
@shared double s_gradX[NUM_QUAD];
@shared double s_gradY[NUM_QUAD];
@shared double s_gradZ[NUM_QUAD];
for (int qOff = 0; qOff < M3_INNER_BATCH; ++qOff; @inner) {
for (int q = qOff; q < NUM_QUAD; q += M3_INNER_BATCH) {
double gradX = 0, gradY = 0, gradZ = 0;
for (int d = 0; d < NUM_DOFS; ++d) {
const double s = solIn(d, e);
gradX += s * quadToDofD(0, d, q);
gradY += s * quadToDofD(1, d, q);
gradZ += s * quadToDofD(2, d, q);
}
const double O11 = oper(0, q, e);
const double O12 = oper(1, q, e);
const double O13 = oper(2, q, e);
const double O22 = oper(3, q, e);
const double O23 = oper(4, q, e);
const double O33 = oper(5, q, e);
s_gradX[q] = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
s_gradY[q] = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
s_gradZ[q] = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
}
}
for (int dOff = 0; dOff < M3_INNER_BATCH; ++dOff) {
for (int d = dOff; d < NUM_DOFS; d += M3_INNER_BATCH) {
double r_sol = 0;
for (int q = 0; q < NUM_QUAD; ++q) {
r_sol += ((s_gradX[q] * quadToDofD(0, d, q)) +
(s_gradY[q] * quadToDofD(1, d, q)) +
(s_gradZ[q] * quadToDofD(2, d, q)));
}
solOut(d, e) += r_sol;
}
}
}
}
//======================================
-370
View File
@@ -1,370 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mfem-occa://defines.okl"
//---[ 1D ]-----------------------------
@kernel void Assemble1D(const int numElements,
const double * restrict quadWeights,
const Jacobian1D_t restrict J,
COEFF_ARGS
SymmOperator1D_t restrict oper) {
for (int e = 0; e < numElements; ++e; @outer) {
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
oper(q, e) = quadWeights[q] * COEFF / J(q, e);
}
}
}
@kernel void MultAdd1D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DofToQuad_t restrict dofToQuadD,
const QuadToDof_t restrict quadToDof,
const QuadToDof_t restrict quadToDofD,
const SymmOperator1D_t restrict oper,
const DLocal1D_t restrict solIn,
DLocal1D_t restrict solOut) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
double grad[NUM_QUAD_1D];
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
grad[qx] = 0;
}
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const double s = solIn(dx, e);
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
grad[qx] += s * dofToQuadD(qx, dx);
}
}
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
grad[qx] *= oper(qx, e);
}
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
const double gradX = grad[qx];
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
solOut(dx, e) += gradX * quadToDofD(dx, qx);
}
}
}
}
}
//======================================
//---[ 2D ]-----------------------------
@kernel void Assemble2D(const int numElements,
const double * restrict quadWeights,
const Jacobian2D_t restrict J,
COEFF_ARGS
SymmOperator2D_t restrict oper) {
for (int e = 0; e < numElements; ++e; @outer) {
for (int q = 0; q < NUM_QUAD_2D; ++q; @inner) {
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
oper(0, q, e) = c_detJ * (J21*J21 + J22*J22); // (1,1)
oper(1, q, e) = -c_detJ * (J21*J11 + J22*J12); // (1,2), (2,1)
oper(2, q, e) = c_detJ * (J11*J11 + J12*J12); // (2,2)
}
}
}
@kernel void MultAdd2D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DofToQuad_t restrict dofToQuadD,
const QuadToDof_t restrict quadToDof,
const QuadToDof_t restrict quadToDofD,
const SymmOperator2D_t restrict oper,
const DLocal2D_t restrict solIn,
DLocal2D_t restrict solOut) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
double grad[NUM_QUAD_1D][NUM_QUAD_1D][2];
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
grad[qy][qx][0] = 0;
grad[qy][qx][1] = 0;
}
}
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
double gradX[NUM_QUAD_1D][2];
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
gradX[qx][0] = 0;
gradX[qx][1] = 0;
}
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const double s = solIn(dx, dy, e);
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
gradX[qx][0] += s * dofToQuad(qx, dx);
gradX[qx][1] += s * dofToQuadD(qx, dx);
}
}
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
const double wy = dofToQuad(qy, dy);
const double wDy = dofToQuadD(qy, dy);
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
grad[qy][qx][0] += gradX[qx][1] * wy;
grad[qy][qx][1] += gradX[qx][0] * wDy;
}
}
}
// Calculate Dxy, xDy in plane
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
const int q = QUAD_2D_ID(qx, qy);
const double O11 = oper(0, q, e);
const double O12 = oper(1, q, e);
const double O22 = oper(2, q, e);
const double gradX = grad[qy][qx][0];
const double gradY = grad[qy][qx][1];
grad[qy][qx][0] = (O11 * gradX) + (O12 * gradY);
grad[qy][qx][1] = (O12 * gradX) + (O22 * gradY);
}
}
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
double gradX[NUM_DOFS_1D][2];
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
gradX[dx][0] = 0;
gradX[dx][1] = 0;
}
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
const double gX = grad[qy][qx][0];
const double gY = grad[qy][qx][1];
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const double wx = quadToDof(dx, qx);
const double wDx = quadToDofD(dx, qx);
gradX[dx][0] += gX * wDx;
gradX[dx][1] += gY * wx;
}
}
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
const double wy = quadToDof(dy, qy);
const double wDy = quadToDofD(dy, qy);
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
solOut(dx, dy, e) += ((gradX[dx][0] * wy) +
(gradX[dx][1] * wDy));
}
}
}
}
}
}
//======================================
//---[ 3D ]-----------------------------
@kernel void Assemble3D(const int numElements,
const double * restrict quadWeights,
const Jacobian3D_t restrict J,
COEFF_ARGS
SymmOperator3D_t restrict oper) {
for (int e = 0; e < numElements; ++e; @outer) {
for (int q = 0; q < NUM_QUAD_3D; ++q; @inner) {
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
const double c_detJ = quadWeights[q] * COEFF / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J23 * J31) - (J21 * J33);
const double A13 = (J21 * J32) - (J22 * J31);
const double A21 = (J13 * J32) - (J12 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J12 * J31) - (J11 * J32);
const double A31 = (J12 * J23) - (J13 * J22);
const double A32 = (J13 * J21) - (J11 * J23);
const double A33 = (J11 * J22) - (J12 * J21);
// adj(J)^Tadj(J)
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2), (2,1)
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3), (3,1)
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3), (3,2)
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
}
}
}
@kernel void MultAdd3D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DofToQuad_t restrict dofToQuadD,
const QuadToDof_t restrict quadToDof,
const QuadToDof_t restrict quadToDofD,
const SymmOperator3D_t restrict oper,
const DLocal3D_t restrict solIn,
DLocal3D_t restrict solOut) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
double grad[NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D][4];
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
grad[qz][qy][qx][0] = 0;
grad[qz][qy][qx][1] = 0;
grad[qz][qy][qx][2] = 0;
}
}
}
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
double gradXY[NUM_QUAD_1D][NUM_QUAD_1D][4];
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
gradXY[qy][qx][0] = 0;
gradXY[qy][qx][1] = 0;
gradXY[qy][qx][2] = 0;
}
}
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
double gradX[NUM_QUAD_1D][2];
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
gradX[qx][0] = 0;
gradX[qx][1] = 0;
}
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const double s = solIn(dx, dy, dz, e);
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
gradX[qx][0] += s * dofToQuad(qx, dx);
gradX[qx][1] += s * dofToQuadD(qx, dx);
}
}
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
const double wy = dofToQuad(qy, dy);
const double wDy = dofToQuadD(qy, dy);
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
const double wx = gradX[qx][0];
const double wDx = gradX[qx][1];
gradXY[qy][qx][0] += wDx * wy;
gradXY[qy][qx][1] += wx * wDy;
gradXY[qy][qx][2] += wx * wy;
}
}
}
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
const double wz = dofToQuad(qz, dz);
const double wDz = dofToQuadD(qz, dz);
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
grad[qz][qy][qx][0] += gradXY[qy][qx][0] * wz;
grad[qz][qy][qx][1] += gradXY[qy][qx][1] * wz;
grad[qz][qy][qx][2] += gradXY[qy][qx][2] * wDz;
}
}
}
}
// Calculate Dxyz, xDyz, xyDz in plane
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
const int q = QUAD_3D_ID(qx, qy, qz);
const double O11 = oper(0, q, e);
const double O12 = oper(1, q, e);
const double O13 = oper(2, q, e);
const double O22 = oper(3, q, e);
const double O23 = oper(4, q, e);
const double O33 = oper(5, q, e);
const double gradX = grad[qz][qy][qx][0];
const double gradY = grad[qz][qy][qx][1];
const double gradZ = grad[qz][qy][qx][2];
grad[qz][qy][qx][0] = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
grad[qz][qy][qx][1] = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
grad[qz][qy][qx][2] = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
}
}
}
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
double gradXY[NUM_DOFS_1D][NUM_DOFS_1D][4];
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
gradXY[dy][dx][0] = 0;
gradXY[dy][dx][1] = 0;
gradXY[dy][dx][2] = 0;
}
}
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
double gradX[NUM_DOFS_1D][4];
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
gradX[dx][0] = 0;
gradX[dx][1] = 0;
gradX[dx][2] = 0;
}
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
const double gX = grad[qz][qy][qx][0];
const double gY = grad[qz][qy][qx][1];
const double gZ = grad[qz][qy][qx][2];
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const double wx = quadToDof(dx, qx);
const double wDx = quadToDofD(dx, qx);
gradX[dx][0] += gX * wDx;
gradX[dx][1] += gY * wx;
gradX[dx][2] += gZ * wx;
}
}
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
const double wy = quadToDof(dy, qy);
const double wDy = quadToDofD(dy, qy);
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
gradXY[dy][dx][0] += gradX[dx][0] * wy;
gradXY[dy][dx][1] += gradX[dx][1] * wDy;
gradXY[dy][dx][2] += gradX[dx][2] * wy;
}
}
}
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
const double wz = quadToDof(dz, qz);
const double wDz = quadToDofD(dz, qz);
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
solOut(dx, dy, dz, e) += ((gradXY[dy][dx][0] * wz) +
(gradXY[dy][dx][1] * wz) +
(gradXY[dy][dx][2] * wDz));
}
}
}
}
}
}
}
//======================================
@@ -1,433 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mfem-occa://defines.okl"
//---[ 1D ]-----------------------------
@kernel void Assemble1D(const int numElements,
const double *quadWeights,
const Jacobian1D_t J,
COEFF_ARGS
SymmOperator1D_t oper) {
for (int eOff = 0; eOff < numElements; eOff += A1_ELEMENT_BATCH; @outer) {
for (int e = eOff; e < (eOff + A1_ELEMENT_BATCH); ++e; @inner) {
if (e < numElements) {
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
oper(q, e) = quadWeights[q] * COEFF / J(q, e);
}
}
}
}
}
@kernel void MultAdd1D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DofToQuad_t restrict dofToQuadD,
const QuadToDof_t restrict quadToDof,
const QuadToDof_t restrict quadToDofD,
const SymmOperator1D_t restrict oper,
const DLocal1D_t restrict solIn,
DLocal1D_t restrict solOut) {
// Iterate over elements
for (int eOff = 0; eOff < numElements; eOff += M1_ELEMENT_BATCHES; @outer) {
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
@exclusive double grad[NUM_QUAD_1D];
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
for (int i = el; i < NUM_QUAD_DOFS_1D; i += M1_INNER_ELEMENT_BATCH) {
s_dofToQuadD[i] = dofToQuadD[i];
s_quadToDofD[i] = quadToDofD[i];
}
}
for (int b = 0; b < M1_OUTER_ELEMENT_BATCH; ++b) {
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
const int e = eOff + b*M1_INNER_ELEMENT_BATCH + el;
if (e < numElements) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
grad[qx] = 0;
}
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const double s = solIn(dx, e);
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
grad[qx] += s * s_dofToQuadD(qx, dx);
}
}
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
grad[qx] *= oper(qx, e);
}
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
double s = 0;
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
s += grad[qx] * s_quadToDofD(dx, qx);
}
solOut(dx, e) += s;
}
}
}
}
}
}
//======================================
//---[ 2D ]-----------------------------
@kernel void Assemble2D(const int numElements,
const double *quadWeights,
const Jacobian2D_t J,
COEFF_ARGS
SymmOperator2D_t oper) {
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
if (e < numElements) {
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
for (int q = qOff; q < NUM_QUAD_2D; q += A2_QUAD_BATCH) {
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
oper(0, q, e) = c_detJ * (J21*J21 + J22*J22); // (1,1)
oper(1, q, e) = -c_detJ * (J21*J11 + J22*J12); // (1,2), (2,1)
oper(2, q, e) = c_detJ * (J11*J11 + J12*J12); // (2,2)
}
}
}
}
}
}
@kernel void MultAdd2D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DofToQuad_t restrict dofToQuadD,
const QuadToDof_t restrict quadToDof,
const QuadToDof_t restrict quadToDofD,
const SymmOperator2D_t restrict oper,
const DLocal2D_t restrict solIn,
DLocal2D_t restrict solOut) {
// Iterate over elements
for (int eOff = 0; eOff < numElements; eOff += M2_ELEMENT_BATCH; @outer) {
// Store dof <--> quad mappings
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
// Store xy planes in shared memory
@shared double s_xy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
@shared double s_xDy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
@shared double s_grad[2 * NUM_QUAD_2D] @dim(2, NUM_QUAD_1D, NUM_QUAD_1D);
@exclusive double r_x[NUM_MAX_1D];
@exclusive double r_y[NUM_QUAD_1D];
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
for (int id = x; id < NUM_QUAD_DOFS_1D; id += NUM_MAX_1D) {
s_dofToQuad[id] = dofToQuad[id];
s_dofToQuadD[id] = dofToQuadD[id];
s_quadToDof[id] = quadToDof[id];
s_quadToDofD[id] = quadToDofD[id];
}
}
for (int e = eOff; e < (eOff + M2_ELEMENT_BATCH); ++e) {
if (e < numElements) {
for (int dx = 0; dx < NUM_DOFS_1D; ++dx; @inner) {
if (dx < NUM_DOFS_1D) {
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
s_xy(dx, qy) = 0;
s_xDy(dx, qy) = 0;
}
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
r_x[dy] = solIn(dx, dy, e);
}
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
double xy = 0;
double xDy = 0;
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
xy += r_x[dy] * s_dofToQuad(qy, dy);
xDy += r_x[dy] * s_dofToQuadD(qy, dy);
}
s_xy(dx, qy) = xy;
s_xDy(dx, qy) = xDy;
}
}
}
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
if (qy < NUM_QUAD_1D) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
double gradX = 0, gradY = 0;
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
gradX += s_xy(dx, qy) * s_dofToQuadD(qx, dx);
gradY += s_xDy(dx, qy) * s_dofToQuad(qx, dx);
}
const int q = QUAD_2D_ID(qx, qy);
const double O11 = oper(0, q, e);
const double O12 = oper(1, q, e);
const double O22 = oper(2, q, e);
s_grad(0, qx, qy) = (O11 * gradX) + (O12 * gradY);
s_grad(1, qx, qy) = (O12 * gradX) + (O22 * gradY);
}
}
}
for (int qx = 0; qx < NUM_QUAD_1D; ++qx; @inner) {
if (qx < NUM_QUAD_1D) {
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
s_xy(dy, qx) = 0;
s_xDy(dy, qx) = 0;
}
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
r_x[qy] = s_grad(0, qx, qy);
r_y[qy] = s_grad(1, qx, qy);
}
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
double xy = 0;
double xDy = 0;
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
xy += r_x[qy] * s_quadToDof(dy, qy);
xDy += r_y[qy] * s_quadToDofD(dy, qy);
}
s_xy(dy, qx) = xy;
s_xDy(dy, qx) = xDy;
}
}
}
for (int dx = 0; dx < NUM_DOFS_1D; ++dx; @inner) {
if (dx < NUM_DOFS_1D) {
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
double s = 0;
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
s += ((s_xy(dy, qx) * s_quadToDofD(dx, qx)) +
(s_xDy(dy, qx) * s_quadToDof(dx, qx)));
}
solOut(dx, dy, e) += s;
}
}
}
}
}
}
}
//======================================
//---[ 3D ]-----------------------------
@kernel void Assemble3D(const int numElements,
const double *quadWeights,
const Jacobian3D_t J,
COEFF_ARGS
SymmOperator3D_t oper) {
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
if (e < numElements) {
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
for (int q = qOff; q < NUM_QUAD_3D; q += A3_QUAD_BATCH) {
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
const double c_detJ = quadWeights[q] * COEFF / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J23 * J31) - (J21 * J33);
const double A13 = (J21 * J32) - (J22 * J31);
const double A21 = (J13 * J32) - (J12 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J12 * J31) - (J11 * J32);
const double A31 = (J12 * J23) - (J13 * J22);
const double A32 = (J13 * J21) - (J11 * J23);
const double A33 = (J11 * J22) - (J12 * J21);
// adj(J)^Tadj(J)
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2), (2,1)
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3), (3,1)
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3), (3,2)
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
}
}
}
}
}
}
@kernel void MultAdd3D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DofToQuad_t restrict dofToQuadD,
const QuadToDof_t restrict quadToDof,
const QuadToDof_t restrict quadToDofD,
const SymmOperator3D_t restrict oper,
const DLocal3D_t restrict solIn,
DLocal3D_t restrict solOut) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
// Store dof <--> quad mappings
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
// Store xy planes in shared memory
@shared double s_z[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
@shared double s_Dz[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
@shared double s_xyDz[NUM_QUAD_2D] @dim(NUM_QUAD_1D, NUM_QUAD_1D);
// Store z axis as registers
@exclusive double r_qz[NUM_QUAD_1D];
@exclusive double r_qDz[NUM_QUAD_1D];
@exclusive double r_dDxyz[NUM_DOFS_1D];
@exclusive double r_dxDyz[NUM_DOFS_1D];
@exclusive double r_dxyDz[NUM_DOFS_1D];
for (int y = 0; y < NUM_MAX_1D; ++y; @inner) {
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
const int id = (y * NUM_MAX_1D) + x;
// Fetch Q <--> D maps
if (id < NUM_QUAD_DOFS_1D) {
s_dofToQuad[id] = dofToQuad[id];
s_dofToQuadD[id] = dofToQuadD[id];
s_quadToDof[id] = quadToDof[id];
s_quadToDofD[id] = quadToDofD[id];
}
// Initialize our Z axis
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
r_qz[qz] = 0;
r_qDz[qz] = 0;
}
// Initialize our solution updates in the Z axis
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
r_dDxyz[dz] = 0;
r_dxDyz[dz] = 0;
r_dxyDz[dz] = 0;
}
}
}
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
const double s = solIn(dx, dy, dz, e);
// Calculate D -> Q in the Z axis
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
r_qz[qz] += s * s_dofToQuad(qz, dz);
r_qDz[qz] += s * s_dofToQuadD(qz, dz);
}
}
}
}
}
// For each xy plane
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
// Fill xy plane at given z position
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
s_z(dx, dy) = r_qz[qz];
s_Dz(dx, dy) = r_qDz[qz];
}
}
}
// Calculate Dxyz, xDyz, xyDz in plane
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
double Dxyz = 0;
double xDyz = 0;
double xyDz = 0;
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
const double wy = s_dofToQuad(qy, dy);
const double wDy = s_dofToQuadD(qy, dy);
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const double wx = s_dofToQuad(qx, dx);
const double wDx = s_dofToQuadD(qx, dx);
const double z = s_z(dx, dy);
const double Dz = s_Dz(dx, dy);
Dxyz += wDx * wy * z;
xDyz += wx * wDy * z;
xyDz += wx * wy * Dz;
}
}
const int q = QUAD_3D_ID(qx, qy, qz);
const double O11 = oper(0, q, e);
const double O12 = oper(1, q, e);
const double O13 = oper(2, q, e);
const double O22 = oper(3, q, e);
const double O23 = oper(4, q, e);
const double O33 = oper(5, q, e);
const double qDxyz = (O11 * Dxyz) + (O12 * xDyz) + (O13 * xyDz);
const double qxDyz = (O12 * Dxyz) + (O22 * xDyz) + (O23 * xyDz);
const double qxyDz = (O13 * Dxyz) + (O23 * xDyz) + (O33 * xyDz);
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
const double wz = s_quadToDof(dz, qz);
const double wDz = s_quadToDofD(dz, qz);
r_dDxyz[dz] += wz * qDxyz;
r_dxDyz[dz] += wz * qxDyz;
r_dxyDz[dz] += wDz * qxyDz;
}
}
}
}
}
// Iterate over xy planes to compute solution
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
// Place xy plane in shared memory
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
s_z(qx, qy) = r_dDxyz[dz];
s_Dz(qx, qy) = r_dxDyz[dz];
s_xyDz(qx, qy) = r_dxyDz[dz];
}
}
}
// Finalize solution in xy plane
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
double solZ = 0;
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
const double wy = s_quadToDof(dy, qy);
const double wDy = s_quadToDofD(dy, qy);
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
const double wx = s_quadToDof(dx, qx);
const double wDx = s_quadToDofD(dx, qx);
const double Dxyz = s_z(qx, qy);
const double xDyz = s_Dz(qx, qy);
const double xyDz = s_xyDz(qx, qy);
solZ += ((wDx * wy * Dxyz) +
(wx * wDy * xDyz) +
(wx * wy * xyDz));
}
}
solOut(dx, dy, dz, e) += solZ;
}
}
}
}
}
}
//======================================
-140
View File
@@ -1,140 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "backend.hpp"
#include "url_handler.hpp"
#include "bilinearform.hpp"
#include "../../general/array.hpp"
namespace mfem
{
namespace occa
{
bool Engine::fileOpenerRegistered = false;
void Engine::Init(const std::string &engine_spec)
{
//
// Initialize inherited fields
//
memory_resources[0] = NULL;
workers_weights[0]= 1.0;
workers_mem_res[0] = 0;
//
// Initialize the OCCA engine
//
::occa::properties props(engine_spec);
device = new ::occa::device[1];
device[0].setup(props);
okl_path = "mfem-occa://";
// okl_defines = "...";
if (!fileOpenerRegistered)
{
// The directories from "MFEM_OCCA_OKL_PATH", if any, have the highest
// priority.
FileOpener *fo = new FileOpener("mfem-occa://", "MFEM_OCCA_OKL_PATH");
// Next in priority is the source path, if it exists.
std::string mfem_src_prefix = mfem::GetSourcePath();
fo->AddDir(mfem_src_prefix + "/backends/occa");
// And last in priority is the install path, if it exists.
std::string mfem_install_prefix = mfem::GetInstallPath();
fo->AddDir(mfem_install_prefix + "/lib/mfem/occa");
::occa::io::fileOpener::add(fo);
fileOpenerRegistered = true;
}
}
Engine::Engine(const std::string &engine_spec)
: mfem::Engine(NULL, 1, 1)
{
Init(engine_spec);
}
#ifdef MFEM_USE_MPI
Engine::Engine(MPI_Comm _comm, const std::string &engine_spec)
: mfem::Engine(NULL, 1, 1)
{
comm = _comm;
Init(engine_spec);
}
#endif
DLayout Engine::MakeLayout(std::size_t size) const
{
return DLayout(new Layout(*this, size));
}
DLayout Engine::MakeLayout(const mfem::Array<std::size_t> &offsets) const
{
MFEM_ASSERT(offsets.Size() == 2,
"multiple workers are not supported yet");
return DLayout(new Layout(*this, offsets.Last()));
}
DArray Engine::MakeArray(PLayout &layout, std::size_t item_size) const
{
MFEM_ASSERT(dynamic_cast<Layout *>(&layout) != NULL,
"invalid input layout");
Layout *lt = static_cast<Layout *>(&layout);
return DArray(new Array(*lt, item_size));
}
DVector Engine::MakeVector(PLayout &layout, int type_id) const
{
MFEM_ASSERT(type_id == ScalarId<double>::value, "invalid type_id");
MFEM_ASSERT(dynamic_cast<Layout *>(&layout) != NULL,
"invalid input layout");
Layout *lt = static_cast<Layout *>(&layout);
return DVector(new Vector(*lt));
}
DFiniteElementSpace Engine::MakeFESpace(mfem::FiniteElementSpace &fespace) const
{
return DFiniteElementSpace(new FiniteElementSpace(*this, fespace));
}
DBilinearForm Engine::MakeBilinearForm(mfem::BilinearForm &bf) const
{
return DBilinearForm(new BilinearForm(*this, bf));
}
void Engine::AssembleLinearForm(LinearForm &l_form) const
{
/// FIXME - What will the actual parameters be?
MFEM_ABORT("FIXME");
}
mfem::Operator *Engine::MakeOperator(const MixedBilinearForm &mbl_form) const
{
/// FIXME - What will the actual parameters be?
MFEM_ABORT("FIXME");
return NULL;
}
mfem::Operator *Engine::MakeOperator(const NonlinearForm &nl_form) const
{
/// FIXME - What will the actual parameters be?
MFEM_ABORT("FIXME");
return NULL;
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-111
View File
@@ -1,111 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_ENGINE_HPP
#define MFEM_BACKENDS_OCCA_ENGINE_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "../base/backend.hpp"
#include <occa.hpp>
namespace mfem
{
namespace occa
{
class Engine : public mfem::Engine
{
protected:
//
// Inherited fields
//
// mfem::Backend *backend;
#ifdef MFEM_USE_MPI
// MPI_Comm comm;
#endif
// int num_mem_res;
// int num_workers;
// MemoryResource **memory_resources;
// double *workers_weights;
// int *workers_mem_res;
static bool fileOpenerRegistered;
::occa::device *device; // An array of OCCA devices
std::string okl_path, okl_defines;
void Init(const std::string &engine_spec);
public:
Engine(const std::string &engine_spec);
#ifdef MFEM_USE_MPI
Engine(MPI_Comm comm, const std::string &engine_spec);
#endif
virtual ~Engine() { delete [] device; }
/**
@name OCCA specific interface, used by other objects in the OCCA backend
*/
///@{
::occa::device GetDevice(int idx = 0) const { return device[idx]; }
/// TODO: doxygen
const std::string &GetOklPath() const { return okl_path; }
/// TODO: doxygen
const std::string &GetOklDefines() const { return okl_defines; }
///@}
// End: OCCA specific interface
/**
@name Virtual interface: finite element data structures and algorithms
*/
///@{
virtual DLayout MakeLayout(std::size_t size) const;
virtual DLayout MakeLayout(const mfem::Array<std::size_t> &offsets) const;
virtual DArray MakeArray(PLayout &layout, std::size_t item_size) const;
virtual DVector MakeVector(PLayout &layout,
int type_id = ScalarId<double>::value) const;
virtual DFiniteElementSpace MakeFESpace(mfem::FiniteElementSpace &
fespace) const;
virtual DBilinearForm MakeBilinearForm(mfem::BilinearForm &bf) const;
/// FIXME - What will the actual parameters be?
virtual void AssembleLinearForm(LinearForm &l_form) const;
/// FIXME - What will the actual parameters be?
virtual mfem::Operator *MakeOperator(const MixedBilinearForm &mbl_form) const;
/// FIXME - What will the actual parameters be?
virtual mfem::Operator *MakeOperator(const NonlinearForm &nl_form) const;
///@}
// End: Virtual interface
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_ENGINE_HPP
-174
View File
@@ -1,174 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "backend.hpp"
#include "fespace.hpp"
#include "interpolation.hpp"
namespace mfem
{
namespace occa
{
FiniteElementSpace::FiniteElementSpace(const Engine &e,
mfem::FiniteElementSpace &fespace)
: PFiniteElementSpace(e, fespace),
e_layout(e, 0) // resized in SetupLocalGlobalMaps()
{
vdim = fespace.GetVDim();
ordering = fespace.GetOrdering();
SetupLocalGlobalMaps();
SetupOperators();
SetupKernels();
}
FiniteElementSpace::~FiniteElementSpace()
{
delete [] elementDofMap;
delete [] elementDofMapInverse;
delete restrictionOp;
delete prolongationOp;
}
void FiniteElementSpace::SetupLocalGlobalMaps()
{
const mfem::FiniteElement &fe = *(fes->GetFE(0));
const mfem::TensorBasisElement *el =
dynamic_cast<const mfem::TensorBasisElement*>(&fe);
const mfem::Table &e2dTable = fes->GetElementToDofTable();
const int *elementMap = e2dTable.GetJ();
const int elements = fes->GetNE();
globalDofs = fes->GetNDofs();
localDofs = fe.GetDof();
e_layout.Resize(localDofs * elements * fes->GetVDim());
elementDofMap = new int[localDofs];
elementDofMapInverse = new int[localDofs];
if (el)
{
::memcpy(elementDofMap,
el->GetDofMap().GetData(),
localDofs * sizeof(int));
}
else
{
for (int i = 0; i < localDofs; ++i)
{
elementDofMap[i] = i;
}
}
for (int i = 0; i < localDofs; ++i)
{
elementDofMapInverse[elementDofMap[i]] = i;
}
// Allocate device offsets and indices
globalToLocalOffsets.allocate(GetDevice(),
globalDofs + 1);
globalToLocalIndices.allocate(GetDevice(),
localDofs, elements);
localToGlobalMap.allocate(GetDevice(),
localDofs, elements);
int *offsets = globalToLocalOffsets.ptr();
int *indices = globalToLocalIndices.ptr();
int *l2gMap = localToGlobalMap.ptr();
// We'll be keeping a count of how many local nodes point
// to its global dof
for (int i = 0; i <= globalDofs; ++i)
{
offsets[i] = 0;
}
for (int e = 0; e < elements; ++e)
{
for (int d = 0; d < localDofs; ++d)
{
const int gid = elementMap[localDofs*e + d];
++offsets[gid + 1];
}
}
// Aggregate to find offsets for each global dof
for (int i = 1; i <= globalDofs; ++i)
{
offsets[i] += offsets[i - 1];
}
// For each global dof, fill in all local nodes that point
// to it
for (int e = 0; e < elements; ++e)
{
for (int d = 0; d < localDofs; ++d)
{
const int gid = elementMap[localDofs*e + elementDofMap[d]];
const int lid = localDofs*e + d;
indices[offsets[gid]++] = lid;
l2gMap[lid] = gid;
}
}
// We shifted the offsets vector by 1 by using it
// as a counter. Now we shift it back.
for (int i = globalDofs; i > 0; --i)
{
offsets[i] = offsets[i - 1];
}
offsets[0] = 0;
globalToLocalOffsets.keepInDevice();
globalToLocalIndices.keepInDevice();
localToGlobalMap.keepInDevice();
}
void FiniteElementSpace::SetupOperators()
{
const mfem::SparseMatrix *R = fes->GetRestrictionMatrix();
const mfem::Operator *P = fes->GetProlongationMatrix();
CreateRPOperators(OccaVLayout(), OccaTrueVLayout(),
R, P,
restrictionOp,
prolongationOp);
}
void FiniteElementSpace::SetupKernels()
{
::occa::properties props("defines: {"
" TILESIZE: 256,"
"}");
props["defines/NUM_VDIM"] = vdim;
props["defines/ORDERING_BY_NODES"] = 0;
props["defines/ORDERING_BY_VDIM"] = 1;
props["defines/VDIM_ORDERING"] = (int) (ordering == Ordering::byVDIM);
::occa::device device = GetDevice();
const std::string &okl_path = OccaEngine().GetOklPath();
const std::string &okl_defines = OccaEngine().GetOklDefines();
globalToLocalKernel = device.buildKernel(okl_path + "fespace.okl",
"GlobalToLocal",
props + okl_defines);
localToGlobalKernel = device.buildKernel(okl_path + "fespace.okl",
"LocalToGlobal",
props + okl_defines);
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-146
View File
@@ -1,146 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_FE_SPACE_HPP
#define MFEM_BACKENDS_OCCA_FE_SPACE_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "engine.hpp"
#include "operator.hpp"
#include "../../fem/fem.hpp"
namespace mfem
{
namespace occa
{
/// TODO: doxygen
class FiniteElementSpace : public mfem::PFiniteElementSpace
{
protected:
//
// Inherited fields
//
// SharedPtr<const mfem::Engine> engine;
// mfem::FiniteElementSpace *fes;
Layout e_layout;
int *elementDofMap;
int *elementDofMapInverse;
::occa::array<int> globalToLocalOffsets;
::occa::array<int> globalToLocalIndices;
::occa::array<int> localToGlobalMap;
::occa::kernel globalToLocalKernel, localToGlobalKernel;
mfem::Ordering::Type ordering;
int globalDofs, localDofs;
int vdim;
mfem::Operator *restrictionOp, *prolongationOp;
void SetupLocalGlobalMaps();
void SetupOperators();
void SetupKernels();
public:
/// TODO: doxygen
FiniteElementSpace(const Engine &e, mfem::FiniteElementSpace &fespace);
/// Virtual destructor
virtual ~FiniteElementSpace();
/// TODO: doxygen
const Engine &OccaEngine() const
{ return *static_cast<const Engine *>(engine.Get()); }
/// TODO: doxygen
::occa::device GetDevice(int idx = 0) const
{ return OccaEngine().GetDevice(idx); }
mfem::Mesh* GetMesh() const { return fes->GetMesh(); }
Layout &OccaVLayout() const
{ return *fes->GetVLayout().As<Layout>(); }
Layout &OccaTrueVLayout() const
{ return *fes->GetTrueVLayout().As<Layout>(); }
Layout &OccaEVLayout() { return e_layout; }
#ifdef MFEM_USE_MPI
bool isDistributed() const { return (OccaEngine().GetComm() != MPI_COMM_NULL); }
#else
bool isDistributed() const { return false; }
#endif
bool hasTensorBasis() const
{ return dynamic_cast<const mfem::TensorBasisElement*>(fes->GetFE(0)); }
mfem::Ordering::Type GetOrdering() const { return ordering; }
int GetGlobalDofs() const { return globalDofs; }
int GetLocalDofs() const { return localDofs; }
int GetDim() const { return fes->GetMesh()->Dimension(); }
int GetVDim() const { return vdim; }
int GetVSize() const { return globalDofs * vdim; }
int GetTrueVSize() const { return fes->GetTrueVSize(); }
int GetGlobalVSize() const { return globalDofs*vdim; /* FIXME: MPI */ }
int GetGlobalTrueVSize() const { return fes->GetTrueVSize(); }
int GetNE() const { return fes->GetNE(); }
const mfem::FiniteElementCollection* FEColl() const
{ return fes->FEColl(); }
const mfem::FiniteElement* GetFE(const int idx) const
{ return fes->GetFE(idx); }
const int* GetElementDofMap() const { return elementDofMap; }
const int* GetElementDofMapInverse() const { return elementDofMapInverse; }
const mfem::Operator* GetRestrictionOperator() { return restrictionOp; }
const mfem::Operator* GetProlongationOperator() { return prolongationOp; }
const ::occa::array<int> GetLocalToGlobalMap() const
{ return localToGlobalMap; }
void GlobalToLocal(const Vector &globalVec, Vector &localVec) const
{
globalToLocalKernel(globalDofs,
localDofs * fes->GetNE(),
globalToLocalOffsets,
globalToLocalIndices,
globalVec.OccaMem(), localVec.OccaMem());
}
void LocalToGlobal(const Vector &localVec, Vector &globalVec) const
{
localToGlobalKernel(globalDofs,
localDofs * fes->GetNE(),
globalToLocalOffsets,
globalToLocalIndices,
localVec.OccaMem(), globalVec.OccaMem());
}
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_FE_SPACE_HPP
-67
View File
@@ -1,67 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
/*
---[ Defines Known At Compile-Time ]------------
TILESIZE : Tilesize for iterating over entries
================================================
*/
#if VDIM_ORDERING == ORDERING_BY_VDIM
typedef double *Global_t @dim(NUM_VDIM, globalEntries);
typedef double *Local_t @dim(NUM_VDIM, localEntries);
#else
typedef double *Global_t @dim(NUM_VDIM, globalEntries) @dimOrder(1, 0);
typedef double *Local_t @dim(NUM_VDIM, localEntries) @dimOrder(1, 0);
#endif
@kernel void GlobalToLocal(const int globalEntries,
const int localEntries,
const int * restrict offsets,
const int * restrict indices,
const Global_t restrict globalX,
Local_t restrict localX) {
for (int i = 0; i < globalEntries; ++i; @tile(TILESIZE, @outer, @inner)) {
if (i < globalEntries) {
const int offset = offsets[i];
const int nextOffset = offsets[i + 1];
for (int v = 0; v < NUM_VDIM; ++v) {
const double dofValue = globalX(v, i);
for (int j = offset; j < nextOffset; ++j) {
localX(v, indices[j]) = dofValue;
}
}
}
}
}
@kernel void LocalToGlobal(const int globalEntries,
const int localEntries,
const int * restrict offsets,
const int * restrict indices,
const Local_t restrict localX,
Global_t restrict globalX) {
for (int i = 0; i < globalEntries; ++i; @tile(TILESIZE, @outer, @inner)) {
if (i < globalEntries) {
const int offset = offsets[i];
const int nextOffset = offsets[i + 1];
for (int v = 0; v < NUM_VDIM; ++v) {
double dofValue = 0;
for (int j = offset; j < nextOffset; ++j) {
dofValue += localX(v, indices[j]);
}
globalX(v, i) = dofValue;
}
}
}
}
-181
View File
@@ -1,181 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef STORE_JACOBIAN
# define STORE_JACOBIAN 1
#endif
#ifndef STORE_JACOBIAN_INV
# define STORE_JACOBIAN_INV 1
#endif
#ifndef STORE_JACOBIAN_DET
# define STORE_JACOBIAN_DET 1
#endif
typedef double* Local1D_t @dim(1, NUM_DOFS, numElements);
typedef double* Local2D_t @dim(2, NUM_DOFS, numElements);
typedef double* Local3D_t @dim(3, NUM_DOFS, numElements);
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
typedef double* DofToQuadD1D_t @dim(NUM_QUAD, NUM_DOFS);
typedef double* DofToQuadD2D_t @dim(2, NUM_QUAD, NUM_DOFS);
typedef double* DofToQuadD3D_t @dim(3, NUM_QUAD, NUM_DOFS);
typedef double* Jacobian1D_t @dim(NUM_QUAD, numElements);
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD, numElements);
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD, numElements);
@kernel void InitGeometryInfo1D(const int numElements,
const DofToQuadD1D_t restrict dofToQuadD,
const Local1D_t restrict nodes,
Jacobian1D_t restrict J,
Jacobian1D_t restrict invJ,
QLocal_t restrict detJ) {
for (int e = 0; e < numElements; ++e; @outer) {
@shared double s_nodes[NUM_DOFS];
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
s_nodes[d] = nodes(0, d, e);
}
}
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
double J11 = 0;
for (int d = 0; d < NUM_DOFS; ++d) {
const double wx = dofToQuadD(q, d);
J11 += wx * s_nodes[d];
}
#if STORE_JACOBIAN
J(q, e) = J11;
#endif
#if STORE_JACOBIAN_INV
invJ(q, e) = 1.0 / J11;
#endif
#if STORE_JACOBIAN_DET
detJ(q, e) = J11;
#endif
}
}
}
@kernel void InitGeometryInfo2D(const int numElements,
const DofToQuadD2D_t restrict dofToQuadD,
const Local2D_t restrict nodes,
Jacobian2D_t restrict J,
Jacobian2D_t restrict invJ,
QLocal_t restrict detJ) {
for (int e = 0; e < numElements; ++e; @outer) {
@shared double s_nodes[2 * NUM_DOFS] @dim(2, NUM_DOFS);
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
s_nodes(0, d) = nodes(0, d, e);
s_nodes(1, d) = nodes(1, d, e);
}
}
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
double J11 = 0, J12 = 0;
double J21 = 0, J22 = 0;
for (int d = 0; d < NUM_DOFS; ++d) {
const double wx = dofToQuadD(0, q, d);
const double wy = dofToQuadD(1, q, d);
const double x = s_nodes(0, d);
const double y = s_nodes(1, d);
J11 += (wx * x); J12 += (wx * y);
J21 += (wy * x); J22 += (wy * y);
}
#if STORE_JACOBIAN_INV || STORE_JACOBIAN_DET
const double r_detJ = (J11 * J22) - (J12 * J21);
#endif
#if STORE_JACOBIAN
J(0, 0, q, e) = J11; J(1, 0, q, e) = J12;
J(0, 1, q, e) = J21; J(1, 1, q, e) = J22;
#endif
#if STORE_JACOBIAN_INV
const double r_idetJ = 1.0 / r_detJ;
invJ(0, 0, q, e) = J22 * r_idetJ;
invJ(1, 0, q, e) = -J12 * r_idetJ;
invJ(0, 1, q, e) = -J21 * r_idetJ;
invJ(1, 1, q, e) = J11 * r_idetJ;
#endif
#if STORE_JACOBIAN_DET
detJ(q, e) = r_detJ;
#endif
}
}
}
@kernel void InitGeometryInfo3D(const int numElements,
const DofToQuadD3D_t restrict dofToQuadD,
const Local3D_t restrict nodes,
Jacobian3D_t restrict J,
Jacobian3D_t restrict invJ,
QLocal_t restrict detJ) {
for (int e = 0; e < numElements; ++e; @outer) {
@shared double s_nodes[3 * NUM_DOFS] @dim(3, NUM_DOFS);
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
s_nodes(0, d) = nodes(0, d, e);
s_nodes(1, d) = nodes(1, d, e);
s_nodes(2, d) = nodes(2, d, e);
}
}
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
double J11 = 0, J12 = 0, J13 = 0;
double J21 = 0, J22 = 0, J23 = 0;
double J31 = 0, J32 = 0, J33 = 0;
for (int d = 0; d < NUM_DOFS; ++d) {
const double wx = dofToQuadD(0, q, d);
const double wy = dofToQuadD(1, q, d);
const double wz = dofToQuadD(2, q, d);
const double x = s_nodes(0, d);
const double y = s_nodes(1, d);
const double z = s_nodes(2, d);
J11 += (wx * x); J12 += (wx * y); J13 += (wx * z);
J21 += (wy * x); J22 += (wy * y); J23 += (wy * z);
J31 += (wz * x); J32 += (wz * y); J33 += (wz * z);
}
#if STORE_JACOBIAN_INV || STORE_JACOBIAN_DET
const double r_detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
#endif
#if STORE_JACOBIAN
J(0, 0, q, e) = J11; J(1, 0, q, e) = J12; J(2, 0, q, e) = J13;
J(0, 1, q, e) = J21; J(1, 1, q, e) = J22; J(2, 1, q, e) = J23;
J(0, 2, q, e) = J31; J(1, 2, q, e) = J32; J(2, 2, q, e) = J33;
#endif
#if STORE_JACOBIAN_INV
const double r_idetJ = 1.0 / r_detJ;
invJ(0, 0, q, e) = r_idetJ * ((J22 * J33) - (J23 * J32));
invJ(1, 0, q, e) = r_idetJ * ((J32 * J13) - (J33 * J12));
invJ(2, 0, q, e) = r_idetJ * ((J12 * J23) - (J13 * J22));
invJ(0, 1, q, e) = r_idetJ * ((J23 * J31) - (J21 * J33));
invJ(1, 1, q, e) = r_idetJ * ((J33 * J11) - (J31 * J13));
invJ(2, 1, q, e) = r_idetJ * ((J13 * J21) - (J11 * J23));
invJ(0, 2, q, e) = r_idetJ * ((J21 * J32) - (J22 * J31));
invJ(1, 2, q, e) = r_idetJ * ((J31 * J12) - (J32 * J11));
invJ(2, 2, q, e) = r_idetJ * ((J11 * J22) - (J12 * J21));
#endif
#if STORE_JACOBIAN_DET
detJ(q, e) = r_detJ;
#endif
}
}
}
-195
View File
@@ -1,195 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "gridfunc.hpp"
#include "bilininteg.hpp"
#include "../../fem/gridfunc.hpp"
namespace mfem
{
namespace occa
{
std::map<std::string, ::occa::kernel> gridFunctionKernels;
::occa::kernel GetGridFunctionKernel(::occa::device device,
FiniteElementSpace &fespace,
const mfem::IntegrationRule &ir)
{
const int numQuad = ir.GetNPoints();
const FiniteElement &fe = *(fespace.GetFE(0));
const int dim = fe.GetDim();
const int vdim = fespace.GetVDim();
std::stringstream ss;
ss << ::occa::hash(device)
<< "FEColl : " << fespace.FEColl()->Name()
<< "Quad: " << numQuad
<< "Dim: " << dim
<< "VDim: " << vdim;
std::string hash = ss.str();
// Kernel defines
::occa::properties props;
props["defines/NUM_VDIM"] = vdim;
SetProperties(fespace, ir, props);
::occa::kernel kernel = gridFunctionKernels[hash];
if (!kernel.isInitialized())
{
const std::string &okl_path = fespace.OccaEngine().GetOklPath();
kernel = device.buildKernel(okl_path + "gridfunc.okl",
stringWithDim("GridFuncToQuad", dim),
props);
}
return kernel;
}
// OccaGridFunction::OccaGridFunction() :
// Vector(),
// ofespace(NULL),
// sequence(0) {}
OccaGridFunction::OccaGridFunction(FiniteElementSpace *ofespace_)
: PArray(ofespace_->OccaVLayout()),
Array(ofespace_->OccaVLayout(), sizeof(double)),
Vector(ofespace_->OccaVLayout()),
ofespace(ofespace_),
sequence(0) {}
// OccaGridFunction::OccaGridFunction(OccaFiniteElementSpace *ofespace_,
// OccaVectorRef ref) :
// OccaVector(ref),
// ofespace(ofespace_),
// sequence(0) {}
OccaGridFunction::OccaGridFunction(const OccaGridFunction &v)
: PArray(v),
Array(v),
Vector(v),
ofespace(v.ofespace),
sequence(v.sequence) {}
OccaGridFunction& OccaGridFunction::operator = (double value)
{
Fill(value);
return *this;
}
OccaGridFunction& OccaGridFunction::operator = (const Vector &v)
{
Assign<double>(v);
return *this;
}
// OccaGridFunction& OccaGridFunction::operator = (const OccaVectorRef &v)
// {
// OccaVector::operator = (v);
// return *this;
// }
OccaGridFunction& OccaGridFunction::operator = (const OccaGridFunction &v)
{
Assign<double>(v);
return *this;
}
// void OccaGridFunction::SetGridFunction(mfem::GridFunction &gf)
// {
// Vector v = *this;
// gf.MakeRef(ofespace->GetFESpace(), v, 0);
// // Make gf the owner of the data
// v.Swap(gf);
// }
void OccaGridFunction::GetTrueDofs(Vector &v)
{
const mfem::Operator *R = ofespace->GetRestrictionOperator();
if (!R)
{
v.MakeRef(*this);
}
else
{
v.Resize<double>(R->OutLayout(), NULL);
mfem::Vector mfem_v(v);
R->Mult(this->Wrap(), mfem_v);
}
}
void OccaGridFunction::SetFromTrueDofs(Vector &v)
{
const mfem::Operator *P = ofespace->GetProlongationOperator();
if (!P)
{
MakeRef(v);
}
else
{
Resize<double>(P->OutLayout(), NULL);
mfem::Vector mfem_this(*this);
P->Mult(v.Wrap(), mfem_this);
}
}
mfem::FiniteElementSpace* OccaGridFunction::GetFESpace()
{
return ofespace->GetFESpace();
}
const mfem::FiniteElementSpace* OccaGridFunction::GetFESpace() const
{
return ofespace->GetFESpace();
}
void OccaGridFunction::ToQuad(const IntegrationRule &ir, Vector &quadValues)
{
const Engine &engine = OccaLayout().OccaEngine();
::occa::device device = engine.GetDevice();
OccaDofQuadMaps &maps = OccaDofQuadMaps::Get(device, *ofespace, ir);
const int elements = ofespace->GetNE();
const int numQuad = ir.GetNPoints();
quadValues.Resize<double>(*(new Layout(engine, numQuad * elements)), NULL);
::occa::kernel g2qKernel = GetGridFunctionKernel(device, *ofespace, ir);
g2qKernel(elements,
maps.dofToQuad,
ofespace->GetLocalToGlobalMap(),
this->OccaMem(),
quadValues.OccaMem());
}
void OccaGridFunction::Distribute(const Vector &v)
{
if (ofespace->isDistributed())
{
mfem::Vector mfem_this(*this);
ofespace->GetProlongationOperator()->Mult(v.Wrap(), mfem_this);
}
else
{
*this = v;
}
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-83
View File
@@ -1,83 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
#define MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "vector.hpp"
#include "fespace.hpp"
namespace mfem
{
class IntegrationRule;
class GridFunction;
namespace occa
{
class OccaIntegrator;
class OccaDofQuadMaps;
// TODO: make this object part of the backend or the engine.
extern std::map<std::string, ::occa::kernel> gridFunctionKernels;
// TODO: make this a method of the backend or the engine.
::occa::kernel GetGridFunctionKernel(::occa::device device,
FiniteElementSpace &fespace,
const mfem::IntegrationRule &ir);
class OccaGridFunction : public Vector
{
protected:
FiniteElementSpace *ofespace;
long sequence;
::occa::kernel gridFuncToQuad[3];
public:
// OccaGridFunction();
OccaGridFunction(FiniteElementSpace *ofespace_);
// OccaGridFunction(FiniteElementSpace *ofespace_,
// OccaVectorRef ref);
OccaGridFunction(const OccaGridFunction &gf);
OccaGridFunction& operator = (double value);
OccaGridFunction& operator = (const Vector &v);
// OccaGridFunction& operator = (const OccaVectorRef &v);
OccaGridFunction& operator = (const OccaGridFunction &gf);
// void SetGridFunction(mfem::GridFunction &gf);
void GetTrueDofs(Vector &v);
void SetFromTrueDofs(Vector &v);
mfem::FiniteElementSpace* GetFESpace();
const mfem::FiniteElementSpace* GetFESpace() const;
void ToQuad(const mfem::IntegrationRule &ir, Vector &quadValues);
void Distribute(const Vector &v);
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
-26
View File
@@ -1,26 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mfem-occa://defines.okl"
#if USING_TENSOR_OPS
# if OCCA_USING_CPU
# include "mfem-occa://gridfunc/tensor/cpu.okl"
# else
# include "mfem-occa://gridfunc/tensor/gpuHighOrder.okl"
# endif
#else
# if OCCA_USING_CPU
# include "mfem-occa://gridfunc/simplex/cpu.okl"
# else
# include "mfem-occa://gridfunc/simplex/gpuHighOrder.okl"
# endif
#endif
-63
View File
@@ -1,63 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mfem-occa://defines.okl"
//---[ 2D ]-----------------------------
@kernel void GridFuncToQuad2D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DLocalMap_t restrict l2gMap,
const double * restrict gf,
QVLocal_t restrict out) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
for (int d = 0; d < NUM_DOFS; ++d) {
const int gid = l2gMap(d, e);
for (int v = 0; v < NUM_VDIM; ++v) {
const double r_gf = gf[v + gid*NUM_VDIM];
double r_out = 0;
for (int q = 0; q < NUM_QUAD; ++q) {
r_out += r_gf * dofToQuad(d, q);
}
out(v, d, e) = r_out;
}
}
}
}
}
//======================================
//---[ 3D ]-----------------------------
@kernel void GridFuncToQuad3D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DLocalMap_t restrict l2gMap,
const double * restrict gf,
QVLocal_t restrict out) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
for (int d = 0; d < NUM_DOFS; ++d) {
const int gid = l2gMap(d, e);
for (int v = 0; v < NUM_VDIM; ++v) {
const double r_gf = gf[v + gid*NUM_VDIM];
double r_out = 0;
for (int q = 0; q < NUM_QUAD; ++q) {
r_out += r_gf * dofToQuad(d, q);
}
out(v, d, e) = r_out;
}
}
}
}
}
//======================================
@@ -1,79 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mfem-occa://defines.okl"
//---[ 2D ]-----------------------------
@kernel void GridFuncToQuad2D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DLocalMap_t restrict l2gMap,
const double * restrict gf,
QVLocal_t restrict out) {
for (int e = 0; e < numElements; ++e; @outer) {
@shared double s_gf[NUM_VDIM][NUM_DOFS];
for (int dOff = 0; dOff < M2_INNER_BATCH; ++dOff; @inner) {
for (int d = dOff; d < NUM_DOFS; d += M2_INNER_BATCH) {
const int gid = l2gMap(d, e);
for (int v = 0; v < NUM_VDIM; ++v) {
s_gf[v][d] = gf[v + gid*NUM_VDIM]];
}
}
}
for (int qOff = 0; qOff < M2_INNER_BATCH; ++qOff; @inner) {
for (int q = qOff; q < NUM_QUAD; q += M2_INNER_BATCH) {
for (int v = 0; v < NUM_VDIM; ++v) {
double r_out = 0;
for (int d = 0; d < NUM_DOFS; ++d) {
r_out += s_gf[v][d] * dofToQuad(d, q);
}
out(v, q, e) = r_out;
}
}
}
}
}
//======================================
//---[ 3D ]-----------------------------
@kernel void GridFuncToQuad3D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DLocalMap_t restrict l2gMap,
const double * restrict gf,
QVLocal_t restrict out) {
for (int e = 0; e < numElements; ++e; @outer) {
@shared double s_gf[NUM_VDIM][NUM_DOFS];
for (int dOff = 0; dOff < M3_INNER_BATCH; ++dOff; @inner) {
for (int d = dOff; d < NUM_DOFS; d += M3_INNER_BATCH) {
const int gid = l2gMap(d, e);
for (int v = 0; v < NUM_VDIM; ++v) {
s_gf[v][d] = gf[v + gid*NUM_VDIM]];
}
}
}
for (int qOff = 0; qOff < M3_INNER_BATCH; ++qOff; @inner) {
for (int q = qOff; q < NUM_QUAD; q += M3_INNER_BATCH) {
for (int v = 0; v < NUM_VDIM; ++v) {
double r_out = 0;
for (int d = 0; d < NUM_DOFS; ++d) {
r_out += s_gf[v][d] * dofToQuad(d, q);
}
out(v, q, e) = r_out;
}
}
}
}
}
//======================================
-188
View File
@@ -1,188 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mfem-occa://defines.okl"
//---[ 1D ]-----------------------------
@kernel void GridFuncToQuad1D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DLocalMap1D_t restrict l2gMap,
const double * restrict gf,
QVLocal1D_t restrict out) {
for (int e = 0; e < numElements; ++e; @outer) {
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
double r_out[NUM_VDIM][NUM_QUAD_1D];
for (int v = 0; v < NUM_VDIM; ++v) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
r_out[v][qx] = 0;
}
}
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const int gid = l2gMap(dx, e);
for (int v = 0; v < NUM_VDIM; ++v) {
const double r_gf = gf[v + gid*NUM_VDIM];
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
r_out[v][qx] += r_gf * dofToQuad(qx, dx);
}
}
}
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
for (int v = 0; v < NUM_VDIM; ++v) {
out(v, qx, e) = r_out[v][qx];
}
}
}
}
}
//======================================
//---[ 2D ]-----------------------------
@kernel void GridFuncToQuad2D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DLocalMap2D_t restrict l2gMap,
const double * restrict gf,
QVLocal2D_t restrict out) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
double out_xy[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D];
for (int v = 0; v < NUM_VDIM; ++v) {
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
out_xy[v][qy][qx] = 0;
}
}
}
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
double out_x[NUM_VDIM][NUM_QUAD_1D];
for (int v = 0; v < NUM_VDIM; ++v) {
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
out_x[v][qy] = 0;
}
}
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const int gid = l2gMap(dx, dy, e);
for (int v = 0; v < NUM_VDIM; ++v) {
const double r_gf = gf[v + gid*NUM_VDIM];
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
out_x[v][qy] += r_gf * dofToQuad(qy, dx);
}
}
}
for (int v = 0; v < NUM_VDIM; ++v) {
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
const double d2q = dofToQuad(qy, dy);
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
out_xy[v][qy][qx] += d2q * out_x[v][qx];
}
}
}
}
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
for (int v = 0; v < NUM_VDIM; ++v) {
out(v, qx, qy, e) = out_xy[v][qy][qx];
}
}
}
}
}
}
//======================================
//---[ 3D ]-----------------------------
@kernel void GridFuncToQuad3D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DLocalMap3D_t restrict l2gMap,
const double * restrict gf,
QVLocal3D_t restrict out) {
for (int e = 0; e < numElements; ++e; @outer) {
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
double out_xyz[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D];
for (int v = 0; v < NUM_VDIM; ++v) {
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
out_xyz[v][qz][qy][qx] = 0;
}
}
}
}
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
double out_xy[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D];
for (int v = 0; v < NUM_VDIM; ++v) {
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
out_xy[v][qy][qx] = 0;
}
}
}
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
double out_x[NUM_VDIM][NUM_QUAD_1D];
for (int v = 0; v < NUM_VDIM; ++v) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
out_x[v][qx] = 0;
}
}
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const int gid = l2gMap(dx, dy, dz, e);
for (int v = 0; v < NUM_VDIM; ++v) {
const double r_gf = gf[v + gid*NUM_VDIM];
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
out_x[v][qx] += r_gf * dofToQuad(qx, dx);
}
}
}
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
const double wy = dofToQuad(qy, dy);
for (int v = 0; v < NUM_VDIM; ++v) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
out_xy[v][qy][qx] += wy * out_x[v][qx];
}
}
}
}
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
const double wz = dofToQuad(qz, dz);
for (int v = 0; v < NUM_VDIM; ++v) {
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
out_xyz[v][qz][qy][qx] += wz * out_xy[v][qy][qx];
}
}
}
}
}
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
for (int v = 0; v < NUM_VDIM; ++v) {
out(v, qx, qy, qz, e) = out_xyz[v][qz][qy][qx];
}
}
}
}
}
}
}
//======================================
@@ -1,183 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mfem-occa://defines.okl"
//---[ 1D ]-----------------------------
@kernel void GridFuncToQuad1D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DLocalMap1D_t restrict l2gMap,
const double * restrict gf,
QLocal1D_t restrict out) {
// Iterate over elements
for (int eOff = 0; eOff < numElements; eOff += M1_ELEMENT_BATCHES; @outer) {
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
@exclusive double r_out[NUM_QUAD_1D];
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
for (int i = el; i < NUM_QUAD_DOFS_1D; i += M1_INNER_ELEMENT_BATCH) {
s_dofToQuad[i] = dofToQuad[i];
}
}
for (int b = 0; b < M1_OUTER_ELEMENT_BATCH; ++b) {
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
const int e = eOff + b*M1_INNER_ELEMENT_BATCH + el;
if (e < numElements) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
r_out[qx] = 0;
}
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const double r_gf = gf[l2gMap(dx, e)];
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
r_out[qx] += r_gf * s_dofToQuad(qx, dx);
}
}
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
out(qx, e) = r_out[qx];
}
}
}
}
}
}
//======================================
//---[ 2D ]-----------------------------
@kernel void GridFuncToQuad2D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DLocalMap2D_t restrict l2gMap,
const double * restrict gf,
QLocal2D_t restrict out) {
// Iterate over elements
for (int eOff = 0; eOff < numElements; eOff += M2_ELEMENT_BATCH; @outer) {
// Store dof <--> quad mappings
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
// Store xy planes in shared memory
@shared double s_xy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
for (int id = x; id < NUM_QUAD_DOFS_1D; id += NUM_MAX_1D) {
s_dofToQuad[id] = dofToQuad[id];
}
}
for (int e = eOff; e < (eOff + M2_ELEMENT_BATCH); ++e) {
if (e < numElements) {
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
if (dx < NUM_DOFS_1D) {
double r_x[NUM_DOFS_1D];
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
r_x[dy] = gf[l2gMap(dx, dy, e)];
}
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
double xy = 0;
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
xy += r_x[dy] * s_dofToQuad(qy, dy);
}
s_xy(dx, qy) = xy;
}
}
}
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
if (qy < NUM_QUAD_1D) {
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
double val = 0;
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
val += s_xy(dx, qy) * s_dofToQuad(qx, dx);
}
out(qx, qy, e) = val;
}
}
}
}
}
}
}
//======================================
//---[ 3D ]-----------------------------
@kernel void GridFuncToQuad3D(const int numElements,
const DofToQuad_t restrict dofToQuad,
const DLocalMap3D_t restrict l2gMap,
const double * restrict gf,
QLocal3D_t restrict out) {
// Iterate over elements
for (int e = 0; e < numElements; ++e; @outer) {
// Store dof <--> quad mappings
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
// Store xy planes in shared memory
@shared double s_z[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
// Store z axis as registers
@exclusive double r_qz[NUM_QUAD_1D];
for (int y = 0; y < NUM_MAX_1D; ++y; @inner) {
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
const int id = (y * NUM_MAX_1D) + x;
// Fetch Q <--> D maps
if (id < NUM_QUAD_DOFS_1D) {
s_dofToQuad[id] = dofToQuad[id];
}
// Initialize our Z axis
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
r_qz[qz] = 0;
}
}
}
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
const double val = gf[l2gMap(dx, dy, dz, e)];
// Calculate D -> Q in the Z axis
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
r_qz[qz] += val * s_dofToQuad(qz, dz);
}
}
}
}
}
// For each xy plane
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
// Fill xy plane at given z position
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
s_z(dx, dy) = r_qz[qz];
}
}
}
// Calculate Dxyz, xDyz, xyDz in plane
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
double val = 0;
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
const double wy = s_dofToQuad(qy, dy);
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
const double wx = s_dofToQuad(qx, dx);
val += wx * wy * s_z(dx, dy);
}
}
out(qx, qy, qz, e) = val;
}
}
}
}
}
}
//======================================
-162
View File
@@ -1,162 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "interpolation.hpp"
namespace mfem
{
namespace occa
{
void CreateRPOperators(Layout &v_layout, Layout &t_layout,
const mfem::SparseMatrix *R, const mfem::Operator *P,
mfem::Operator *&OccaR, mfem::Operator *&OccaP)
{
if (!P)
{
OccaR = new IdentityOperator(t_layout);
OccaP = new IdentityOperator(t_layout);
return;
}
const mfem::SparseMatrix *pmat = dynamic_cast<const mfem::SparseMatrix*>(P);
::occa::device device = v_layout.OccaEngine().GetDevice();
if (R)
{
OccaSparseMatrix *occaR =
CreateMappedSparseMatrix(v_layout, t_layout, *R);
::occa::array<int> reorderIndices = occaR->reorderIndices;
delete occaR;
OccaR = new RestrictionOperator(v_layout, t_layout, reorderIndices);
}
if (pmat)
{
const mfem::SparseMatrix *pmatT = Transpose(*pmat);
OccaSparseMatrix *occaP =
CreateMappedSparseMatrix(t_layout, v_layout, *pmat);
OccaSparseMatrix *occaPT =
CreateMappedSparseMatrix(v_layout, t_layout, *pmatT);
OccaP = new ProlongationOperator(*occaP, *occaPT);
}
else
{
OccaP = new ProlongationOperator(t_layout, v_layout, P);
}
}
RestrictionOperator::RestrictionOperator(Layout &in_layout, Layout &out_layout,
::occa::array<int> indices) :
Operator(in_layout, out_layout)
{
entries = indices.size() / 2;
trueIndices = indices;
// FIXME: paths ...
::occa::device device = in_layout.OccaEngine().GetDevice();
const std::string &okl_path = in_layout.OccaEngine().GetOklPath();
const std::string &okl_defines = in_layout.OccaEngine().GetOklDefines();
multOp = device.buildKernel(okl_path + "mappings.okl",
"ExtractSubVector",
"defines: { TILESIZE: 256 }" + okl_defines);
multTransposeOp = device.buildKernel(okl_path + "mappings.okl",
"SetSubVector",
"defines: { TILESIZE: 256 }" +
okl_defines);
}
void RestrictionOperator::Mult_(const Vector &x, Vector &y) const
{
multOp(entries, trueIndices, x.OccaMem(), y.OccaMem());
}
void RestrictionOperator::MultTranspose_(const Vector &x, Vector &y) const
{
y.Fill<double>(0.0);
multTransposeOp(entries, trueIndices, x.OccaMem(), y.OccaMem());
}
ProlongationOperator::ProlongationOperator(OccaSparseMatrix &multOp_,
OccaSparseMatrix &multTransposeOp_) :
Operator(multOp_),
pmat(NULL),
multOp(multOp_),
multTransposeOp(multTransposeOp_) {}
ProlongationOperator::ProlongationOperator(Layout &in_layout,
Layout &out_layout,
const mfem::Operator *pmat_) :
Operator(in_layout, out_layout),
pmat(pmat_),
multOp(*this),
multTransposeOp(*this)
{ }
void ProlongationOperator::Mult_(const Vector &x, Vector &y) const
{
MFEM_VERIFY(pmat == NULL, "");
multOp.Mult_(x, y);
}
void ProlongationOperator::MultTranspose_(const Vector &x, Vector &y) const
{
MFEM_VERIFY(pmat == NULL, "");
multTransposeOp.Mult_(x, y);
}
void ProlongationOperator::Mult(const mfem::Vector &x, mfem::Vector &y) const
{
if (pmat)
{
// FIXME: create an OCCA version of 'pmat'
x.Pull();
y.Pull(false);
pmat->Mult(x, y);
y.Push();
}
else
{
multOp.Mult(x, y);
}
}
void ProlongationOperator::MultTranspose(const mfem::Vector &x,
mfem::Vector &y) const
{
if (pmat)
{
// FIXME: create an OCCA version of 'pmat'
x.Pull();
y.Pull(false);
pmat->MultTranspose(x, y);
y.Push();
}
else
{
multTransposeOp.Mult(x, y);
}
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-79
View File
@@ -1,79 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
#define MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include <occa.hpp>
#include "vector.hpp"
#include "engine.hpp"
#include "sparsemat.hpp"
#include "../../fem/fem.hpp"
namespace mfem
{
namespace occa
{
// [MISSING] Proper destructors
void CreateRPOperators(Layout &v_layout, Layout &t_layout,
const mfem::SparseMatrix *R, const mfem::Operator *P,
mfem::Operator *&OccaR, mfem::Operator *&OccaP);
class RestrictionOperator : public Operator
{
protected:
int entries;
::occa::array<int> trueIndices;
::occa::kernel multOp, multTransposeOp;
public:
RestrictionOperator(Layout &in_layout, Layout &out_layout,
::occa::array<int> indices);
// overrides
virtual void Mult_(const Vector &x, Vector &y) const;
virtual void MultTranspose_(const Vector &x, Vector &y) const;
};
class ProlongationOperator : public Operator
{
protected:
const mfem::Operator *pmat;
OccaSparseMatrix multOp, multTransposeOp;
public:
ProlongationOperator(OccaSparseMatrix &multOp_,
OccaSparseMatrix &multTransposeOp_);
ProlongationOperator(Layout &in_layout, Layout &out_layout,
const mfem::Operator *pmat_);
// overrides
virtual void Mult_(const Vector &x, Vector &y) const;
virtual void MultTranspose_(const Vector &x, Vector &y) const;
// overrides
virtual void Mult(const mfem::Vector &x, mfem::Vector &y) const;
virtual void MultTranspose(const mfem::Vector &x, mfem::Vector &y) const;
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
-40
View File
@@ -1,40 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "layout.hpp"
#include "../../general/array.hpp"
namespace mfem
{
namespace occa
{
void Layout::Resize(std::size_t new_size)
{
size = new_size;
}
void Layout::Resize(const Array<std::size_t> &offsets)
{
MFEM_ASSERT(offsets.Size() == 2,
"multiple workers are not supported yet");
size = offsets.Last();
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-68
View File
@@ -1,68 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_LAYOUT_HPP
#define MFEM_BACKENDS_OCCA_LAYOUT_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "../base/layout.hpp"
#include "engine.hpp"
namespace mfem
{
namespace occa
{
class Layout : public PLayout
{
protected:
//
// Inherited fields
//
// SharedPtr<const mfem::Engine> engine;
// std::size_t size;
public:
Layout(const Engine &e, std::size_t s = 0) : PLayout(e, s) { }
const Engine &OccaEngine() const
{ return *static_cast<const Engine *>(engine.Get()); }
::occa::memory Alloc(std::size_t bytes) const
{ return OccaEngine().GetDevice().malloc(bytes); }
virtual ~Layout() { }
/**
@name Virtual interface
*/
///@{
/// Resize the layout
virtual void Resize(std::size_t new_size);
/// Resize the layout based on the given worker offsets
virtual void Resize(const Array<std::size_t> &offsets);
///@}
// End: Virtual interface
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_LAYOUT_HPP
-54
View File
@@ -1,54 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
/*
---[ Defines Known At Compile-Time ]------------
TILESIZE : Tilesize for iterating over entries
================================================
*/
@kernel void ExtractSubVector(const int entries,
const int * restrict indices,
const double * restrict in,
double * restrict out) {
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
if (i < entries) {
out[i] = in[indices[i]];
}
}
}
@kernel void SetSubVector(const int entries,
const int * restrict indices,
const double * restrict in,
double * restrict out) {
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
if (i < entries) {
out[indices[i]] = in[i];
}
}
}
@kernel void MapSubVector(const int entries,
const int * restrict indices,
const double * restrict in,
double * restrict out) {
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
if (i < entries) {
const int fromIdx = indices[2*i + 0];
const int toIdx = indices[2*i + 1];
out[toIdx] = in[fromIdx];
}
}
}
-135
View File
@@ -1,135 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "operator.hpp"
namespace mfem
{
namespace occa
{
// FIXME: move this object to the Backend?
::occa::kernelBuilder OccaConstrainedOperator::mapDofBuilder =
::occa::linalg::customLinearMethod(
"vector_map_dofs",
"const int idx = v2[i];"
"v0[idx] = v1[idx];",
"defines: {"
" VTYPE0: 'double',"
" VTYPE1: 'double',"
" VTYPE2: 'int',"
" TILESIZE: 128,"
"}");
// FIXME: move this object to the Backend?
::occa::kernelBuilder OccaConstrainedOperator::clearDofBuilder =
::occa::linalg::customLinearMethod(
"vector_clear_dofs",
"v0[v1[i]] = 0.0;",
"defines: {"
" VTYPE0: 'double',"
" VTYPE1: 'int',"
" TILESIZE: 128,"
"}");
OccaConstrainedOperator::OccaConstrainedOperator(
mfem::Operator *A_,
const mfem::Array<int> &constraintList_,
bool own_A_)
: Operator(A_->InLayout()->As<Layout>()),
z(OutLayout_()),
w(OutLayout_()),
mfem_z((z.DontDelete(), z)),
mfem_w((w.DontDelete(), w))
{
Setup(OutLayout_().OccaEngine().GetDevice(), A_, constraintList_, own_A_);
}
void OccaConstrainedOperator::Setup(::occa::device device_,
mfem::Operator *A_,
const mfem::Array<int> &constraintList_,
bool own_A_)
{
device = device_;
A = A_;
own_A = own_A_;
constraintIndices = constraintList_.Size();
constraintList = constraintList_.Get_PArray()->As<Array>().OccaMem();
}
void OccaConstrainedOperator::EliminateRHS(const Vector &x, Vector &b) const
{
const std::string &okl_defines = InLayout_().OccaEngine().GetOklDefines();
::occa::kernel mapDofs = mapDofBuilder.build(device, okl_defines);
w.Fill<double>(0.0);
if (constraintIndices)
{
mapDofs(constraintIndices, w.OccaMem(), x.OccaMem(), constraintList);
}
A->Mult(mfem_w, mfem_z);
b.Axpby<double>(1.0, b, -1.0, z);
if (constraintIndices)
{
mapDofs(constraintIndices, b.OccaMem(), x.OccaMem(), constraintList);
}
}
void OccaConstrainedOperator::Mult_(const Vector &x, Vector &y) const
{
mfem::Vector mfem_y(y);
if (constraintIndices == 0)
{
A->Mult(x.Wrap(), mfem_y);
return;
}
const std::string &okl_defines = InLayout_().OccaEngine().GetOklDefines();
::occa::kernel mapDofs = mapDofBuilder.build(device, okl_defines);
::occa::kernel clearDofs = clearDofBuilder.build(device, okl_defines);
z.Assign<double>(x); // z = x
clearDofs(constraintIndices, z.OccaMem(), constraintList);
A->Mult(mfem_z, mfem_y);
mapDofs(constraintIndices, y.OccaMem(), x.OccaMem(), constraintList);
}
OccaConstrainedOperator::~OccaConstrainedOperator()
{
if (own_A)
{
delete A;
}
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-129
View File
@@ -1,129 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_OPERATOR_HPP
#define MFEM_BACKENDS_OCCA_OPERATOR_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "vector.hpp"
#include "../../linalg/operator.hpp"
namespace mfem
{
namespace occa
{
class Operator : public mfem::Operator
{
public:
/// Creare an operator with the same dimensions as @a orig.
Operator(const Operator &orig)
: mfem::Operator(orig) { }
Operator(Layout &layout)
: mfem::Operator(layout) { }
Operator(Layout &in_layout, Layout &out_layout)
: mfem::Operator(in_layout, out_layout) { }
Layout &InLayout_() const
{ return *static_cast<Layout*>(in_layout.Get()); }
Layout &OutLayout_() const
{ return *static_cast<Layout*>(out_layout.Get()); }
virtual void Mult_(const Vector &x, Vector &y) const = 0;
virtual void MultTranspose_(const Vector &x, Vector &y) const
{ MFEM_ABORT("method is not supported"); }
// override
virtual void Mult(const mfem::Vector &x, mfem::Vector &y) const
{
Mult_(x.Get_PVector()->As<Vector>(),
y.Get_PVector()->As<Vector>());
}
// override
virtual void MultTranspose(const mfem::Vector &x, mfem::Vector &y) const
{
MultTranspose_(x.Get_PVector()->As<Vector>(),
y.Get_PVector()->As<Vector>());
}
};
class OccaConstrainedOperator : public Operator
{
protected:
::occa::device device;
mfem::Operator *A; //< The unconstrained Operator.
bool own_A; //< Ownership flag for A.
::occa::memory constraintList; //< List of constrained indices/dofs.
int constraintIndices;
mutable Vector z, w; //< Auxiliary vectors.
mutable mfem::Vector mfem_z, mfem_w; // Wrap z, w
static ::occa::kernelBuilder mapDofBuilder, clearDofBuilder;
public:
/** @brief Constructor from a general Operator and a list of essential
indices/dofs.
Specify the unconstrained operator @a *A and a @a list of indices to
constrain, i.e. each entry @a list[i] represents an essential-dof. If the
ownership flag @a own_A is true, the operator @a *A will be destroyed
when this object is destroyed. */
OccaConstrainedOperator(mfem::Operator *A_,
const mfem::Array<int> &constraintList_,
bool own_A_ = false);
void Setup(::occa::device device_,
mfem::Operator *A_,
const mfem::Array<int> &constraintList_,
bool own_A_ = false);
/** @brief Eliminate "essential boundary condition" values specified in @a x
from the given right-hand side @a b.
Performs the following steps:
z = A((0,x_b)); b_i -= z_i; b_b = x_b;
where the "_b" subscripts denote the essential (boundary) indices/dofs of
the vectors, and "_i" -- the rest of the entries. */
void EliminateRHS(const Vector &x, Vector &b) const;
/** @brief Constrained operator action.
Performs the following steps:
z = A((x_i,0)); y_i = z_i; y_b = x_b;
where the "_b" subscripts denote the essential (boundary) indices/dofs of
the vectors, and "_i" -- the rest of the entries. */
virtual void Mult_(const Vector &x, Vector &y) const;
// Destructor: destroys the unconstrained Operator @a A if @a own_A is true.
virtual ~OccaConstrainedOperator();
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_OPERATOR_HPP
-57
View File
@@ -1,57 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
/*
---[ Defines Known At Compile-Time ]------------
TILESIZE : Tilesize for iterating over dofs
================================================
*/
@kernel void Mult(const int entries,
const int * restrict offsets,
const int * restrict indices,
const double * restrict weights,
const double * restrict in,
double * restrict out) {
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
if (i < entries) {
const int offset = offsets[i];
const int nextOffset = offsets[i + 1];
double value = 0;
for (int j = offset; j < nextOffset; ++j) {
value += weights[j] * in[indices[j]];
}
out[i] = value;
}
}
}
@kernel void MappedMult(const int entries,
const int * restrict offsets,
const int * restrict indices,
const double * restrict weights,
const int * restrict outIndices,
const double * restrict in,
double * restrict out) {
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
if (i < entries) {
const int offset = offsets[i];
const int nextOffset = offsets[i + 1];
double value = 0;
for (int j = offset; j < nextOffset; ++j) {
value += weights[j] * in[indices[j]];
}
out[outIndices[i]] = value;
}
}
}
-248
View File
@@ -1,248 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "sparsemat.hpp"
namespace mfem
{
namespace occa
{
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
const mfem::SparseMatrix &m,
const ::occa::properties &props) :
Operator(in_layout, out_layout)
{
Setup(in_layout.OccaEngine().GetDevice(), m, props);
}
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
const mfem::SparseMatrix &m,
::occa::array<int> reorderIndices_,
::occa::array<int> mappedIndices_,
const ::occa::properties &props) :
Operator(in_layout, out_layout)
{
Setup(in_layout.OccaEngine().GetDevice(), m,
reorderIndices, mappedIndices_, props);
}
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
::occa::array<int> offsets_,
::occa::array<int> indices_,
::occa::array<double> weights_,
const ::occa::properties &props) :
Operator(in_layout, out_layout),
offsets(offsets_),
indices(indices_),
weights(weights_)
{
SetupKernel(in_layout.OccaEngine().GetDevice(), props);
}
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
::occa::array<int> offsets_,
::occa::array<int> indices_,
::occa::array<double> weights_,
::occa::array<int> reorderIndices_,
::occa::array<int> mappedIndices_,
const ::occa::properties &props) :
Operator(in_layout, out_layout),
offsets(offsets_),
indices(indices_),
weights(weights_),
reorderIndices(reorderIndices_),
mappedIndices(mappedIndices_)
{
SetupKernel(in_layout.OccaEngine().GetDevice(), props);
}
void OccaSparseMatrix::Setup(::occa::device device, const mfem::SparseMatrix &m,
const ::occa::properties &props)
{
Setup(device, m, ::occa::array<int>(), ::occa::array<int>(), props);
}
void OccaSparseMatrix::Setup(::occa::device device, const SparseMatrix &m,
::occa::array<int> reorderIndices_,
::occa::array<int> mappedIndices_,
const ::occa::properties &props)
{
const int nnz = m.GetI()[height];
offsets.allocate(device,
height + 1, m.GetI());
indices.allocate(device,
nnz, m.GetJ());
weights.allocate(device,
nnz, m.GetData());
offsets.keepInDevice();
indices.keepInDevice();
weights.keepInDevice();
reorderIndices = reorderIndices_;
mappedIndices = mappedIndices_;
SetupKernel(device, props);
}
void OccaSparseMatrix::SetupKernel(::occa::device device,
const ::occa::properties &props)
{
const bool hasOutIndices = mappedIndices.isInitialized();
const ::occa::properties defaultProps("defines: {"
" TILESIZE: 256,"
"}");
const std::string &okl_path = InLayout_().OccaEngine().GetOklPath();
const std::string &okl_defines = InLayout_().OccaEngine().GetOklDefines();
mapKernel = device.buildKernel(okl_path + "mappings.okl",
"MapSubVector",
defaultProps + props + okl_defines);
multKernel = device.buildKernel(okl_path + "sparse.okl",
hasOutIndices ? "MappedMult" : "Mult",
defaultProps + props + okl_defines);
}
void OccaSparseMatrix::Mult_(const Vector &x, Vector &y) const
{
if (reorderIndices.isInitialized() ||
mappedIndices.isInitialized())
{
if (reorderIndices.isInitialized())
{
mapKernel((int) (reorderIndices.size() / 2),
reorderIndices,
x.OccaMem(), y.OccaMem());
}
if (mappedIndices.isInitialized())
{
multKernel((int) (mappedIndices.size()),
offsets, indices, weights,
mappedIndices,
x.OccaMem(), y.OccaMem());
}
}
else
{
multKernel((int) height,
offsets, indices, weights,
x.OccaMem(), y.OccaMem());
}
}
OccaSparseMatrix* CreateMappedSparseMatrix(Layout &in_layout,
Layout &out_layout,
const mfem::SparseMatrix &m,
const ::occa::properties &props)
{
const int mHeight = m.Height();
// const int mWidth = m.Width();
// Count indices that are only reordered (true dofs)
const int *I = m.GetI();
const int *J = m.GetJ();
const double *D = m.GetData();
int trueCount = 0;
for (int i = 0; i < mHeight; ++i)
{
trueCount += ((I[i + 1] - I[i]) == 1);
}
const int dupCount = (mHeight - trueCount);
// Create the reordering map for entries that aren't modified (true dofs)
::occa::device device(in_layout.OccaEngine().GetDevice());
::occa::array<int> reorderIndices(device,
2 * trueCount);
::occa::array<int> mappedIndices, offsets, indices;
::occa::array<double> weights;
if (dupCount)
{
mappedIndices.allocate(device,
dupCount);
}
int trueIdx = 0, dupIdx = 0;
for (int i = 0; i < mHeight; ++i)
{
const int i1 = I[i];
if ((I[i + 1] - i1) == 1)
{
reorderIndices[trueIdx++] = J[i1];
reorderIndices[trueIdx++] = i;
}
else
{
mappedIndices[dupIdx++] = i;
}
}
reorderIndices.keepInDevice();
if (dupCount)
{
mappedIndices.keepInDevice();
// Extract sparse matrix without reordered identity
const int dupNnz = I[mHeight] - trueCount;
offsets.allocate(device,
dupCount + 1);
indices.allocate(device,
dupNnz);
weights.allocate(device,
dupNnz);
int nnz = 0;
offsets[0] = 0;
for (int i = 0; i < dupCount; ++i)
{
const int idx = mappedIndices[i];
const int offStart = I[idx];
const int offEnd = I[idx + 1];
offsets[i + 1] = offsets[i] + (offEnd - offStart);
for (int j = offStart; j < offEnd; ++j)
{
indices[nnz] = J[j];
weights[nnz] = D[j];
++nnz;
}
}
offsets.keepInDevice();
indices.keepInDevice();
weights.keepInDevice();
}
return new OccaSparseMatrix(in_layout, out_layout,
offsets, indices, weights,
reorderIndices, mappedIndices,
props);
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-95
View File
@@ -1,95 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
#define MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include <occa.hpp>
#include "vector.hpp"
#include "engine.hpp"
#include "operator.hpp"
#include "../../linalg/sparsemat.hpp"
namespace mfem
{
namespace occa
{
/// TODO: doxygen
class OccaSparseMatrix : public Operator
{
public:
::occa::array<int> offsets, indices;
::occa::array<double> weights;
::occa::array<int> reorderIndices, mappedIndices;
::occa::kernel mapKernel, multKernel;
/// Construct an empty OccaSparseMatrix.
OccaSparseMatrix(const Operator &orig)
: Operator(orig) { }
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
const mfem::SparseMatrix &m,
const ::occa::properties &props = ::occa::properties());
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
const mfem::SparseMatrix &m,
::occa::array<int> reorderIndices_,
::occa::array<int> mappedIndices_,
const ::occa::properties &props = ::occa::properties());
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
::occa::array<int> offsets_,
::occa::array<int> indices_,
::occa::array<double> weights_,
const ::occa::properties &props = ::occa::properties());
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
::occa::array<int> offsets_,
::occa::array<int> indices_,
::occa::array<double> weights_,
::occa::array<int> reorderIndices_,
::occa::array<int> mappedIndices_,
const ::occa::properties &props = ::occa::properties());
void Setup(::occa::device device, const mfem::SparseMatrix &m,
const ::occa::properties &props);
void Setup(::occa::device device, const mfem::SparseMatrix &m,
::occa::array<int> reorderIndices_,
::occa::array<int> mappedIndices_,
const ::occa::properties &props);
void SetupKernel(::occa::device device,
const ::occa::properties &props);
// override
virtual void Mult_(const Vector &x, Vector &y) const;
};
/// TODO: doxygen
OccaSparseMatrix* CreateMappedSparseMatrix(
Layout &in_layout, Layout &out_layout,
const mfem::SparseMatrix &m,
const ::occa::properties &props = ::occa::properties());
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
-81
View File
@@ -1,81 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "url_handler.hpp"
#include "../../general/error.hpp"
#include <cstdlib>
#include <sys/stat.h>
namespace mfem
{
namespace occa
{
FileOpener::FileOpener(const std::string &prefix,
const std::string &env_variable)
: pfx(prefix)
{
const char *env_path = getenv(env_variable.c_str());
if (!env_path) { return; }
std::string path(env_path);
for (std::size_t start = 0, end; start < path.size(); start = end + 1)
{
end = path.find(':', start);
if (end == std::string::npos)
{
AddDir(path.substr(start, end));
break;
}
AddDir(path.substr(start, end - start));
}
}
bool FileOpener::AddDir(const std::string &dir)
{
if (dir.size() == 0 || dir[0] != '/') { return false; }
struct stat dir_stat;
if (stat(dir.c_str(), &dir_stat)) { return false; }
if (!S_ISDIR(dir_stat.st_mode)) { return false; }
paths.push_back(dir + (*dir.rbegin() == '/' ? "" : "/"));
return true;
}
bool FileOpener::handles(const std::string &filename)
{
return filename.size() >= pfx.size() &&
filename.compare(0, pfx.size(), pfx) == 0;
}
std::string FileOpener::expand(const std::string &filename)
{
std::string sfx(filename.substr(pfx.size()));
for (std::size_t i = 0; i < paths.size(); i++)
{
std::string file = paths[i] + sfx;
struct stat file_stat;
if (stat(file.c_str(), &file_stat) == 0 && S_ISREG(file_stat.st_mode))
{
return file;
}
}
MFEM_ABORT("invalid url: " << filename);
return sfx;
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-47
View File
@@ -1,47 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
#define MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include <occa.hpp>
namespace mfem
{
namespace occa
{
class FileOpener : public ::occa::io::fileOpener
{
protected:
std::string pfx; // prefix, e.g. "mfem://"
std::vector<std::string> paths; // paths to search for prefix replacement
public:
FileOpener(const std::string &prefix, const std::string &env_variable);
bool AddDir(const std::string &dir);
virtual bool handles(const std::string &filename);
virtual std::string expand(const std::string &filename);
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
-22
View File
@@ -1,22 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
typedef double* Local_t @dim(numDofs, numElements);
@kernel void InitLocalVector(const int numElements,
const int numDofs,
Local_t restrict sol) {
for (int e = 0; e < numElements; ++e; @outer) {
for (int d = 0; d < numDofs; ++d; @inner) {
sol(d, e) = 0;
}
}
}
-204
View File
@@ -1,204 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include "vector.hpp"
#include "../../linalg/vector.hpp"
namespace mfem
{
namespace occa
{
PVector *Vector::DoVectorClone(bool copy_data, void **buffer,
int buffer_type_id) const
{
MFEM_ASSERT(buffer_type_id == ScalarId<double>::value, "");
Vector *new_vector = new Vector(OccaLayout());
if (copy_data)
{
new_vector->slice.copyFrom(slice);
}
if (buffer)
{
*buffer = new_vector->GetBuffer();
}
return new_vector;
}
void Vector::DoDotProduct(const PVector &x, void *result,
int result_type_id) const
{
// Can be called when Size() == 0, e.g. when an MPI-parallel vector has a
// local size of 0.
MFEM_ASSERT(result_type_id == ScalarId<double>::value, "");
double *res = (double *)result;
MFEM_ASSERT(dynamic_cast<const Vector *>(&x) != NULL, "invalid Vector type");
const Vector *xp = static_cast<const Vector *>(&x);
MFEM_ASSERT(this->Size() == xp->Size(), "");
*res = ::occa::linalg::dot<double, double, double>(this->slice, xp->slice);
#ifdef MFEM_USE_MPI
double local_dot = *res;
if (IsParallel())
{
MPI_Allreduce(&local_dot, res, 1, MPI_DOUBLE, MPI_SUM,
OccaLayout().OccaEngine().GetComm());
}
#endif
}
void Vector::DoAxpby(const void *a, const PVector &x,
const void *b, const PVector &y,
int ab_type_id)
{
const std::string &okl_defines = OccaLayout().OccaEngine().GetOklDefines();
//
// TODO: move all kernel builders to class mfem::occa::Backend
//
static ::occa::kernelBuilder axpby1_builder =
::occa::linalg::customLinearMethod(
"mfem_occa_axpby1",
"v0[i] = c0 * v1[i];",
"defines: {"
" CTYPE0: 'double',"
" VTYPE0: 'double',"
" VTYPE1: 'double',"
" TILESIZE: '128',"
"}");
static ::occa::kernelBuilder axpby2_builder =
::occa::linalg::customLinearMethod(
"mfem_occa_axpby2",
"v0[i] = c0 * v0[i] + c1 * v1[i];",
"defines: {"
" CTYPE0: 'double',"
" CTYPE1: 'double',"
" VTYPE0: 'double',"
" VTYPE1: 'double',"
" TILESIZE: '128',"
"}");
static ::occa::kernelBuilder axpby3_builder =
::occa::linalg::customLinearMethod(
"mfem_occa_axpby3",
"v0[i] = c0 * v1[i] + c1 * v2[i];",
"defines: {"
" CTYPE0: 'double',"
" CTYPE1: 'double',"
" VTYPE0: 'double',"
" VTYPE1: 'double',"
" VTYPE2: 'double',"
" TILESIZE: '128',"
"}");
// called only when Size() != 0
MFEM_ASSERT(ab_type_id == ScalarId<double>::value, "");
const double da = *static_cast<const double *>(a);
const double db = *static_cast<const double *>(b);
MFEM_ASSERT(da == 0.0 || dynamic_cast<const Vector *>(&x) != NULL,
"invalid Vector x");
MFEM_ASSERT(db == 0.0 || dynamic_cast<const Vector *>(&y) != NULL,
"invalid Vector y");
const Vector *xp = static_cast<const Vector *>(&x);
const Vector *yp = static_cast<const Vector *>(&y);
MFEM_ASSERT(da == 0.0 || this->Size() == xp->Size(), "");
MFEM_ASSERT(db == 0.0 || this->Size() == yp->Size(), "");
if (da == 0.0)
{
if (db == 0.0)
{
OccaFill(&da);
}
else
{
if (this->slice == yp->slice)
{
// *this *= db
::occa::linalg::operator_mult_eq(slice, db);
}
else
{
// *this = db * y
::occa::kernel kernel = axpby1_builder.build(slice.getDevice(),
okl_defines);
kernel((int)Size(), db, slice, yp->slice);
}
}
}
else
{
if (db == 0.0)
{
if (this->slice == xp->slice)
{
// *this *= da
::occa::linalg::operator_mult_eq(slice, da);
}
else
{
// *this = da * x
::occa::kernel kernel = axpby1_builder.build(slice.getDevice(),
okl_defines);
kernel((int)Size(), da, slice, xp->slice);
}
}
else
{
MFEM_ASSERT(xp->slice != yp->slice, "invalid input");
if (this->slice == xp->slice)
{
// *this = da * (*this) + db * y
::occa::kernel kernel = axpby2_builder.build(slice.getDevice(),
okl_defines);
kernel((int)Size(), da, db, slice, yp->slice);
}
else if (this->slice == yp->slice)
{
// *this = da * x + db * (*this)
::occa::kernel kernel = axpby2_builder.build(slice.getDevice(),
okl_defines);
kernel((int)Size(), db, da, slice, xp->slice);
}
else
{
// *this = da * x + db * y
::occa::kernel kernel = axpby3_builder.build(slice.getDevice(),
okl_defines);
kernel((int)Size(), da, db, slice, xp->slice, yp->slice);
}
}
}
}
mfem::Vector Vector::Wrap()
{
return mfem::Vector(*this);
}
const mfem::Vector Vector::Wrap() const
{
return mfem::Vector(*const_cast<Vector*>(this));
}
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
-74
View File
@@ -1,74 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OCCA_VECTOR_HPP
#define MFEM_BACKENDS_OCCA_VECTOR_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#include <occa.hpp>
#include "../base/vector.hpp"
#include "array.hpp"
namespace mfem
{
namespace occa
{
class Vector : virtual public Array, public PVector
{
protected:
//
// Inherited fields
//
// DLayout layout;
/**
@name Virtual interface
*/
///@{
virtual PVector *DoVectorClone(bool copy_data, void **buffer,
int buffer_type_id) const;
virtual void DoDotProduct(const PVector &x, void *result,
int result_type_id) const;
virtual void DoAxpby(const void *a, const PVector &x,
const void *b, const PVector &y,
int ab_type_id);
///@}
// End: Virtual interface
public:
Vector(Layout &lt)
: PArray(lt), Array(lt, sizeof(double)), PVector(lt)
{ }
mfem::Vector Wrap();
const mfem::Vector Wrap() const;
#if defined(MFEM_USE_MPI)
bool IsParallel() const { return (OccaLayout().OccaEngine().GetComm() != MPI_COMM_NULL); }
#endif
};
} // namespace mfem::occa
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
#endif // MFEM_BACKENDS_OCCA_VECTOR_HPP
-675
View File
@@ -1,675 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && \
defined(MFEM_USE_OMP) && \
defined(MFEM_USE_ACROTENSOR)
#include "adiffusioninteg.hpp"
namespace mfem
{
namespace omp
{
PAIntegrator::PAIntegrator(Coefficient &q, FiniteElementSpace &f)
{
Q = &q;
ofes = &f;
fes = ofes->GetFESpace();
onGPU = (ofes->OmpEngine().ExecTarget() == Device);
fe = fes->GetFE(0);
tfe = dynamic_cast<const TensorBasisElement*>(fe);
if (tfe)
{
tDofMap = tfe->GetDofMap();
}
else
{
tDofMap.SetSize(nDof);
for (int i = 0; i < nDof; ++i)
{
tDofMap[i] = i;
}
}
nElem = fes->GetNE();
GeomType = fe->GetGeomType();
FEOrder = fe->GetOrder();
nDim = fe->GetDim();
nDof = fe->GetDof();
ElementTransformation *Trans = fes->GetElementTransformation(0);
int irorder = 2*fe->GetOrder() + Trans->OrderW();
ir = &IntRules.Get(GeomType, irorder);
nQuad = ir->GetNPoints();
hasTensorBasis = tfe ? true : false;
if (nDim > 3)
{
mfem_error("AcroIntegrator tensor computations don't support dim > 3.");
}
}
PAIntegrator::~PAIntegrator()
{
}
AcroDiffusionIntegrator::AcroDiffusionIntegrator(Coefficient &q, FiniteElementSpace &f) :
PAIntegrator(q,f)
{
if (onGPU)
{
//TE.SetExecutorType("OneOutPerThread");
TE.SetExecutorType("Cuda");
//TODO: Set to an existing cuda context if one exists
}
else
{
TE.SetExecutorType("CPUInterpreted");
}
const IntegrationRule *ir1D = &IntRules.Get(Geometry::SEGMENT, ir->GetOrder());
nDof1D = FEOrder + 1;
nQuad1D = ir1D->GetNPoints();
if (hasTensorBasis)
{
H1_FECollection fec(FEOrder,1);
const FiniteElement *fe1D = fec.FiniteElementForGeometry(Geometry::SEGMENT);
mfem::Vector eval(nDof1D);
DenseMatrix deval(nDof1D,1);
B.Init(nQuad1D, nDof1D);
G.Init(nQuad1D, nDof1D);
std::vector<int> wdims(nDim, nQuad1D);
W.Init(wdims);
mfem::Vector w(nQuad1D);
for (int k = 0; k < nQuad1D; ++k)
{
const IntegrationPoint &ip = ir1D->IntPoint(k);
fe1D->CalcShape(ip, eval);
fe1D->CalcDShape(ip, deval);
B(k,0) = eval(0);
B(k,nDof1D-1) = eval(1);
G(k,0) = deval(0,0);
G(k,nDof1D-1) = deval(1,0);
for (int i = 1; i < nDof1D-1; ++i)
{
B(k,i) = eval(i+1);
G(k,i) = deval(i+1,0);
}
w(k) = ip.weight;
}
if (nDim == 1)
{
for (int k1 = 0; k1 < nQuad1D; ++k1)
{
W(k1) = w(k1);
}
}
else if (nDim == 2)
{
for (int k1 = 0; k1 < nQuad1D; ++k1)
{
for (int k2 = 0; k2 < nQuad1D; ++k2)
{
W(k1,k2) = w(k1)*w(k2);
}
}
}
else if (nDim == 3)
{
for (int k1 = 0; k1 < nQuad1D; ++k1)
{
for (int k2 = 0; k2 < nQuad1D; ++k2)
{
for (int k3 = 0; k3 < nQuad1D; ++k3)
{
W(k1,k2,k3) = w(k1)*w(k2)*w(k3);
}
}
}
}
}
else
{
mfem::Vector eval(nDof);
DenseMatrix deval(nDof,nDim);
G.Init(nQuad, nDof,nDim);
W.Init(nQuad);
for (int k = 0; k < nQuad; ++k)
{
const IntegrationPoint &ip = ir->IntPoint(k);
fe->CalcDShape(ip, deval);
for (int i = 0; i < nDof; ++i)
{
for (int d = 0; d < nDim; ++d)
{
G(k,i,d) = deval(i,d);
}
}
W(k) = ip.weight;
}
}
if (onGPU)
{
B.MapToGPU();
G.MapToGPU();
W.MapToGPU();
}
// Assemble in the constructor!
BatchedPartialAssemble();
}
AcroDiffusionIntegrator::~AcroDiffusionIntegrator()
{
for (int i = 0; i < Btil.Size(); i++) delete Btil[i];
}
void AcroDiffusionIntegrator::ComputeBTilde()
{
Btil.SetSize(nDim);
for (int d = 0; d < nDim; ++d)
{
Btil[d] = new acro::Tensor(nDim, nDim, nQuad1D, nDof1D, nDof1D);
for (int m = 0; m < nDim; ++m)
{
for (int n = 0; n < nDim; ++n)
{
acro::Tensor &BGM = (m == d) ? G : B;
acro::Tensor &BGN = (n == d) ? G : B;
for (int k = 0; k < nQuad1D; ++k)
{
for (int i = 0; i < nDof1D; ++i)
{
for (int j = 0; j < nDof1D; ++j)
{
(*Btil[d])(m, n, k, i, j) = BGM(k,i)*BGN(k,j);
}
}
}
}
}
}
}
void AcroDiffusionIntegrator::BatchedPartialAssemble()
{
//Initilze the tensors
acro::Tensor J,Jinv,Jdet,C;
if (hasTensorBasis)
{
const IntegrationRule *ir1D = &IntRules.Get(Geometry::SEGMENT, ir->GetOrder());
IntegrationPoint ip;
if (nDim == 1)
{
D.Init(nElem, nDim, nDim, nQuad1D);
J.Init(nElem, nQuad1D, nDim, nDim);
Jinv.Init(nElem, nQuad1D, nDim, nDim);
Jdet.Init(nElem, nQuad1D);
C.Init(nElem, nQuad1D);
for (int e = 0; e < nElem; ++e)
{
ElementTransformation *Trans = fes->GetElementTransformation(e);
for (int k1 = 0; k1 < nQuad1D; ++k1)
{
ip.x = ir1D->IntPoint(k1).x;
ip.y = 0.0;
ip.z = 0.0;
Trans->SetIntPoint(&ip);
C(e,k1) = Q->Eval(*Trans, ip);
const DenseMatrix &JMat = Trans->Jacobian();
for (int m = 0; m < nDim; ++m)
{
for (int n = 0; n < nDim; ++n)
{
J(e,k1,m,n) = JMat.Elem(m,n);
}
}
}
}
}
else if (nDim == 2)
{
D.Init(nElem, nDim, nDim, nQuad1D, nQuad1D);
J.Init(nElem, nQuad1D, nQuad1D, nDim, nDim);
Jinv.Init(nElem, nQuad1D, nQuad1D, nDim, nDim);
Jdet.Init(nElem, nQuad1D, nQuad1D);
C.Init(nElem, nQuad1D, nQuad1D);
for (int e = 0; e < nElem; ++e)
{
ElementTransformation *Trans = fes->GetElementTransformation(e);
for (int k1 = 0; k1 < nQuad1D; ++k1)
{
for (int k2 = 0; k2 < nQuad1D; ++k2)
{
ip.x = ir1D->IntPoint(k1).x;
ip.y = ir1D->IntPoint(k2).y;
ip.z = 0.0;
Trans->SetIntPoint(&ip);
C(e,k1,k2) = Q->Eval(*Trans, ip);
const DenseMatrix &JMat = Trans->Jacobian();
for (int m = 0; m < nDim; ++m)
{
for (int n = 0; n < nDim; ++n)
{
J(e,k1,k2,m,n) = JMat.Elem(m,n);
}
}
}
}
}
}
else if (nDim == 3)
{
D.Init(nElem, nDim, nDim, nQuad1D, nQuad1D, nQuad1D);
J.Init(nElem, nQuad1D, nQuad1D, nQuad1D, nDim, nDim);
Jinv.Init(nElem, nQuad1D, nQuad1D, nQuad1D, nDim, nDim);
Jdet.Init(nElem, nQuad1D, nQuad1D, nQuad1D);
C.Init(nElem, nQuad1D, nQuad1D, nQuad1D);
for (int e = 0; e < nElem; ++e)
{
ElementTransformation *Trans = fes->GetElementTransformation(e);
for (int k1 = 0; k1 < nQuad1D; ++k1)
{
for (int k2 = 0; k2 < nQuad1D; ++k2)
{
for (int k3 = 0; k3 < nQuad1D; ++k3)
{
ip.x = ir1D->IntPoint(k1).x;
ip.y = ir1D->IntPoint(k2).y;
ip.z = ir1D->IntPoint(k3).z;
Trans->SetIntPoint(&ip);
C(e,k1,k2,k3) = Q->Eval(*Trans, ip);
const DenseMatrix &JMat = Trans->Jacobian();
for (int m = 0; m < nDim; ++m)
{
for (int n = 0; n < nDim; ++n)
{
J(e,k1,k2,k3,m,n) = JMat.Elem(m,n);
}
}
}
}
}
}
}
}
else
{
D.Init(nElem, nDim, nDim, nQuad);
J.Init(nElem, nQuad, nDim, nDim);
Jinv.Init(nElem, nQuad, nDim, nDim);
Jdet.Init(nElem, nQuad);
C.Init(nElem, nQuad);
for (int e = 0; e < nElem; ++e)
{
ElementTransformation *Trans = fes->GetElementTransformation(e);
for (int k = 0; k < nQuad; ++k)
{
const IntegrationPoint &ip = ir->IntPoint(k);
Trans->SetIntPoint(&ip);
C(e,k) = Q->Eval(*Trans, ip);
const DenseMatrix &JMat = Trans->Jacobian();
for (int m = 0; m < nDim; ++m)
{
for (int n = 0; n < nDim; ++n)
{
J(e,k,m,n) = JMat.Elem(m,n);
}
}
}
}
}
TE.BatchMatrixInvDet(Jinv, Jdet, J);
if (hasTensorBasis)
{
if (nDim == 1)
{
TE("D_e_m_n_k = W_k C_e_k Jdet_e_k Jinv_e_k_m_j Jinv_e_k_n_j",
D, W, C, Jdet, Jinv, Jinv);
}
else if (nDim == 2)
{
TE("D_e_m_n_k1_k2 = W_k1_k2 C_e_k1_k2 Jdet_e_k1_k2 Jinv_e_k1_k2_m_j Jinv_e_k1_k2_n_j",
D, W, C, Jdet, Jinv, Jinv);
}
else if (nDim == 3)
{
TE("D_e_m_n_k1_k2_k3 = W_k1_k2_k3 C_e_k1_k2_k3 Jdet_e_k1_k2_k3 Jinv_e_k1_k2_k3_n_j Jinv_e_k1_k2_k3_m_j",
D, W, C, Jdet, Jinv, Jinv);
}
}
else
{
TE("D_e_m_n_k = W_k C_e_k Jdet_e_k Jinv_e_k_m_j Jinv_e_k_n_j",
D, W, C, Jdet, Jinv, Jinv);
}
}
void AcroDiffusionIntegrator::BatchedAssembleElementMatrices(DenseTensor &elmats)
{
if (hasTensorBasis && Btil.Size() == 0)
{
ComputeBTilde();
}
if (!D.IsInitialized())
{
BatchedPartialAssemble();
}
if (!S.IsInitialized())
{
if (hasTensorBasis)
{
if (nDim == 1)
{
S.Init(nElem, nDof1D, nDof1D);
}
else if (nDim == 2)
{
S.Init(nElem, nDof1D, nDof1D, nDof1D, nDof1D);
}
else if (nDim == 3)
{
S.Init(nElem, nDof1D, nDof1D, nDof1D, nDof1D, nDof1D, nDof1D);
}
}
else
{
S.Init(nElem, nDof, nDof);
}
if (onGPU) {S.SwitchToGPU();}
}
if (hasTensorBasis) {
if (nDim == 1) {
TE("S_e_i1_j1 = Btil_m_n_k1_i1_j1 D_e_m_n_k1",
S, *Btil[0], D);
}
else if (nDim == 2)
{
TE("S_e_i1_i2_j1_j2 = Btil1_m_n_k1_i1_j1 Btil2_m_n_k2_i2_j2 D_e_m_n_k1_k2",
S, *Btil[0], *Btil[1], D);
}
else if (nDim == 3)
{
TE("S_e_i1_i2_i3_j1_j2_j3 = Btil1_m_n_k1_i1_j1 Btil2_m_n_k2_i2_j2 Btil3_m_n_k3_i3_j3 D_e_m_n_k1_k2_k3",
S, *Btil[0], *Btil[1], *Btil[2], D);
}
}
else
{
TE("S_e_i_j = G_k_i_m G_k_i_n D_e_m_n_k",
S, G, G, D);
}
S.MoveFromGPU();
for (int e = 0; e < nElem; ++e)
{
for (int ei = 0; ei < nDof; ++ei)
{
for (int ej = 0; ej < nDof; ++ej)
{
elmats(tDofMap[ei], tDofMap[ej], e) = S[e*nDof*nDof + ei*nDof + ej];
}
}
}
}
void AcroDiffusionIntegrator::ComputeElementMatrices(Vector &elmats)
{
if (hasTensorBasis && Btil.Size() == 0)
{
ComputeBTilde();
}
if (!D.IsInitialized())
{
BatchedPartialAssemble();
}
if (!S.IsInitialized())
{
if (hasTensorBasis)
{
if (nDim == 1)
{
S.Init(nElem, nDof1D, nDof1D);
}
else if (nDim == 2)
{
S.Init(nElem, nDof1D, nDof1D, nDof1D, nDof1D);
}
else if (nDim == 3)
{
S.Init(nElem, nDof1D, nDof1D, nDof1D, nDof1D, nDof1D, nDof1D);
}
}
else
{
S.Init(nElem, nDof, nDof);
}
if (onGPU) {S.SwitchToGPU();}
}
if (hasTensorBasis) {
if (nDim == 1) {
TE("S_e_i1_j1 += Btil_m_n_k1_i1_j1 D_e_m_n_k1",
S, *Btil[0], D);
}
else if (nDim == 2)
{
TE("S_e_i1_i2_j1_j2 += Btil1_m_n_k1_i1_j1 Btil2_m_n_k2_i2_j2 D_e_m_n_k1_k2",
S, *Btil[0], *Btil[1], D);
}
else if (nDim == 3)
{
TE("S_e_i1_i2_i3_j1_j2_j3 += Btil1_m_n_k1_i1_j1 Btil2_m_n_k2_i2_j2 Btil3_m_n_k3_i3_j3 D_e_m_n_k1_k2_k3",
S, *Btil[0], *Btil[1], *Btil[2], D);
}
}
else
{
TE("S_e_i_j += G_k_i_m G_k_i_n D_e_m_n_k",
S, G, G, D);
}
S.MoveFromGPU();
double *edata = elmats.GetData<double>();
for (int e = 0; e < nElem; ++e)
{
const int e_offset = e * nDof * nDof;
for (int ei = 0; ei < nDof; ++ei)
{
const int offset = e_offset + ei * tDofMap[ei] * nDof;
for (int ej = 0; ej < nDof; ++ej)
{
const int index = offset + tDofMap[ej];
edata[index] = S[e*nDof*nDof + ei*nDof + ej];
}
}
}
}
void AcroDiffusionIntegrator::ReassembleOperator()
{
BatchedPartialAssemble();
}
void AcroDiffusionIntegrator::PAMult(const Vector &x, Vector &y)
{
MFEM_ASSERT(hasTensorBasis,"AcroDiffusionIntegrator PAMult on simplices not supported");
if (!U.IsInitialized())
{
// NOTE: x and y are already sized for the fespace in the constructor
double *Xptr = const_cast<double*>(x.GetData<double>());
double *Yptr = y.GetData<double>();
if (nDim == 1) {
X.Init(nElem,nDof1D,Xptr,Xptr,onGPU);
Y.Init(nElem,nDof1D,Yptr,Yptr,onGPU);
U.Init(nDim, nElem, nQuad1D);
Z.Init(nDim, nElem, nQuad1D);
if (onGPU)
{
U.SwitchToGPU();
Z.SwitchToGPU();
}
}
else if (nDim == 2)
{
X.Init(nElem,nDof1D,nDof1D,Xptr,Xptr,onGPU);
Y.Init(nElem,nDof1D,nDof1D,Yptr,Yptr,onGPU);
U.Init(nDim, nElem, nQuad1D, nQuad1D);
Z.Init(nDim, nElem, nQuad1D, nQuad1D);
T1.Init(nElem,nDof1D,nQuad1D);
if (onGPU)
{
U.SwitchToGPU();
Z.SwitchToGPU();
T1.SwitchToGPU();
}
}
else if (nDim == 3)
{
X.Init(nElem,nDof1D,nDof1D,nDof1D,Xptr,Xptr,onGPU);
Y.Init(nElem,nDof1D,nDof1D,nDof1D,Yptr,Yptr,onGPU);
U.Init(nDim, nElem, nQuad1D, nQuad1D, nQuad1D);
Z.Init(nDim, nElem, nQuad1D, nQuad1D, nQuad1D);
T1.Init(nElem, nDof1D, nQuad1D, nQuad1D);
T2.Init(nElem, nDof1D, nDof1D, nQuad1D);
if (onGPU)
{
U.SwitchToGPU();
Z.SwitchToGPU();
T1.SwitchToGPU();
T2.SwitchToGPU();
}
}
}
else
{
// NOTE: x and y are already sized for the fespace in the constructor
double *Xptr = const_cast<double*>(x.GetData<double>());
double *Yptr = y.GetData<double>();
X.Retarget(Xptr,Xptr);
Y.Retarget(Yptr,Yptr);
}
acro::SliceTensor U1,U2,U3,Z1,Z2,Z3;
if (nDim == 1)
{
TE("U_n_e_k1 = G_k1_i1 X_e_i1", U, G, X);
TE("Z_m_e_k1 = D_e_m_n_k1 U_n_e_k1", Z, D, U);
TE("Y_e_i1 = G_k1_i1 Z_m_e_k1", Y, G, Z);
}
else if (nDim == 2)
{
U1.SliceInit(U, 0); U2.SliceInit(U, 1);
Z1.SliceInit(Z, 0); Z2.SliceInit(Z, 1);
//U1_e_k1_k2 = G_k1_i1 B_k2_i2 X_e_i1_i2
TE("BX_e_i1_k2 = B_k2_i2 X_e_i2_i1", T1, B, X);
TE("U1_e_k1_k2 = G_k1_i1 BX_e_i1_k2", U1, G, T1);
//U2_e_k1_k2 = B_k1_i1 G_k2_i2 X_e_i1_i2
TE("GX_e_i1_k2 = G_k2_i2 X_e_i2_i1", T1, G, X);
TE("U2_e_k1_k2 = B_k1_i1 GX_e_i1_k2", U2, B, T1);
TE("Z_m_e_k1_k2 = D_e_m_n_k1_k2 U_n_e_k1_k2", Z, D, U);
//Y_e_i1_i2 = G_k1_i1 B_k2_i2 Z1_e_k1_k2
TE("BZ1_e_i2_k1 = B_k2_i2 Z1_e_k1_k2", T1, B, Z1);
TE("Y_e_i2_i1 = G_k1_i1 BZ1_e_i2_k1", Y, G, T1);
//Y_e_i1_i2 += B_k1_i1 G_k2_i2 Z2_e_k1_k2
TE("GZ2_e_i2_k1 = G_k2_i2 Z2_e_k1_k2", T1, G, Z2);
TE("Y_e_i2_i1 += B_k1_i1 GZ2_e_i2_k1", Y, B, T1);
}
else if (nDim == 3)
{
U1.SliceInit(U, 0); U2.SliceInit(U, 1); U3.SliceInit(U, 2);
Z1.SliceInit(Z, 0); Z2.SliceInit(Z, 1); Z3.SliceInit(Z, 2);
TE.BeginMultiKernelLaunch();
//U1_e_k1_k2_k3 = G_k1_i1 B_k2_i2 B_k3_i3 X_e_i1_i2_i3
TE("T2_e_i1_i2_k3 = B_k3_i3 X_e_i1_i2_i3", T2, B, X);
TE("T1_e_i1_k2_k3 = B_k2_i2 T2_e_i1_i2_k3", T1, B, T2);
TE("U1_e_k1_k2_k3 = G_k1_i1 T1_e_i1_k2_k3", U1, G, T1);
//U2_e_k1_k2_k3 = B_k1_i1 G_k2_i2 B_k3_i3 X_e_i1_i2_i3
TE("T1_e_i1_k2_k3 = G_k2_i2 T2_e_i1_i2_k3", T1, G, T2);
TE("U2_e_k1_k2_k3 = B_k1_i1 T1_e_i1_k2_k3", U2, B, T1);
//U3_e_k1_k2_k3 = B_k1_i1 B_k2_i2 G_k3_i3 X_e_i1_i2_i3
TE("T2_e_i1_i2_k3 = G_k3_i3 X_e_i1_i2_i3", T2, G, X);
TE("T1_e_i1_k2_k3 = B_k2_i2 T2_e_i1_i2_k3", T1, B, T2);
TE("U3_e_k1_k2_k3 = B_k1_i1 T1_e_i1_k2_k3", U3, B, T1);
TE("Z_m_e_k1_k2_k3 = D_e_m_n_k1_k2_k3 U_n_e_k1_k2_k3", Z, D, U);
//Y_e_i1_i2_i3 = G_k1_i1 B_k2_i2 B_k3_i3 Z1_e_k1_k2_k3
TE("T1_e_i3_k1_k2 = B_k3_i3 Z1_e_k1_k2_k3", T1, B, Z1);
TE("T2_e_i2_i3_k1 = B_k2_i2 T1_e_i3_k1_k2", T2, B, T1);
TE("Y_e_i1_i2_i3 = G_k1_i1 T2_e_i2_i3_k1", Y, G, T2);
//Y_e_i1_i2_i3 += B_k1_i1 G_k2_i2 B_k3_i3 Z2_e_k1_k2_k3
TE("T1_e_i3_k1_k2 = B_k3_i3 Z2_e_k1_k2_k3", T1, B, Z2);
TE("T2_e_i2_i3_k1 = G_k2_i2 T1_e_i3_k1_k2", T2, G, T1);
TE("Y_e_i1_i2_i3 += B_k1_i1 T2_e_i2_i3_k1", Y, B, T2);
//Y_e_i1_i2_i3 += B_k1_i1 B_k2_i2 G_k3_i3 Z3_e_k1_k2_k3
TE("T1_e_i3_k1_k2 = G_k3_i3 Z3_e_k1_k2_k3", T1, G, Z3);
TE("T2_e_i2_i3_k1 = B_k2_i2 T1_e_i3_k1_k2", T2, B, T1);
TE("Y_e_i1_i2_i3 += B_k1_i1 T2_e_i2_i3_k1", Y, B, T2);
TE.EndMultiKernelLaunch();
}
}
void AcroDiffusionIntegrator::MultAdd(const Vector &x, Vector &y) const
{
const_cast<AcroDiffusionIntegrator*>(this)->PAMult(x, y);
}
void AcroDiffusionIntegrator::MultTransposeAdd(const Vector &x, Vector &y) const
{
mfem_error("Not supported");
}
} // namespace mfem::omp
} // namespace mfem
#endif
-95
View File
@@ -1,95 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OMP_ADIFFUSIONINTEG_HPP
#define MFEM_BACKENDS_OMP_ADIFFUSIONINTEG_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && \
defined(MFEM_USE_OMP) && \
defined(MFEM_USE_ACROTENSOR)
#include "../../fem/bilininteg.hpp"
#include "../../fem/fem.hpp"
#include "vector.hpp"
#include "fespace.hpp"
#include "bilinearform.hpp"
#include "AcroTensor.hpp"
namespace mfem
{
namespace omp
{
class PAIntegrator : public TensorBilinearFormIntegrator
{
protected:
Coefficient *Q;
FiniteElementSpace *ofes;
mfem::FiniteElementSpace *fes;
const FiniteElement *fe;
const TensorBasisElement *tfe;
const IntegrationRule *ir;
mfem::Array<int> tDofMap;
int GeomType;
int FEOrder;
bool onGPU;
bool hasTensorBasis;
int nDim;
int nElem;
int nDof;
int nQuad;
public:
PAIntegrator(Coefficient &q, FiniteElementSpace &f);
virtual ~PAIntegrator();
};
class AcroDiffusionIntegrator : public PAIntegrator
{
private:
acro::TensorEngine TE;
int nDof1D;
int nQuad1D;
acro::Tensor B, G; //Basis and dbasis evaluated on the quad points
acro::Tensor W; //Integration weights
mfem::Array<acro::Tensor*> Btil; //Btilde used to compute stiffness matrix
acro::Tensor D; //Product of integration weight, physical consts, and element shape info
acro::Tensor S; //The assembled local stiffness matrices
acro::Tensor U, Z, T1, T2; //Intermediate computations for tensor product partial assembly
acro::Tensor X, Y;
void ComputeBTilde();
public:
AcroDiffusionIntegrator(BilinearFormIntegrator *integ);
AcroDiffusionIntegrator(Coefficient &q, FiniteElementSpace &f);
virtual ~AcroDiffusionIntegrator();
void BatchedPartialAssemble();
void BatchedAssembleElementMatrices(DenseTensor &elmats);
void ComputeElementMatrices(Vector &elmats);
void PAMult(const Vector &x, Vector &y);
virtual void MultTransposeAdd(const Vector &x, Vector &y) const;
virtual void MultAdd(const Vector &x, Vector &y) const;
virtual void ReassembleOperator();
};
} // namespace mfem::omp
} // namespace mfem
#endif
#endif
-128
View File
@@ -1,128 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include <cstring>
#include "array.hpp"
namespace mfem
{
namespace omp
{
PArray *Array::DoClone(bool copy_data, void **buffer,
std::size_t item_size) const
{
Array *new_array = new Array(OmpLayout(), item_size);
if (copy_data)
{
if (!ComputeOnDevice())
std::memcpy(new_array->GetData<void>(), data, bytes);
else
{
char *new_data = new_array->GetData<char>();
const bool use_target = ComputeOnDevice();
const bool use_parallel = Size() > 1000;
#pragma omp target teams distribute parallel for \
if (target: use_target) if (parallel: use_parallel) \
is_device_ptr(new_data)
for (std::size_t i = 0; i < bytes; i++) new_data[i] = data[i];
}
}
if (buffer)
{
*buffer = new_array->GetData<void>();
}
return new_array;
}
int Array::DoResize(PLayout &new_layout, void **buffer,
std::size_t item_size)
{
MFEM_ASSERT(dynamic_cast<Layout *>(&new_layout) != NULL,
"new_layout is not an OMP Layout");
Layout *lt = static_cast<Layout *>(&new_layout);
layout.Reset(lt); // Reset() checks if the pointer is the same
int err = ResizeData(lt, item_size);
if (!err && buffer)
{
*buffer = GetData<void>();
}
return err;
}
void *Array::DoPullData(void *buffer, std::size_t item_size)
{
// called only when Size() != 0
if (!IsUnifiedMemory() && ComputeOnDevice() && (buffer != NULL))
{
#pragma omp target update from(data)
std::memcpy(buffer, data, bytes);
}
else
{
buffer = data;
}
return buffer;
}
void Array::DoFill(const void *value_ptr, std::size_t item_size)
{
// called only when Size() != 0
switch (item_size)
{
case sizeof(int):
OmpFill((const int *)value_ptr);
break;
case sizeof(double):
OmpFill((const double *)value_ptr);
break;
default:
MFEM_ABORT("item_size = " << item_size << " is not supported");
}
}
void Array::DoPushData(const void *src_buffer, std::size_t item_size)
{
// called only when Size() != 0
std::memcpy(data, (char *) src_buffer, bytes);
if ((!IsUnifiedMemory() && ComputeOnDevice()) && (data != src_buffer))
{
#pragma omp target update to(data)
}
}
void Array::DoAssign(const PArray &src, std::size_t item_size)
{
// called only when Size() != 0
// Note: static_cast can not be used here since PArray is a virtual base
// class.
const Array *source = dynamic_cast<const Array *>(&src);
MFEM_ASSERT(source != NULL, "invalid source Array type");
MFEM_ASSERT(Size() == source->Size(), "");
// All arrays from this engine are of the same type, so we can simply check *this and assume the same is used in src.
DoPushData(source->GetData<void>(), item_size);
}
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
-143
View File
@@ -1,143 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OMP_ARRAY_HPP
#define MFEM_BACKENDS_OMP_ARRAY_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "layout.hpp"
#include "../base/array.hpp"
namespace mfem
{
namespace omp
{
class Array : public virtual mfem::PArray
{
protected:
//
// Inherited fields
//
// DLayout layout;
bool own_data;
std::size_t bytes;
char *data;
//
// Virtual interface
//
virtual void *DoGetData() const { return (void *) data; }
virtual PArray *DoClone(bool copy_data, void **buffer,
std::size_t item_size) const;
virtual int DoResize(PLayout &new_layout, void **buffer,
std::size_t item_size);
virtual void *DoPullData(void *buffer, std::size_t item_size);
virtual void DoFill(const void *value_ptr, std::size_t item_size);
virtual void DoPushData(const void *src_buffer, std::size_t item_size);
virtual void DoAssign(const PArray &src, std::size_t item_size);
//
// Auxiliary methods
//
inline int ResizeData(const Layout *lt, std::size_t item_size);
inline bool IsUnifiedMemory() const { return OmpLayout().OmpEngine().UnifiedMemory(); }
template <typename T>
void OmpFill(const T *pval)
{
T *ptr = (T*) data;
T val = *pval;
const bool use_target = ComputeOnDevice();
const bool use_parallel = (use_target || layout->Size() > 1000);
const std::size_t size = layout->Size();
#pragma omp target teams distribute parallel for \
if (target: use_target) \
if (parallel: use_parallel) map (to: ptr, val)
for (int i = 0; i < size; i++) ptr[i] = val;
}
public:
Array(Layout &lt, std::size_t item_size)
: PArray(lt),
own_data(true),
bytes(lt.Size() * item_size),
data(static_cast<char *>(lt.Alloc(bytes)))
{
#pragma omp target enter data map(alloc:data[:bytes]) if (!IsUnifiedMemory() && ComputeOnDevice())
}
Array(const Array &array)
: PArray(array.GetLayout()),
own_data(false),
bytes(array.bytes),
data(array.data) { }
inline bool ComputeOnDevice() const { return (OmpLayout().OmpEngine().ExecTarget() == Device); }
virtual ~Array()
{
#pragma omp target exit data map(delete:data[:bytes]) if (!IsUnifiedMemory() && ComputeOnDevice())
if (own_data) layout->As<Layout>().Dealloc(data);
}
inline void MakeRef(Array &master);
Layout &OmpLayout() const
{ return *static_cast<Layout *>(layout.Get()); }
};
//
// Inline methods
//
inline int Array::ResizeData(const Layout *lt, std::size_t item_size)
{
const std::size_t new_bytes = lt->Size() * item_size;
if (bytes < new_bytes)
{
#pragma omp target exit data map(delete:data)
OmpLayout().Dealloc(data);
data = static_cast<char *>(OmpLayout().Alloc(new_bytes));
MFEM_VERIFY(data != NULL, "");
// If memory allocation fails - an exception is thrown.
#pragma omp target enter data map(alloc:data[:new_bytes])
}
return 0;
}
inline void Array::MakeRef(Array &master)
{
layout = master.layout;
data = master.data;
}
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#endif // MFEM_BACKENDS_OMP_ARRAY_HPP
-46
View File
@@ -1,46 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "backend.hpp"
#include "engine.hpp"
namespace mfem
{
namespace omp
{
bool Backend::Supports(const std::string &engine_spec) const
{
return true;
}
mfem::Engine *Create(const std::string &engine_spec)
{
return new Engine(engine_spec);
}
#ifdef MFEM_USE_MPI
mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec)
{
return new Engine(comm, engine_spec);
}
#endif
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
-48
View File
@@ -1,48 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OMP_BACKEND_HPP
#define MFEM_BACKENDS_OMP_BACKEND_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
// Only the Backend and Engine classes should be exposed through "backend.hpp"
#include "../base/backend.hpp"
#include "engine.hpp"
namespace mfem
{
namespace omp
{
class Backend : public mfem::Backend
{
public:
virtual ~Backend();
virtual bool Supports(const std::string &engine_spec) const;
virtual mfem::Engine *Create(const std::string &engine_spec);
#ifdef MFEM_USE_MPI
virtual mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec);
#endif
};
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#endif // MFEM_BACKENDS_OMP_BACKEND_HPP
-399
View File
@@ -1,399 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "backend.hpp"
#include "bilinearform.hpp"
#include "adiffusioninteg.hpp"
namespace mfem
{
namespace omp
{
BilinearForm::~BilinearForm()
{
// Make sure all integrators free their data
for (int i = 0; i < tbfi.Size(); i++) delete tbfi[i];
delete element_matrices;
}
void BilinearForm::TransferIntegrators()
{
mfem::Array<mfem::BilinearFormIntegrator*> &dbfi = *bform->GetDBFI();
for (int i = 0; i < dbfi.Size(); i++)
{
std::string integ_name(dbfi[i]->Name());
Coefficient *scal_coeff = dbfi[i]->GetScalarCoefficient();
// ConstantCoefficient *const_coeff =
// dynamic_cast<ConstantCoefficient*>(scal_coeff);
// // TODO: other types of coefficients ...
// double val = const_coeff ? const_coeff->constant : 1.0;
if (integ_name == "(undefined)")
{
MFEM_ABORT("BilinearFormIntegrator does not define Name()");
}
else if (integ_name == "diffusion")
{
switch (OmpEngine().IntegType())
{
case Acrotensor:
tbfi.Append(new AcroDiffusionIntegrator(*scal_coeff, bform->FESpace()->Get_PFESpace()->As<FiniteElementSpace>()));
break;
default:
mfem_error("integrator is not supported for any MultType");
break;
}
}
else
{
MFEM_ABORT("BilinearFormIntegrator [Name() = " << integ_name
<< "] is not supported");
}
}
}
void BilinearForm::InitRHS(const mfem::Array<int> &ess_tdof_list,
mfem::Vector &mfem_x, mfem::Vector &mfem_b,
mfem::OperatorHandle &A,
mfem::Vector &mfem_X, mfem::Vector &mfem_B,
int copy_interior) const
{
const mfem::Operator *P = GetProlongation();
const mfem::Operator *R = GetRestriction();
if (P)
{
// Variational restriction with P
mfem_B.Resize(P->InLayout());
P->MultTranspose(mfem_b, mfem_B);
mfem_X.Resize(R->OutLayout());
R->Mult(mfem_x, mfem_X);
}
else
{
// rap, X and B point to the same data as this, x and b
mfem_X.MakeRef(mfem_x);
mfem_B.MakeRef(mfem_b);
}
if (A.Type() != mfem::Operator::ANY_TYPE)
{
A.EliminateBC(mat_e, ess_tdof_list, mfem_X, mfem_B);
}
if (!copy_interior && ess_tdof_list.Size() > 0)
{
Vector &X = mfem_X.Get_PVector()->As<Vector>();
const Array &constraint_list = ess_tdof_list.Get_PArray()->As<Array>();
double *X_data = X.GetData<double>();
const int* constraint_data = constraint_list.GetData<int>();
Vector subvec(constraint_list.OmpLayout());
double *subvec_data = subvec.GetData<double>();
const std::size_t num_constraint = constraint_list.Size();
const bool use_target = constraint_list.ComputeOnDevice();
const bool use_parallel = (use_target || num_constraint > 1000);
// This operation is a general version of mfem::Vector::SetSubVectorComplement()
// {
#pragma omp target teams distribute parallel for \
map(to: subvec_data, constraint_data, X_data) \
if (target: use_target) \
if (parallel: use_parallel)
for (std::size_t i = 0; i < num_constraint; i++) subvec_data[i] = X_data[constraint_data[i]];
X.Fill(0.0);
#pragma omp target teams distribute parallel for \
map(to: X_data, constraint_data, subvec_data) \
if (target: use_target) \
if (parallel: use_parallel)
for (std::size_t i = 0; i < num_constraint; i++) X_data[constraint_data[i]] = subvec_data[i];
// }
}
if (A.Type() == mfem::Operator::ANY_TYPE)
{
ConstrainedOperator *A_constrained = static_cast<ConstrainedOperator*>(A.Ptr());
A_constrained->EliminateRHS(mfem_X, mfem_B);
}
}
bool BilinearForm::Assemble()
{
if (!has_assembled)
{
TransferIntegrators();
has_assembled = true;
}
return true;
}
void BilinearForm::ComputeElementMatrices()
{
// Only called if performing full assembly
const int nelements = trial_fes->GetFESpace()->GetNE();
const int trial_ndofs = trial_fes->GetFESpace()->GetFE(0)->GetDof() * trial_fes->GetFESpace()->GetVDim();
const int test_ndofs = test_fes->GetFESpace()->GetFE(0)->GetDof() * test_fes->GetFESpace()->GetVDim();
const std::size_t length = nelements * trial_ndofs * test_ndofs;
if (!element_matrices) element_matrices = new mfem::Vector(*(new Layout(OmpEngine(), length)));
else element_matrices->Push();
element_matrices->Fill(0.0);
Vector &elmats = element_matrices->Get_PVector()->As<Vector>();
tbfi[0]->ComputeElementMatrices(elmats);
if (tbfi.Size() > 1)
{
for (int k = 1; k < tbfi.Size(); k++)
{
tbfi[k]->ComputeElementMatrices(elmats);
}
}
}
void BilinearForm::FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
mfem::OperatorHandle &A)
{
if (A.Type() == mfem::Operator::ANY_TYPE)
{
// FIXME: Support different test and trial spaces (MixedBilinearForm)
const mfem::Operator *P = GetProlongation();
mfem::Operator *rap = this;
if (P != NULL) rap = new mfem::RAPOperator(*P, *this, *P);
A.Reset(new ConstrainedOperator(rap, ess_tdof_list, (rap != this)));
return;
}
else
{
// ASSUMPTION: some sort of sparse matrix
// Compute the local matrices (stored in bform->element_matrices
ComputeElementMatrices();
bform->AllocateMatrix();
mfem::SparseMatrix &mat = bform->SpMat();
element_matrices->Pull();
double *data = element_matrices->GetData();
const bool skip_zeros = true;
mfem::Array<int> tr_vdofs, te_vdofs;
for (int i = 0; i < trial_fes->GetFESpace()->GetNE(); i++)
{
trial_fes->GetFESpace()->GetElementVDofs(i, tr_vdofs);
test_fes->GetFESpace()->GetElementVDofs(i, te_vdofs);
const mfem::DenseMatrix elmat(data, te_vdofs.Size(), tr_vdofs.Size());
mat.AddSubMatrix(te_vdofs, tr_vdofs, elmat, skip_zeros);
data += tr_vdofs.Size() * te_vdofs.Size();
}
}
if (A.Type() == mfem::Operator::MFEM_SPARSEMAT)
{
// This works because the FormSystemMatrix call with an explicit
// SparseMatrix doesnt call the backend version... This might
// change in the future.
bform->FormSystemMatrix(ess_tdof_list, static_cast<mfem::SparseMatrix&>(*A.Ptr()));
}
#ifdef MFEM_USE_MPI
else if (A.Type() == mfem::Operator::Hypre_ParCSR)
{
mfem::SparseMatrix &mat = bform->SpMat();
mfem::ParBilinearForm *pbform = dynamic_cast<mfem::ParBilinearForm*>(bform);
const bool skip_zeros = false;
mat.Finalize(skip_zeros);
// -------- FOR SOME VERY AGGREVATING REASON THIS DOESN'T WORK ---------
// mfem::ParFiniteElementSpace *pfes = pbform->ParFESpace();
// OperatorHandle dA(Operator::Hypre_ParCSR);
// // construct a parallel block-diagonal matrix 'A' based on 'a'
// dA.MakeSquareBlockDiag(pfes->GetComm(), *engine->MakeLayout(pfes->GlobalTrueVSize()),
// pfes->GetDofOffsets(), &mat);
// OperatorHandle Ph(pfes->Dof_TrueDof_Matrix());
// A.MakePtAP(dA, Ph);
// A.SetOperatorOwner(false);
// -------- BUT THIS DOES ---------
pbform->ParallelAssemble(A, &mat);
A.SetOperatorOwner(false);
// ---------------------
mat.Clear();
mat_e.Clear();
std::cout << "operator size (FormSystemMatrix): " << A.Ptr()->InLayout()->Size() << " " << A.Ptr()->OutLayout()->Size() << std::endl;
mat_e.EliminateRowsCols(A, ess_tdof_list);
}
#endif
else
{
MFEM_ABORT("Operator::Type is not supported, type = " << A.Type());
}
}
void BilinearForm::FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
mfem::Vector &x, mfem::Vector &b,
mfem::OperatorHandle &A, mfem::Vector &X, mfem::Vector &B,
int copy_interior)
{
FormSystemMatrix(ess_tdof_list, A);
std::cout << "operator size (FormLinearSystem 1): " << A.Ptr()->InLayout()->Size() << " " << A.Ptr()->OutLayout()->Size() << std::endl;
InitRHS(ess_tdof_list, x, b, A, X, B, copy_interior);
}
void BilinearForm::RecoverFEMSolution(const mfem::Vector &X, const mfem::Vector &b,
mfem::Vector &x)
{
const mfem::Operator *P = GetProlongation();
if (P)
{
// Apply conforming prolongation
x.Resize(P->OutLayout());
P->Mult(X, x);
}
// Otherwise X and x point to the same data
}
void BilinearForm::Mult(const mfem::Vector &x, mfem::Vector &y) const
{
trial_fes->ToEVector(x.Get_PVector()->As<Vector>(), x_local);
y_local.Fill<double>(0.0);
for (int i = 0; i < tbfi.Size(); i++) tbfi[i]->MultAdd(x_local, y_local);
test_fes->ToLVector(y_local, y.Get_PVector()->As<Vector>());
}
void BilinearForm::MultTranspose(const mfem::Vector &x, mfem::Vector &y) const
{ mfem_error("mfem::omp::BilinearForm::MultTranspose() is not supported!"); }
ConstrainedOperator::ConstrainedOperator(mfem::Operator *A_,
const mfem::Array<int> &constraint_list_,
bool own_A_)
: Operator(A_->InLayout()->As<Layout>()),
A(A_),
own_A(own_A_),
// FIXME: @dudouit1 has a general fix for this
constraint_list(constraint_list_.Get_PArray()->As<Array>()),
z(OutLayout()->As<Layout>()),
w(OutLayout()->As<Layout>()),
mfem_z((z.DontDelete(), z)),
mfem_w((w.DontDelete(), w)) { }
void ConstrainedOperator::EliminateRHS(const mfem::Vector &mfem_x, mfem::Vector &mfem_b) const
{
w.Fill<double>(0.0);
const Vector &x = mfem_x.Get_PVector()->As<Vector>();
Vector &b = mfem_b.Get_PVector()->As<Vector>();
const double *x_data = x.GetData<double>();
double *b_data = b.GetData<double>();
double *w_data = w.GetData<double>();
const int* constraint_data = constraint_list.GetData<int>();
const std::size_t num_constraint = constraint_list.Size();
const bool use_target = constraint_list.ComputeOnDevice();
const bool use_parallel = (use_target || num_constraint > 1000);
if (num_constraint > 0)
{
#pragma omp target teams distribute parallel for \
map(to: w_data, constraint_data, x_data) \
if (target: use_target) \
if (parallel: use_parallel)
for (std::size_t i = 0; i < num_constraint; i++)
w_data[constraint_data[i]] = x_data[constraint_data[i]];
}
A->Mult(mfem_w, mfem_z);
b.Axpby<double>(1.0, b, -1.0, z);
if (num_constraint > 0)
{
#pragma omp target teams distribute parallel for \
map(to: b_data, constraint_data, x_data) \
if (target: use_target) \
if (parallel: use_parallel)
for (std::size_t i = 0; i < num_constraint; i++)
b_data[constraint_data[i]] = x_data[constraint_data[i]];
}
}
void ConstrainedOperator::Mult(const mfem::Vector &mfem_x, mfem::Vector &mfem_y) const
{
if (constraint_list.Size() == 0)
{
A->Mult(mfem_x, mfem_y);
return;
}
const Vector &x = mfem_x.Get_PVector()->As<Vector>();
Vector &y = mfem_y.Get_PVector()->As<Vector>();
const double *x_data = x.GetData<double>();
double *y_data = y.GetData<double>();
double *z_data = z.GetData<double>();
const int* constraint_data = constraint_list.GetData<int>();
const std::size_t num_constraint = constraint_list.Size();
const bool use_target = constraint_list.ComputeOnDevice();
const bool use_parallel = (use_target || num_constraint > 1000);
z.Assign<double>(x); // z = x
// z[constraint_list] = 0.0
#pragma omp target teams distribute parallel for \
map(to: z_data, constraint_data) \
if (target: use_target) \
if (parallel: use_parallel)
for (std::size_t i = 0; i < num_constraint; i++)
z_data[constraint_data[i]] = 0.0;
// y = A * z
A->Mult(mfem_z, mfem_y);
// y[constraint_list] = x[constraint_list]
#pragma omp target teams distribute parallel for \
map(to: y_data, constraint_data, x_data) \
if (target: use_target) \
if (parallel: use_parallel)
for (std::size_t i = 0; i < num_constraint; i++)
y_data[constraint_data[i]] = x_data[constraint_data[i]];
}
// Destructor: destroys the unconstrained Operator @a A if @a own_A is true.
ConstrainedOperator::~ConstrainedOperator()
{
if (own_A) delete A;
}
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
-176
View File
@@ -1,176 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OMP_BILINEARFORM_HPP
#define MFEM_BACKENDS_OMP_BILINEARFORM_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "fespace.hpp"
#include "array.hpp"
#include "vector.hpp"
#include "../../fem/bilininteg.hpp"
namespace mfem
{
namespace omp
{
class TensorBilinearFormIntegrator
{
public:
virtual ~TensorBilinearFormIntegrator() { }
virtual void ReassembleOperator() = 0;
virtual void ComputeElementMatrices(Vector &element_matrices)
{ mfem_error("TensorBilinaerFormIntegrator::ComputeElementMatrices is not overloaded"); }
virtual void MultAdd(const Vector &x, Vector &y) const = 0;
virtual void Mult(const Vector &x, Vector &y) const
{ y.Fill<double>(0.0); MultAdd(x, y); }
};
/// TODO: doxygen
class BilinearForm : public mfem::PBilinearForm, public mfem::Operator
{
protected:
//
// Inherited fields
//
// SharedPtr<const mfem::Engine> engine;
// mfem::BilinearForm *bform;
mfem::Array<TensorBilinearFormIntegrator*> tbfi;
bool has_assembled;
mutable FiniteElementSpace *trial_fes, *test_fes;
mutable Vector x_local, y_local;
mfem::Vector *element_matrices;
OperatorHandle mat_e;
void TransferIntegrators();
void ComputeElementMatrices();
void InitRHS(const mfem::Array<int> &constraint_list,
mfem::Vector &mfem_x, mfem::Vector &mfem_b,
mfem::OperatorHandle &A,
mfem::Vector &mfem_X, mfem::Vector &mfem_B,
int copy_interior = 0) const;
public:
/// TODO: doxygen
BilinearForm(const Engine &e, mfem::BilinearForm &bf)
: mfem::PBilinearForm(e, bf),
// FIXME: for mixed bilinear forms
mfem::Operator(*bf.FESpace()->GetVLayout().As<Layout>()),
tbfi(),
has_assembled(false),
trial_fes(&bf.FESpace()->Get_PFESpace()->As<FiniteElementSpace>()),
test_fes(&bf.FESpace()->Get_PFESpace()->As<FiniteElementSpace>()),
x_local(trial_fes->GetELayout()),
y_local(test_fes->GetELayout()),
element_matrices(NULL),
mat_e() { }
/// Virtual destructor
virtual ~BilinearForm();
/// Return the engine as an OpenMP engine
const Engine &OmpEngine() { return static_cast<const Engine&>(*engine); }
/** @brief Prolongation operator from linear algebra (linear system) vectors,
to input vectors for the operator. `NULL` means identity. */
virtual const Operator *GetProlongation() const { return trial_fes->GetProlongation(); }
/** @brief Restriction operator from input vectors for the operator to linear
algebra (linear system) vectors. `NULL` means identity. */
virtual const Operator *GetRestriction() const { return test_fes->GetRestriction(); }
/// Assemble the PBilinearForm.
/** This method is called from the method BilinearForm::Assemble() of the
associated BilinearForm #bform.
@returns True, if the host assembly should be skipped. */
virtual bool Assemble();
/// TODO: doxygen
virtual void FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
mfem::OperatorHandle &A);
/// TODO: doxygen
virtual void FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
mfem::Vector &x, mfem::Vector &b,
mfem::OperatorHandle &A, mfem::Vector &mfem_X, mfem::Vector &mfem_B,
int copy_interior);
/// TODO: doxygen
virtual void RecoverFEMSolution(const mfem::Vector &mfem_X, const mfem::Vector &mfem_b,
mfem::Vector &mfem_x);
/// Operator application: `y=A(x)`.
virtual void Mult(const mfem::Vector &mfem_x, mfem::Vector &mfem_y) const;
/** @brief Action of the transpose operator: `y=A^t(x)`. The default behavior
in class Operator is to generate an error. */
virtual void MultTranspose(const mfem::Vector &mfem_x, mfem::Vector &mfem_y) const;
};
class ConstrainedOperator : public mfem::Operator
{
const mfem::Operator *A;
const bool own_A;
const Array constraint_list;
mutable Vector z, w;
mutable mfem::Vector mfem_z, mfem_w;
public:
ConstrainedOperator(mfem::Operator *A_,
const mfem::Array<int> &constraint_list_,
bool own_A_ = false);
// Destructor: destroys the unconstrained Operator @a A if @a own_A is true.
virtual ~ConstrainedOperator();
/** @brief Eliminate "essential boundary condition" values specified in @a x
from the given right-hand side @a b.
Performs the following steps:
z = A((0,x_b)); b_i -= z_i; b_b = x_b;
where the "_b" subscripts denote the essential (boundary) indices/dofs of
the vectors, and "_i" -- the rest of the entries. */
void EliminateRHS(const mfem::Vector &mfem_x, mfem::Vector &mfem_b) const;
/** @brief Constrained operator action.
Performs the following steps:
z = A((x_i,0)); y_i = z_i; y_b = x_b;
where the "_b" subscripts denote the essential (boundary) indices/dofs of
the vectors, and "_i" -- the rest of the entries. */
virtual void Mult(const mfem::Vector &mfem_x, mfem::Vector &mfem_y) const;
};
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#endif // MFEM_BACKENDS_OMP_BILINEAR_FORM_HPP
-253
View File
@@ -1,253 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "engine.hpp"
#include "array.hpp"
#include "layout.hpp"
#include "vector.hpp"
#include "fespace.hpp"
#include "bilinearform.hpp"
#include "memory_resource.hpp"
#include <map>
namespace mfem
{
namespace omp
{
typedef std::map<std::string, std::string> keyval_pair_t;
template<typename T, typename P>
static T remove_if(T beg, T end, P pred)
{
T dest = beg;
for (T itr = beg;itr != end; ++itr)
if (!pred(*itr))
*(dest++) = *itr;
return dest;
}
void parse_token(const std::string &token, std::string &key, std::string &val)
{
std::size_t sep = token.find_first_of(':');
if (sep > token.size()) mfem_error("Parse error");
key = token.substr(0, sep);
key.erase(mfem::omp::remove_if(key.begin(), key.end(), isspace), key.end());
key.erase(std::remove(key.begin(), key.end(), '\''), key.end());
val = token.substr(sep+1);
val.erase(mfem::omp::remove_if(val.begin(), val.end(), isspace), val.end());
val.erase(std::remove(val.begin(), val.end(), '\''), val.end());
}
keyval_pair_t parse_engine_spec(const std::string &engine_spec)
{
keyval_pair_t map;
std::size_t token_extent = 0;
std::string key, val;
while (token_extent < engine_spec.size())
{
const std::string remaining(engine_spec, token_extent);
std::size_t next_comma = remaining.find_first_of(',');
if (next_comma == std::string::npos) next_comma = engine_spec.size() - 1;
const std::string token(remaining, 0, next_comma);
parse_token(token, key, val);
map[key] = val;
token_extent += next_comma+1;
}
return map;
}
void Engine::Init(const std::string &engine_spec)
{
keyval_pair_t tokens(parse_engine_spec(engine_spec));
keyval_pair_t::iterator it;
it = tokens.find("exec_target");
if (it != tokens.end())
{
if (!std::strncmp(it->second.data(), "device", 6))
{
exec_target = Device;
device_number = 0;
}
else if (!std::strncmp(it->second.data(), "host", 4))
{
exec_target = Host;
device_number = -1;
}
else
{
mfem_error("Parse error. Possible values for exec_target are: ['host', 'device']");
}
}
else
{
// Default to host if not specified
mfem::out << "Did not specify exec_target. Defaulting to host..." << std::endl;
exec_target = Host;
device_number = -1;
}
it = tokens.find("mem_type");
if (it != tokens.end())
{
if (!std::strncmp(it->second.data(), "unified", 7))
{
#if defined(MFEM_USE_CUDAUM)
memory_resources[0] = new UnifiedMemoryResource();
unified_memory = true;
#else
mfem_error("Have not compiled support for CUDA unified memory.");
#endif
}
else if (!std::strncmp(it->second.data(), "separate", 4))
{
memory_resources[0] = new NewDeleteMemoryResource();
unified_memory = false;
}
else
{
mfem_error("Parse error. Possible values for mem_type are: ['separate', 'unified']");
}
}
else {
if (exec_target == Device)
{
#if defined(MFEM_USE_CUDAUM)
mfem::out << "Did not specify mem_type in engine spec. Defaulting to unified memory..." << std::endl;
// Default to unified memory
memory_resources[0] = new UnifiedMemoryResource();
unified_memory = true;
#else
mfem::out << "Did not specify mem_type in engine spec. Defaulting to standard host memory..." << std::endl;
memory_resources[0] = new NewDeleteMemoryResource();
unified_memory = false;
#endif
}
else
{
mfem::out << "Did not specify mem_type in engine spec. Defaulting to standard host memory..." << std::endl;
memory_resources[0] = new NewDeleteMemoryResource();
unified_memory = false;
}
}
it = tokens.find("mult_engine");
if (it != tokens.end())
{
if (!std::strncmp(it->second.data(), "acrotensor", 10))
{
mult_type = Acrotensor;
}
else
{
mfem_error("Parse error. Possible values for mem_type are: ['acrotensor'].");
}
}
else
{
mfem::out << "Did not specify mult_engine in engine spec. Defaulting to Acrotensor..." << std::endl;
#ifndef MFEM_USE_ACROTENSOR
mfem_error("Must compile with Acrotensor support");
#endif
mult_type = Acrotensor;
}
}
Engine::Engine(const std::string &engine_spec)
: mfem::Engine(NULL, 1, 1)
{
Init(engine_spec);
}
#ifdef MFEM_USE_MPI
Engine::Engine(MPI_Comm _comm, const std::string &engine_spec)
: mfem::Engine(NULL, 1, 1)
{
comm = _comm;
Init(engine_spec);
}
#endif
DLayout Engine::MakeLayout(std::size_t size) const
{
return DLayout(new Layout(*this, size));
}
DLayout Engine::MakeLayout(const mfem::Array<std::size_t> &offsets) const
{
MFEM_ASSERT(offsets.Size() == 2,
"multiple workers are not supported yet");
return DLayout(new Layout(*this, offsets.Last()));
}
DArray Engine::MakeArray(PLayout &layout, std::size_t item_size) const
{
MFEM_ASSERT(dynamic_cast<Layout *>(&layout) != NULL,
"invalid input layout");
Layout *lt = static_cast<Layout *>(&layout);
return DArray(new Array(*lt, item_size));
}
DVector Engine::MakeVector(PLayout &layout, int type_id) const
{
MFEM_ASSERT(type_id == ScalarId<double>::value, "invalid type_id");
MFEM_ASSERT(dynamic_cast<Layout *>(&layout) != NULL,
"invalid input layout");
Layout *lt = static_cast<Layout *>(&layout);
return DVector(new Vector(*lt));
}
DFiniteElementSpace Engine::MakeFESpace(mfem::FiniteElementSpace &fespace) const
{
return DFiniteElementSpace(new FiniteElementSpace(*this, fespace));
}
DBilinearForm Engine::MakeBilinearForm(mfem::BilinearForm &bf) const
{
return DBilinearForm(new BilinearForm(*this, bf));
}
void Engine::AssembleLinearForm(LinearForm &l_form) const
{
/// FIXME - What will the actual parameters be?
MFEM_ABORT("FIXME");
}
mfem::Operator *Engine::MakeOperator(const MixedBilinearForm &mbl_form) const
{
/// FIXME - What will the actual parameters be?
MFEM_ABORT("FIXME");
return NULL;
}
mfem::Operator *Engine::MakeOperator(const NonlinearForm &nl_form) const
{
/// FIXME - What will the actual parameters be?
MFEM_ABORT("FIXME");
return NULL;
}
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
-123
View File
@@ -1,123 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OMP_ENGINE_HPP
#define MFEM_BACKENDS_OMP_ENGINE_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "../base/engine.hpp"
namespace mfem
{
namespace omp
{
enum ExecutionTarget { Host, Device };
enum IntegratorType { Acrotensor };
class Engine : public mfem::Engine
{
protected:
//
// Inherited fields
//
// mfem::Backend *backend;
#ifdef MFEM_USE_MPI
// MPI_Comm comm;
#endif
// int num_mem_res;
// int num_workers;
// MemoryResource **memory_resources;
// double *workers_weights;
// int *workers_mem_res;
enum ExecutionTarget exec_target;
bool unified_memory;
int device_number;
IntegratorType mult_type;
void Init(const std::string &engine_spec);
public:
Engine(const std::string &engine_spec);
#ifdef MFEM_USE_MPI
Engine(MPI_Comm comm, const std::string &engine_spec);
#endif
virtual ~Engine() { }
/**
@name OMP specific interface, used by other objects in the OMP backend
*/
///@{
IntegratorType IntegType() const { return mult_type; }
ExecutionTarget ExecTarget() const { return exec_target; }
inline bool UnifiedMemory() const { return unified_memory; }
void* Malloc(std::size_t bytes) const
{
return memory_resources[0]->Allocate(bytes, 16);
}
void Dealloc(void *ptr, std::size_t bytes = 0) const
{
memory_resources[0]->Deallocate(ptr, bytes);
}
///@}
// End: OMP specific interface
/**
@name Virtual interface: finite element data structures and algorithms
*/
///@{
virtual DLayout MakeLayout(std::size_t size) const;
virtual DLayout MakeLayout(const mfem::Array<std::size_t> &offsets) const;
virtual DArray MakeArray(PLayout &layout, std::size_t item_size) const;
virtual DVector MakeVector(PLayout &layout,
int type_id = ScalarId<double>::value) const;
virtual DFiniteElementSpace MakeFESpace(mfem::FiniteElementSpace &
fespace) const;
virtual DBilinearForm MakeBilinearForm(mfem::BilinearForm &bf) const;
/// FIXME - What will the actual parameters be?
virtual void AssembleLinearForm(LinearForm &l_form) const;
/// FIXME - What will the actual parameters be?
virtual mfem::Operator *MakeOperator(const MixedBilinearForm &mbl_form) const;
/// FIXME - What will the actual parameters be?
virtual mfem::Operator *MakeOperator(const NonlinearForm &nl_form) const;
///@}
// End: Virtual interface
};
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#endif // MFEM_BACKENDS_OMP_ENGINE_HPP
-237
View File
@@ -1,237 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "fespace.hpp"
namespace mfem
{
namespace omp
{
FiniteElementSpace::FiniteElementSpace(const Engine &e,
mfem::FiniteElementSpace &fespace)
: PFiniteElementSpace(e, fespace),
e_layout(e, 0),
tensor_offsets(NULL),
tensor_indices(NULL),
prolongation(NULL),
restriction(NULL)
{
std::size_t lsize = 0;
for (int e = 0; e < fespace.GetNE(); e++) { lsize += fespace.GetFE(e)->GetDof(); }
e_layout.Resize(lsize);
// The e_layout will be stored inside multiple shared DLayout objects
e_layout.DontDelete();
}
void FiniteElementSpace::BuildDofMaps()
{
mfem::FiniteElementSpace *mfem_fes = GetFESpace();
const int local_size = GetELayout().Size();
const int global_size = mfem_fes->GetVLayout()->Size();
const int vdim = mfem_fes->GetVDim();
// Now we can allocate and fill the global map
tensor_offsets = new mfem::Array<int>(*(new Layout(OmpEngine(), global_size + 1)));
tensor_indices = new mfem::Array<int>(*(new Layout(OmpEngine(), local_size)));
mfem::Array<int> &offsets = *tensor_offsets;
mfem::Array<int> &indices = *tensor_indices;
mfem::Array<int> global_map(local_size);
mfem::Array<int> elem_vdof;
int offset = 0;
for (int e = 0; e < mfem_fes->GetNE(); e++)
{
const FiniteElement *fe = mfem_fes->GetFE(e);
const int dofs = fe->GetDof();
const TensorBasisElement *tfe = dynamic_cast<const TensorBasisElement *>(fe);
const mfem::Array<int> &dof_map = tfe->GetDofMap();
mfem_fes->GetElementVDofs(e, elem_vdof);
for (int vd = 0; vd < vdim; vd++)
for (int i = 0; i < dofs; i++)
{
global_map[offset + dofs*vd + i] = elem_vdof[dofs*vd + dof_map[i]];
}
offset += dofs * vdim;
}
// global_map[i] = index in global vector for local dof i
// NOTE: multiple i values will yield same global_map[i] for shared DOF.
// We want to now invert this map so we have indices[j] = (local dof for global dof j).
// Zero the offset vector
offsets = 0;
// Keep track of how many local dof point to its global dof
// Count how many times each dof gets hit
for (int i = 0; i < local_size; i++)
{
const int g = global_map[i];
++offsets[g + 1];
}
// Aggregate the offsets
for (int i = 1; i <= global_size; i++)
{
offsets[i] += offsets[i - 1];
}
for (int i = 0; i < local_size; i++)
{
const int g = global_map[i];
indices[offsets[g]++] = i;
}
// Shift the offset vector back by one, since it was used as a
// counter above.
for (int i = global_size; i > 0; i--)
{
offsets[i] = offsets[i - 1];
}
offsets[0] = 0;
offsets.Push();
indices.Push();
}
/// Convert an E vector to L vector
void FiniteElementSpace::ToLVector(const Vector &e_vector, Vector &l_vector)
{
if (tensor_indices == NULL) BuildDofMaps();
if (l_vector.Size() != (std::size_t) GetFESpace()->GetVSize())
{
l_vector.Resize<double>(GetFESpace()->GetVLayout(), NULL);
}
const int lsize = l_vector.Size();
const int *offsets = tensor_offsets->Get_PArray()->As<Array>().GetData<int>();
const int *indices = tensor_indices->Get_PArray()->As<Array>().GetData<int>();
const double *e_data = e_vector.GetData<double>();
double *l_data = l_vector.GetData<double>();
const bool use_target = l_vector.ComputeOnDevice();
const bool use_parallel = (use_target || lsize > 1000);
#pragma omp target teams distribute parallel for \
map (to: offsets, indices, l_data, e_data) \
if (target: use_target) \
if (parallel: use_parallel)
for (int i = 0; i < lsize; i++)
{
const int offset = offsets[i];
const int next_offset = offsets[i + 1];
double dof_value = 0;
for (int j = offset; j < next_offset; j++)
{
dof_value += e_data[indices[j]];
}
l_data[i] = dof_value;
}
}
/// Covert an L vector to E vector
void FiniteElementSpace::ToEVector(const Vector &l_vector, Vector &e_vector)
{
if (tensor_indices == NULL) BuildDofMaps();
if (e_vector.Size() != (std::size_t) e_layout.Size())
{
e_vector.Resize<double>(GetELayout(), NULL);
}
const int lsize = l_vector.Size();
const int *offsets = tensor_offsets->Get_PArray()->As<Array>().GetData<int>();
const int *indices = tensor_indices->Get_PArray()->As<Array>().GetData<int>();
const double *l_data = l_vector.GetData<double>();
double *e_data = e_vector.GetData<double>();
const bool use_target = l_vector.ComputeOnDevice();
const bool use_parallel = (use_target || lsize > 1000);
#pragma omp target teams distribute parallel for \
map (to: offsets, indices, l_data, e_data) \
if (target: use_target) \
if (parallel: use_parallel)
for (int i = 0; i < lsize; i++)
{
const int offset = offsets[i];
const int next_offset = offsets[i + 1];
const double dof_value = l_data[i];
for (int j = offset; j < next_offset; j++)
{
e_data[indices[j]] = dof_value;
}
}
}
/// Get the finite element space prolongation matrix
const Operator *FiniteElementSpace::GetProlongation() const
{
// FIXME: This relies on unified memory if using a device other than the CPU
if (!prolongation)
{
Layout &v_layout = GetVLayout();
Layout &t_layout = GetTrueVLayout();
const mfem::Operator *op = GetFESpace()->GetProlongationMatrix();
if (!op)
{
prolongation = new mfem::IdentityOperator(t_layout);
}
else
{
prolongation = new BackendOperator(t_layout, v_layout, op);
}
}
return prolongation;
}
/// Get the finite element space restriction matrix
const Operator *FiniteElementSpace::GetRestriction() const
{
// FIXME: This relies on unified memory if using a device other than the CPU
if (!restriction)
{
Layout &v_layout = GetVLayout();
Layout &t_layout = GetTrueVLayout();
const mfem::Operator *op = GetFESpace()->GetRestrictionMatrix();
if (!op)
{
restriction = new mfem::IdentityOperator(t_layout);
}
else
{
restriction = new BackendOperator(v_layout, t_layout, op);
}
}
return restriction;
}
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
-106
View File
@@ -1,106 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OMP_FESPACE_HPP
#define MFEM_BACKENDS_OMP_FESPACE_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "engine.hpp"
#include "array.hpp"
#include "vector.hpp"
#include "../../fem/fem.hpp"
namespace mfem
{
namespace omp
{
/*
Wraps an mfem::Operator that does not contain layout information.
*/
class BackendOperator : public mfem::Operator
{
const mfem::Operator *op;
public:
BackendOperator(Layout &in_layout, Layout &out_layout,
const mfem::Operator *op_) : Operator(in_layout, out_layout), op(op_) { }
virtual void Mult(const mfem::Vector &x, mfem::Vector &y) const { op->Mult(x, y); }
virtual void MultTranspose(const mfem::Vector &x, mfem::Vector &y) const { op->MultTranspose(x, y); }
};
/// TODO: doxygen
class FiniteElementSpace : public mfem::PFiniteElementSpace
{
protected:
//
// Inherited fields
//
// SharedPtr<const mfem::Engine> engine;
// mfem::FiniteElementSpace *fes;
Layout e_layout;
mfem::Array<int> *tensor_offsets, *tensor_indices;
mutable mfem::Operator *prolongation, *restriction;
void BuildDofMaps();
public:
/// Nearly-empty class that stores a pointer to a mfem::FiniteElementSpace instance and the engine
FiniteElementSpace(const Engine &e, mfem::FiniteElementSpace &fespace);
/// Virtual destructor
virtual ~FiniteElementSpace()
{
delete tensor_offsets;
delete tensor_indices;
delete prolongation;
delete restriction;
}
Layout &GetELayout() { return e_layout; }
Layout &GetVLayout() const
{ return *fes->GetVLayout().As<Layout>(); }
Layout &GetTrueVLayout() const
{ return *fes->GetTrueVLayout().As<Layout>(); }
/// Return the engine as an OpenMP engine
const Engine &OmpEngine() { return static_cast<const Engine&>(*engine); }
/// Convert an E vector to L vector
void ToLVector(const Vector &e_vector, Vector &l_vector);
/// Covert an L vector to E vector
void ToEVector(const Vector &l_vector, Vector &e_vector);
/// Get the finite element space prolongation matrix
const Operator *GetProlongation() const;
/// Get the finite element space restriction matrix
const Operator *GetRestriction() const;
};
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#endif // MFEM_BACKENDS_OMP_FESPACE_HPP
-40
View File
@@ -1,40 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "layout.hpp"
#include "../../general/array.hpp"
namespace mfem
{
namespace omp
{
void Layout::Resize(std::size_t new_size)
{
size = new_size;
}
void Layout::Resize(const Array<std::size_t> &offsets)
{
MFEM_ASSERT(offsets.Size() == 2,
"multiple workers are not supported yet");
size = offsets.Last();
}
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
-71
View File
@@ -1,71 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OMP_LAYOUT_HPP
#define MFEM_BACKENDS_OMP_LAYOUT_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "../base/layout.hpp"
#include "engine.hpp"
namespace mfem
{
namespace omp
{
class Layout : public mfem::PLayout
{
protected:
//
// Inherited fields
//
// SharedPtr<const mfem::Engine> engine;
// std::size_t size;
public:
Layout(const Engine &e, std::size_t s = 0) : PLayout(e, s) { }
const Engine &OmpEngine() const
{ return *static_cast<const Engine *>(engine.Get()); }
void *Alloc(std::size_t bytes) const
{ return OmpEngine().Malloc(bytes); }
void Dealloc(void *ptr) const
{ return OmpEngine().Dealloc(ptr); }
virtual ~Layout() { }
/**
@name Virtual interface
*/
///@{
/// Resize the layout
virtual void Resize(std::size_t new_size);
/// Resize the layout based on the given worker offsets
virtual void Resize(const Array<std::size_t> &offsets);
///@}
// End: Virtual interface
};
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#endif // MFEM_BACKENDS_OMP_LAYOUT_HPP
-57
View File
@@ -1,57 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "memory_resource.hpp"
#include "../../general/error.hpp"
#ifdef MFEM_USE_CUDAUM
#include "cuda_runtime.h"
#include "cuda.h"
#endif
namespace mfem
{
namespace omp
{
#ifdef MFEM_USE_CUDAUM
void *UnifiedMemoryResource::DoAllocate(std::size_t bytes,
std::size_t alignment)
{
void *p = NULL;
if (bytes > 0)
{
cudaError_t ret = cudaMallocManaged(&p, bytes);
MFEM_VERIFY(ret == cudaSuccess, "");
}
return p;
}
void UnifiedMemoryResource::DoDeallocate(void *p, std::size_t bytes,
std::size_t alignment)
{
if (p != NULL)
{
cudaFree(p);
}
}
#endif
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
-44
View File
@@ -1,44 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OMP_MEMORY_RESOURCE_HPP
#define MFEM_BACKENDS_OMP_MEMORY_RESOURCE_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "../../backends/base/memory_resource.hpp"
namespace mfem
{
namespace omp
{
/// Polymorphic memory resource. Similar to C++17's std::pmr::memory_resource.
#ifdef MFEM_USE_CUDAUM
/** @brief Memory resource using unified memory. */
class UnifiedMemoryResource : public MemoryResource
{
protected:
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment);
virtual void DoDeallocate(void *p, std::size_t bytes, std::size_t alignment);
};
#endif
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#endif // MFEM_BACKENDS_OMP_MEMORY_RESOURCE_HPP
-205
View File
@@ -1,205 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "vector.hpp"
#include "../../linalg/vector.hpp"
namespace mfem
{
namespace omp
{
PVector *Vector::DoVectorClone(bool copy_data, void **buffer,
int buffer_type_id) const
{
MFEM_ASSERT(buffer_type_id == ScalarId<double>::value, "");
Vector *new_vector = new Vector(OmpLayout());
if (copy_data)
{
const std::size_t total_size = sizeof(double) * OmpLayout().Size();
if (!ComputeOnDevice())
std::memcpy(new_vector->GetData<void>(), data, total_size);
else
{
char *new_data = new_vector->GetData<char>();
#pragma omp target teams distribute parallel for is_device_ptr(new_data)
for (std::size_t i = 0; i < total_size; i++) new_data[i] = data[i];
}
}
if (buffer)
{
*buffer = new_vector->GetData<void>();
}
return new_vector;
}
void Vector::DoDotProduct(const PVector &x, void *result,
int result_type_id) const
{
// Can be called when Size() == 0, e.g. when an MPI-parallel vector has a
// local size of 0.
MFEM_ASSERT(result_type_id == ScalarId<double>::value, "");
double *res = (double *)result;
double local_dot = 0.;
MFEM_ASSERT(dynamic_cast<const Vector *>(&x) != NULL, "invalid Vector type");
const Vector *xp = static_cast<const Vector *>(&x);
MFEM_ASSERT(this->Size() == xp->Size(), "");
const double *ptr = GetData<double>();
const double *xptr = xp->GetData<double>();
const std::size_t size = Size();
if (!ComputeOnDevice())
{
for (std::size_t i = 0; i < size; i++) local_dot += ptr[i] * xptr[i];
}
else
{
#pragma omp target teams distribute parallel for map(to: ptr, xptr) reduction(+:local_dot)
for (std::size_t i = 0; i < size; i++) local_dot += ptr[i] * xptr[i];
}
*res = local_dot;
#ifdef MFEM_USE_MPI
MPI_Comm comm = OmpLayout().OmpEngine().GetComm();
if (comm != MPI_COMM_NULL)
{
MPI_Allreduce(&local_dot, res, 1, MPI_DOUBLE, MPI_SUM, comm);
}
#endif
}
void Vector::DoAxpby(const void *a, const PVector &x,
const void *b, const PVector &y,
int ab_type_id)
{
// called only when Size() != 0
MFEM_ASSERT(ab_type_id == ScalarId<double>::value, "");
const double da = *static_cast<const double *>(a);
const double db = *static_cast<const double *>(b);
MFEM_ASSERT(da == 0.0 || dynamic_cast<const Vector *>(&x) != NULL,
"invalid Vector x");
MFEM_ASSERT(db == 0.0 || dynamic_cast<const Vector *>(&y) != NULL,
"invalid Vector y");
const Vector *xp = static_cast<const Vector *>(&x);
const Vector *yp = static_cast<const Vector *>(&y);
MFEM_ASSERT(da == 0.0 || this->Size() == xp->Size(), "");
MFEM_ASSERT(db == 0.0 || this->Size() == yp->Size(), "");
const std::size_t size = Size();
const std::size_t critical_size = 1000;
const double *xd = xp->GetData<double>();
const double *yd = yp->GetData<double>();
double *td = GetData<double>();
const bool use_target = ComputeOnDevice();
const bool use_parallel = (use_target || size > critical_size);
if (da == 0.0)
{
if (db == 0.0)
{
OmpFill(&da);
}
else
{
if (td == yd)
{
// *this *= db
#pragma omp target teams distribute parallel for \
if (target: use_target) \
if (parallel: use_parallel) map (to: db)
for (std::size_t i = 0; i < size; i++) td[i] *= db;
}
else
{
// *this = db * y
#pragma omp target teams distribute parallel for \
if (target: use_target) \
if (parallel: use_parallel) map (to: yd, db)
for (std::size_t i = 0; i < size; i++) td[i] = yd[i] * db;
}
}
}
else
{
if (db == 0.0)
{
if (td == xd)
{
// *this *= da
#pragma omp target teams distribute parallel for \
if (target: use_target) \
if (parallel: use_parallel) map (to: da)
for (std::size_t i = 0; i < size; i++) td[i] *= da;
}
else
{
// *this = da * x
#pragma omp target teams distribute parallel for \
if (target: use_target) \
if (parallel: use_parallel) map (to: xd, da)
for (std::size_t i = 0; i < size; i++) td[i] = xd[i] * da;
}
}
else
{
MFEM_ASSERT(xd != yd, "invalid input");
if (td == xd)
{
// *this = da * (*this) + db * y
#pragma omp target teams distribute parallel for \
if (target: use_target) \
if (parallel: use_parallel) map (to: da, td, db, yd)
for (std::size_t i = 0; i < size; i++) td[i] = da * td[i] + db * yd[i];
}
else if (td == yd)
{
// *this = da * x + db * (*this)
#pragma omp target teams distribute parallel for \
if (target: use_target) \
if (parallel: use_parallel) map (to: da, xd, db, td)
for (std::size_t i = 0; i < size; i++) td[i] = da * xd[i] + db * td[i];
}
else
{
// *this = da * x + db * y
#pragma omp target teams distribute parallel for \
if (target: use_target) \
if (parallel: use_parallel) map (to: da, xd, db, yd)
for (std::size_t i = 0; i < size; i++) td[i] = da * xd[i] + db * yd[i];
}
}
}
}
mfem::Vector Vector::Wrap()
{
return mfem::Vector(*this);
}
const mfem::Vector Vector::Wrap() const
{
return mfem::Vector(*const_cast<Vector*>(this));
}
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
-71
View File
@@ -1,71 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_BACKENDS_OMP_VECTOR_HPP
#define MFEM_BACKENDS_OMP_VECTOR_HPP
#include "../../config/config.hpp"
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#include "../base/vector.hpp"
#include "array.hpp"
namespace mfem
{
namespace omp
{
class Vector : virtual public Array, public mfem::PVector
{
protected:
//
// Inherited fields
//
// DLayout layout;
// char *data;
// std::size_t size;
/**
@name Virtual interface
*/
///@{
virtual PVector *DoVectorClone(bool copy_data, void **buffer,
int buffer_type_id) const;
virtual void DoDotProduct(const PVector &x, void *result,
int result_type_id) const;
virtual void DoAxpby(const void *a, const PVector &x,
const void *b, const PVector &y,
int ab_type_id);
///@}
// End: Virtual interface
public:
Vector(Layout &lt)
: PArray(lt), Array(lt, sizeof(double)), PVector(lt)
{ }
mfem::Vector Wrap();
const mfem::Vector Wrap() const;
};
} // namespace mfem::omp
} // namespace mfem
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OMP)
#endif // MFEM_BACKENDS_OMP_VECTOR_HPP
-182
View File
@@ -1,182 +0,0 @@
##################################################################################
#
# Set defaults for XSDK CMake projects
#
##################################################################################
#
# This module implements standard behavior for XSDK CMake projects. The main
# thing it does in XSDK mode (i.e. USE_XSDK_DEFAULTS=TRUE) is to print out
# when the env vars CC, CXX, FC and compiler flags CFLAGS, CXXFLAGS, and
# FFLAGS/FCFLAGS are used to select the compilers and compiler flags (raw
# CMake does this silently) and to set BUILD_SHARED_LIBS=TRUE and
# CMAKE_BUILD_TYPE=DEBUG by default. It does not implement *all* of the
# standard XSDK configuration parameters. The parent CMake project must do
# that.
#
# Note that when USE_XSDK_DEFAULTS=TRUE, then the Fortran flags will be read
# from either of the env vars FFLAGS or FCFLAGS. If both are set, but are the
# same, then FFLAGS it used (which is the same as FCFLAGS). However, if both
# are set but are not equal, then a FATAL_ERROR is raised and CMake configure
# processing is stopped.
#
# To be used in a parent project, this module must be included after
#
# PROJECT(${PROJECT_NAME} NONE)
#
# is called but before the compilers are defined and processed using:
#
# ENABLE_LANGUAGE(<LANG>)
#
# For example, one would do:
#
# PROJECT(${PROJECT_NAME} NONE)
# ...
# SET(USE_XSDK_DEFAULTS_DEFAULT TRUE) # Set to false if desired
# INCLUDE("${CMAKE_CURRENT_SOURCE_DIR}/stdk/XSDKDefaults.cmake")
# ...
# ENABLE_LANGUAGE(C)
# ENABLE_LANGUAGE(C++)
# ENABLE_LANGUAGE(Fortran)
#
# The variable `USE_XSDK_DEFAULTS_DEFAULT` is used as the default for the
# cache var `USE_XSDK_DEFAULTS`. That way, a project can decide if it wants
# XSDK defaults turned on or off by default and users can independently decide
# if they want the CMake project to use standard XSDK behavior or raw CMake
# behavior.
#
# By default, the XSDKDefaults.cmake module assumes that the project will need
# C, C++, and Fortran. If any language is not needed then, set
# XSDK_ENABLE_C=OFF, XSDK_ENABLE_CXX=OFF, or XSDK_ENABLE_Fortran=OFF *before*
# including this module. Note, these variables are *not* cache vars because a
# project either does or does not have C, C++ or Fortran source files, the
# user has nothing to do with this so there is no need for cache vars. The
# parent CMake project just needs to tell XSDKDefault.cmake what languages is
# needs or does not need.
#
# For example, if the parent CMake project only needs C, then it would do:
#
# PROJECT(${PROJECT_NAME} NONE)'
# ...
# SET(USE_XSDK_DEFAULTS_DEFAULT TRUE)
# SET(XSDK_ENABLE_CXX OFF)
# SET(XSDK_ENABLE_Fortran OFF)
# INCLUDE("${CMAKE_CURRENT_SOURCE_DIR}/stdk/XSDKDefaults.cmake")
# ...
# ENABLE_LANGAUGE(C)
#
# This module code will announce when it sets any variables.
#
#
# Helper functions
#
IF (NOT COMMAND PRINT_VAR)
FUNCTION(PRINT_VAR VAR_NAME)
MESSAGE("-- " "${VAR_NAME} = '${${VAR_NAME}}'")
ENDFUNCTION()
ENDIF()
IF (NOT COMMAND SET_DEFAULT)
MACRO(SET_DEFAULT VAR)
IF ("${${VAR}}" STREQUAL "")
SET(${VAR} ${ARGN})
ENDIF()
ENDMACRO()
ENDIF()
#
# XSDKDefaults.cmake control variables
#
# USE_XSDK_DEFAULTS
IF ("${USE_XSDK_DEFAULTS_DEFAULT}" STREQUAL "")
SET(USE_XSDK_DEFAULTS_DEFAULT FALSE)
ENDIF()
SET(USE_XSDK_DEFAULTS ${USE_XSDK_DEFAULTS_DEFAULT} CACHE BOOL
"Use XSDK defaults and behavior.")
PRINT_VAR(USE_XSDK_DEFAULTS)
SET_DEFAULT(XSDK_ENABLE_C TRUE)
SET_DEFAULT(XSDK_ENABLE_CXX TRUE)
SET_DEFAULT(XSDK_ENABLE_Fortran TRUE)
# Handle the compiler and flags for a language
MACRO(XSDK_HANDLE_LANG_DEFAULTS CMAKE_LANG_NAME ENV_LANG_NAME
ENV_LANG_FLAGS_NAMES
)
# Announce using env var ${ENV_LANG_NAME}
IF (NOT "$ENV{${ENV_LANG_NAME}}" STREQUAL "" AND
"${CMAKE_${CMAKE_LANG_NAME}_COMPILER}" STREQUAL ""
)
MESSAGE("-- " "XSDK: Setting CMAKE_${CMAKE_LANG_NAME}_COMPILER from env var"
" ${ENV_LANG_NAME}='$ENV{${ENV_LANG_NAME}}'!")
SET(CMAKE_${CMAKE_LANG_NAME}_COMPILER "$ENV{${ENV_LANG_NAME}}" CACHE FILEPATH
"XSDK: Set by default from env var ${ENV_LANG_NAME}")
ENDIF()
# Announce using env var ${ENV_LANG_FLAGS_NAME}
FOREACH(ENV_LANG_FLAGS_NAME ${ENV_LANG_FLAGS_NAMES})
IF (NOT "$ENV{${ENV_LANG_FLAGS_NAME}}" STREQUAL "" AND
"${CMAKE_${CMAKE_LANG_NAME}_FLAGS}" STREQUAL ""
)
MESSAGE("-- " "XSDK: Setting CMAKE_${CMAKE_LANG_NAME}_FLAGS from env var"
" ${ENV_LANG_FLAGS_NAME}='$ENV{${ENV_LANG_FLAGS_NAME}}'!")
SET(CMAKE_${CMAKE_LANG_NAME}_FLAGS "$ENV{${ENV_LANG_FLAGS_NAME}} " CACHE STRING
"XSDK: Set by default from env var ${ENV_LANG_FLAGS_NAME}")
# NOTE: CMake adds the space after $ENV{${ENV_LANG_FLAGS_NAME}} so we
# duplicate that here!
ENDIF()
ENDFOREACH()
ENDMACRO()
#
# Set XSDK Defaults
#
# Set default compilers and flags
IF (USE_XSDK_DEFAULTS)
# Handle env vars for languages C, C++, and Fortran
IF (XSDK_ENABLE_C)
XSDK_HANDLE_LANG_DEFAULTS(C CC CFLAGS)
ENDIF()
IF (XSDK_ENABLE_CXX)
XSDK_HANDLE_LANG_DEFAULTS(CXX CXX CXXFLAGS)
ENDIF()
IF (XSDK_ENABLE_Fortran)
SET(ENV_FFLAGS "$ENV{FFLAGS}")
SET(ENV_FCFLAGS "$ENV{FCFLAGS}")
IF (
(NOT "${ENV_FFLAGS}" STREQUAL "") AND (NOT "${ENV_FCFLAGS}" STREQUAL "")
AND
("${CMAKE_Fortran_FLAGS}" STREQUAL "")
)
IF (NOT "${ENV_FFLAGS}" STREQUAL "${ENV_FCFLAGS}")
MESSAGE(FATAL_ERROR "Error, env vars FFLAGS='${ENV_FFLAGS}' and"
" FCFLAGS='${ENV_FCFLAGS}' are both set in the env but are not equal!")
ENDIF()
ENDIF()
XSDK_HANDLE_LANG_DEFAULTS(Fortran FC "FFLAGS;FCFLAGS")
ENDIF()
# Set XSDK defaults for other CMake variables
IF ("${BUILD_SHARED_LIBS}" STREQUAL "")
MESSAGE("-- " "XSDK: Setting default BUILD_SHARED_LIBS=TRUE")
SET(BUILD_SHARED_LIBS TRUE CACHE BOOL "Set by default in XSDK mode")
ENDIF()
IF ("${CMAKE_BUILD_TYPE}" STREQUAL "")
MESSAGE("-- " "XSDK: Setting default CMAKE_BUILD_TYPE=DEBUG")
SET(CMAKE_BUILD_TYPE DEBUG CACHE STRING "Set by default in XSDK mode")
ENDIF()
ENDIF()
+2 -12
View File
@@ -12,41 +12,31 @@
include(${CMAKE_CURRENT_LIST_DIR}/MFEMConfigVersion.cmake)
set(MFEM_VERSION ${PACKAGE_VERSION})
set(MFEM_VERSION_INT @MFEM_VERSION@)
set(MFEM_GIT_STRING "@MFEM_GIT_STRING@")
set(MFEM_USE_MPI @MFEM_USE_MPI@)
set(MFEM_USE_METIS @MFEM_USE_METIS@)
set(MFEM_USE_METIS_5 @MFEM_USE_METIS_5@)
set(MFEM_DEBUG @MFEM_DEBUG@)
set(MFEM_USE_EXCEPTIONS @MFEM_USE_EXCEPTIONS@)
set(MFEM_USE_GZSTREAM @MFEM_USE_GZSTREAM@)
set(MFEM_USE_LIBUNWIND @MFEM_USE_LIBUNWIND@)
set(MFEM_USE_LAPACK @MFEM_USE_LAPACK@)
set(MFEM_THREAD_SAFE @MFEM_THREAD_SAFE@)
set(MFEM_USE_OPENMP @MFEM_USE_OPENMP@)
set(MFEM_USE_MEMALLOC @MFEM_USE_MEMALLOC@)
set(MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@)
set(MFEM_USE_SUNDIALS @MFEM_USE_SUNDIALS@)
set(MFEM_USE_MESQUITE @MFEM_USE_MESQUITE@)
set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
set(MFEM_USE_GECKO @MFEM_USE_GECKO@)
set(MFEM_USE_GNUTLS @MFEM_USE_GNUTLS@)
set(MFEM_USE_NETCDF @MFEM_USE_NETCDF@)
set(MFEM_USE_PETSC @MFEM_USE_PETSC@)
set(MFEM_USE_MPFR @MFEM_USE_MPFR@)
set(MFEM_USE_SIDRE @MFEM_USE_SIDRE@)
set(MFEM_USE_CONDUIT @MFEM_USE_CONDUIT@)
set(MFEM_CXX_COMPILER "@CMAKE_CXX_COMPILER@")
set(MFEM_CXX_FLAGS "@CMAKE_CXX_FLAGS@")
@PACKAGE_INIT@
set(MFEM_INCLUDE_DIRS "@PACKAGE_INCLUDE_INSTALL_DIRS@")
foreach (dir ${MFEM_INCLUDE_DIRS})
message("DIR = ${dir}")
set_and_check(MFEM_INCLUDE_DIR "${dir}")
endforeach (dir "${MFEM_INCLUDE_DIRS}")
+5 -37
View File
@@ -12,27 +12,6 @@
#ifndef MFEM_CONFIG_HEADER
#define MFEM_CONFIG_HEADER
// MFEM version: integer of the form: (major*100 + minor)*100 + patch.
#cmakedefine MFEM_VERSION @MFEM_VERSION@
// MFEM version string of the form "3.3" or "3.3.1".
#cmakedefine MFEM_VERSION_STRING "@MFEM_VERSION_STRING@"
// MFEM version type, see the MFEM_VERSION_TYPE_* constants below.
#define MFEM_VERSION_TYPE ((MFEM_VERSION)%2)
// MFEM version type constants.
#define MFEM_VERSION_TYPE_RELEASE 0
#define MFEM_VERSION_TYPE_DEVELOPMENT 1
// Separate MFEM version numbers for major, minor, and patch.
#define MFEM_VERSION_MAJOR ((MFEM_VERSION)/10000)
#define MFEM_VERSION_MINOR (((MFEM_VERSION)/100)%100)
#define MFEM_VERSION_PATCH ((MFEM_VERSION)%100)
// Description of the git commit used to build MFEM.
#cmakedefine MFEM_GIT_STRING "@MFEM_GIT_STRING@"
// Build the parallel MFEM library.
// Requires an MPI compiler, and the libraries HYPRE and METIS.
#cmakedefine MFEM_USE_MPI
@@ -40,18 +19,12 @@
// Enable debug checks in MFEM.
#cmakedefine MFEM_DEBUG
// Throw an exception on errors.
#cmakedefine MFEM_USE_EXCEPTIONS
// Enable gzstream in MFEM.
#cmakedefine MFEM_USE_GZSTREAM
// Enable backtraces for mfem_error through libunwind.
#cmakedefine MFEM_USE_LIBUNWIND
// Enable MFEM features that use the METIS library (parallel MFEM).
#cmakedefine MFEM_USE_METIS
// Enable this option if linking with METIS version 5 (parallel MFEM).
#cmakedefine MFEM_USE_METIS_5
@@ -74,9 +47,6 @@
// Enable MFEM functionality based on the SuperLU_DIST library.
#cmakedefine MFEM_USE_SUPERLU
// Enable MFEM functionality based on the STRUMPACK library.
#cmakedefine MFEM_USE_STRUMPACK
// Internal MFEM option: enable group/batch allocation for some small objects.
#cmakedefine MFEM_USE_MEMALLOC
@@ -95,11 +65,12 @@
// Enable MFEM functionality based on the Sidre library
#cmakedefine MFEM_USE_SIDRE
// Enable MFEM functionality based on Conduit
#cmakedefine MFEM_USE_CONDUIT
// Which library functions to use in class StopWatch for measuring time.
// For a list of the available options, see INSTALL.
// The available options are:
// 0 - use std::clock from <ctime>
// 1 - use times from <sys/times.h>
// 2 - use high-resolution POSIX clocks
// 3 - use QueryPerformanceCounter from <windows.h>
// If not defined, an option is selected automatically.
#define MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@
@@ -110,7 +81,4 @@
// Macro needed to get defines like M_PI from <cmath>. (Visual Studio C++ only?)
#cmakedefine _USE_MATH_DEFINES
// Version of HYPRE used for building MFEM.
#cmakedefine MFEM_HYPRE_VERSION @MFEM_HYPRE_VERSION@
#endif // MFEM_CONFIG_HEADER
@@ -10,14 +10,15 @@
# Software Foundation) version 2.1 dated February 1999.
# Defines the following variables:
# - AXOM_FOUND
# - AXOM_LIBRARIES
# - AXOM_INCLUDE_DIRS
# - ATK_FOUND
# - ATK_LIBRARIES
# - ATK_INCLUDE_DIRS
include(MfemCmakeUtilities)
# Note: components are enabled based on the find_package() parameters.
mfem_find_package(Axom AXOM AXOM_DIR "include" "" "lib" ""
"Paths to headers required by Axom." "Libraries required by Axom."
mfem_find_package(ATK ATK ATK_DIR "include" "" "lib" ""
"Paths to headers required by ATK." "Libraries required by ATK."
ADD_COMPONENT Sidre "include" sidre/sidre.hpp "lib" sidre
ADD_COMPONENT SPIO "include" spio/IOManager.hpp "lib" spio
ADD_COMPONENT SLIC "include" slic/slic.hpp "lib" slic
ADD_COMPONENT axom_utils "include" axom_utils/Utilities.hpp "lib" axom_utils)
ADD_COMPONENT common "include" common/ATKMacros.hpp "lib" common)
+1 -14
View File
@@ -14,22 +14,9 @@
# - CONDUIT_LIBRARIES
# - CONDUIT_INCLUDE_DIRS
# check to see if relay requires hdf5, if so make sure to set HDF5
# as a required dep
if(EXISTS ${CONDUIT_DIR}/include/conduit/conduit_relay_hdf5.hpp)
message(STATUS "Conduit Relay HDF5 Support is ENABLED")
# we only need HDF5 if Conduit was built with HDF5 support
set(Conduit_REQUIRED_PACKAGES "HDF5" CACHE STRING
"Additional packages required by Conduit.")
else()
message(STATUS "Conduit Relay HDF5 Support is DISABLED")
endif()
include(MfemCmakeUtilities)
mfem_find_package(Conduit CONDUIT CONDUIT_DIR
"include;include/conduit" conduit.hpp "lib" conduit
"Paths to headers required by Conduit." "Libraries required by Conduit."
ADD_COMPONENT relay
"include;include/conduit" conduit_relay.hpp "lib" conduit_relay
ADD_COMPONENT blueprint
"include;include/conduit" conduit_blueprint.hpp "lib" conduit_blueprint)
"include;include/conduit" conduit_relay.hpp "lib" conduit_relay)
-16
View File
@@ -13,23 +13,7 @@
# - HYPRE_FOUND
# - HYPRE_LIBRARIES
# - HYPRE_INCLUDE_DIRS
# - HYPRE_VERSION
include(MfemCmakeUtilities)
mfem_find_package(HYPRE HYPRE HYPRE_DIR "include" "HYPRE.h" "lib" "HYPRE"
"Paths to headers required by HYPRE." "Libraries required by HYPRE.")
if (HYPRE_FOUND AND (NOT HYPRE_VERSION))
try_run(HYPRE_VERSION_RUN_RESULT HYPRE_VERSION_COMPILE_RESULT
${CMAKE_CURRENT_BINARY_DIR}/config
${CMAKE_CURRENT_SOURCE_DIR}/config/get_hypre_version.cpp
CMAKE_FLAGS -DINCLUDE_DIRECTORIES:STRING=${HYPRE_INCLUDE_DIRS}
RUN_OUTPUT_VARIABLE HYPRE_VERSION_OUTPUT)
if ((HYPRE_VERSION_RUN_RESULT EQUAL 0) AND HYPRE_VERSION_OUTPUT)
string(STRIP "${HYPRE_VERSION_OUTPUT}" HYPRE_VERSION)
set(HYPRE_VERSION ${HYPRE_VERSION} CACHE STRING "HYPRE version." FORCE)
message(STATUS "Found HYPRE version ${HYPRE_VERSION}")
else()
message(FATAL_ERROR "Unable to determine HYPRE version.")
endif()
endif()
-36
View File
@@ -1,36 +0,0 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Sets the following variables:
# - STRUMPACK_FOUND
# - STRUMPACK_INCLUDE_DIRS
# - STRUMPACK_LIBRARIES
include(MfemCmakeUtilities)
mfem_find_package(STRUMPACK STRUMPACK STRUMPACK_DIR
"include" "StrumpackSparseSolverMPIDist.hpp"
"lib" "strumpack;strumpack_sparse" # add NAMES_PER_DIR?
"Paths to headers required by STRUMPACK."
"Libraries required by STRUMPACK."
CHECK_BUILD STRUMPACK_VERSION_OK TRUE
"
#include <StrumpackSparseSolverMPIDist.hpp>
using namespace strumpack;
int main(int argc, char *argv[])
{
MPI_Init(&argc, &argv);
MPI_Comm comm = MPI_COMM_WORLD;
StrumpackSparseSolverMPIDist<double,int> solver(comm, argc, argv, false);
solver.options().set_from_command_line();
return 0;
}
"
)
-29
View File
@@ -1,29 +0,0 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Sets the following variables:
# - Scotch_FOUND
# - Scotch_INCLUDE_DIRS
# - Scotch_LIBRARIES
include(MfemCmakeUtilities)
mfem_find_package(Scotch Scotch Scotch_DIR "" "" "" ""
"Paths to headers required by Scotch."
"Libraries required by Scotch."
ADD_COMPONENT "scotch" "include" scotch.h "lib" scotch
ADD_COMPONENT "scotcherr" "" "" "lib" scotcherr
ADD_COMPONENT "scotcherrexit" "" "" "lib" scotcherrexit
ADD_COMPONENT "scotchmetis" "include" "metis.h" "lib" scotchmetis
ADD_COMPONENT "ptscotch" "include" ptscotch.h "lib" ptscotch
ADD_COMPONENT "ptscotcherr" "" "" "lib" ptscotcherr
ADD_COMPONENT "ptscotcherrexit" "" "" "lib" ptscotcherrexit
ADD_COMPONENT "ptscotchparmetis" "include" "parmetis.h" "lib" ptscotchparmetis
)
+44 -197
View File
@@ -9,30 +9,6 @@
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Function that converts a version string of the form 'major[.minor[.patch]]' to
# the integer ((major * 100) + minor) * 100 + patch.
function(mfem_version_to_int VersionString VersionIntVar)
if ("${VersionString}" MATCHES "^([0-9]+)(.*)$")
set(Major "${CMAKE_MATCH_1}")
set(MinorPatchString "${CMAKE_MATCH_2}")
else()
set(Major 0)
endif()
if ("${MinorPatchString}" MATCHES "^\\.([0-9]+)(.*)$")
set(Minor "${CMAKE_MATCH_1}")
set(PatchString "${CMAKE_MATCH_2}")
else()
set(Minor 0)
endif()
if ("${PatchString}" MATCHES "^\\.([0-9]+)(.*)$")
set(Patch "${CMAKE_MATCH_1}")
else()
set(Patch 0)
endif()
math(EXPR VersionInt "(${Major}*100+${Minor})*100+${Patch}")
set(${VersionIntVar} ${VersionInt} PARENT_SCOPE)
endfunction()
# A handy function to add the current source directory to a local
# filename. To be used for creating a list of sources.
function(convert_filenames_to_full_paths NAMES)
@@ -74,8 +50,10 @@ function(add_mfem_examples EXE_SRCS)
string(REPLACE ".cpp" "" EXE_NAME "${EXE_PREFIX}${SRC_FILENAME}")
add_executable(${EXE_NAME} ${SRC_FILE})
add_dependencies(${MFEM_ALL_EXAMPLES_TARGET_NAME} ${EXE_NAME})
if (EXE_NEEDED_BY)
# If given a prefix, don't add the example to the list of examples to build.
if (NOT EXE_PREFIX)
add_dependencies(${MFEM_ALL_EXAMPLES_TARGET_NAME} ${EXE_NAME})
elseif (EXE_NEEDED_BY)
add_dependencies(${EXE_NEEDED_BY} ${EXE_NAME})
endif()
add_dependencies(${EXE_NAME}
@@ -83,8 +61,7 @@ function(add_mfem_examples EXE_SRCS)
target_link_libraries(${EXE_NAME} mfem)
if (MFEM_USE_MPI)
# Not needed: (mfem already links with MPI_CXX_LIBRARIES)
# target_link_libraries(${EXE_NAME} ${MPI_CXX_LIBRARIES})
target_link_libraries(${EXE_NAME} ${MPI_CXX_LIBRARIES})
# Language-specific include directories:
if (MPI_CXX_INCLUDE_PATH)
@@ -155,7 +132,6 @@ function(add_mfem_miniapp MFEM_EXE_NAME)
# Handle the MPI separately
if (MFEM_USE_MPI)
# Add MPI_CXX_LIBRARIES, in case this target does not link with mfem.
if(CMAKE_VERSION VERSION_GREATER 2.8.11)
target_link_libraries(${MFEM_EXE_NAME} PRIVATE ${MPI_CXX_LIBRARIES})
else()
@@ -183,26 +159,26 @@ function(mfem_find_component Prefix DirVar IncSuffixes Header LibSuffixes Lib
if (Lib)
if (${DirVar} OR EnvDirVar)
find_library(${Prefix}_LIBRARY ${Lib}
find_library(${Prefix}_LIBRARIES ${Lib}
HINTS ${${DirVar}} ENV ${DirVar}
PATH_SUFFIXES ${LibSuffixes}
NO_DEFAULT_PATH
DOC "${LibDoc}")
endif()
find_library(${Prefix}_LIBRARY ${Lib}
find_library(${Prefix}_LIBRARIES ${Lib}
PATH_SUFFIXES ${LibSuffixes}
DOC "${LibDoc}")
endif()
if (Header)
if (${DirVar} OR EnvDirVar)
find_path(${Prefix}_INCLUDE_DIR ${Header}
find_path(${Prefix}_INCLUDE_DIRS ${Header}
HINTS ${${DirVar}} ENV ${DirVar}
PATH_SUFFIXES ${IncSuffixes}
NO_DEFAULT_PATH
DOC "${IncDoc}")
endif()
find_path(${Prefix}_INCLUDE_DIR ${Header}
find_path(${Prefix}_INCLUDE_DIRS ${Header}
PATH_SUFFIXES ${IncSuffixes}
DOC "${IncDoc}")
endif()
@@ -214,9 +190,8 @@ endfunction(mfem_find_component)
# successful, optionally checks building (compile + link) one or more given
# code snippets. Additionally, a list of required/optional/alternative
# packages (given by ${Name}_REQUIRED_PACKAGES) are searched for and added to
# the ${Prefix}_INCLUDE_DIRS and ${Prefix}_LIBRARIES lists. The variable
# ${Name}_REQUIRED_LIBRARIES can be set to spcecify any additional libraries
# that are needed. This function defines the following CACHE variables:
# the ${Prefix}_INCLUDE_DIRS and ${Prefix}_LIBRARIES lists. The function
# defines the following CACHE variables:
#
# ${Prefix}_FOUND
# ${Prefix}_INCLUDE_DIRS
@@ -255,14 +230,12 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
mfem_find_component("${Prefix}" "${DirVar}" "${IncSuffixes}" "${Header}"
"${LibSuffixes}" "${Lib}" "${IncDoc}" "${LibDoc}")
if (((NOT Lib) OR ${Prefix}_LIBRARY) AND
((NOT Header) OR ${Prefix}_INCLUDE_DIR))
if (((NOT Lib) OR ${Prefix}_LIBRARIES) AND
((NOT Header) OR ${Prefix}_INCLUDE_DIRS))
set(Found TRUE)
else()
set(Found FALSE)
endif()
set(${Prefix}_LIBRARIES ${${Prefix}_LIBRARY})
set(${Prefix}_INCLUDE_DIRS ${${Prefix}_INCLUDE_DIR})
set(ReqVars "")
@@ -301,22 +274,25 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
"${CompLibSuffixes}" "${CompLib}" "" "")
if (CompRequired)
if (CompLib)
list(APPEND ReqVars ${FullPrefix}_LIBRARY)
list(APPEND ReqVars ${FullPrefix}_LIBRARIES)
endif()
if (CompHeader)
list(APPEND ReqVars ${FullPrefix}_INCLUDE_DIR)
list(APPEND ReqVars ${FullPrefix}_INCLUDE_DIRS)
endif()
endif(CompRequired)
if (((NOT CompLib) OR ${FullPrefix}_LIBRARY) AND
((NOT CompHeader) OR ${FullPrefix}_INCLUDE_DIR))
if (((NOT CompLib) OR ${FullPrefix}_LIBRARIES) AND
((NOT CompHeader) OR ${FullPrefix}_INCLUDE_DIRS))
# Component found
list(APPEND ${Prefix}_LIBRARIES ${${FullPrefix}_LIBRARY})
list(APPEND ${Prefix}_INCLUDE_DIRS ${${FullPrefix}_INCLUDE_DIR})
set(${FullPrefix}_FOUND TRUE CACHE BOOL
"${Name}/${CompPrefix} was found." FORCE)
list(APPEND ${Prefix}_LIBRARIES ${${FullPrefix}_LIBRARIES})
list(APPEND ${Prefix}_INCLUDE_DIRS ${${FullPrefix}_INCLUDE_DIRS})
if (NOT ${Name}_FIND_QUIETLY)
# message(STATUS "${Name}: ${CompPrefix}: found")
message(STATUS
"${Name}: ${CompPrefix}: ${${FullPrefix}_LIBRARY}")
"${Name}: ${CompPrefix}: ${${FullPrefix}_LIBRARIES}")
# message(STATUS
# "${Name}: ${CompPrefix}: ${${FullPrefix}_INCLUDE_DIR}")
# "${Name}: ${CompPrefix}: ${${FullPrefix}_INCLUDE_DIRS}")
endif()
else()
# Let FindPackageHandleStandardArgs() handle errors
@@ -369,8 +345,6 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
if (NOT ${Name}_FIND_QUIETLY)
message(STATUS "${Name}: trying alternative package: ${ReqPackM}")
endif()
# Do not add ${Required} here, since that will prevent other potential
# alternative packages from being found.
find_package(${ReqPack} ${Quiet} COMPONENTS ${PackComps})
string(TOUPPER ${ReqPack} ReqPACK)
if (${ReqPack}_FOUND)
@@ -384,7 +358,7 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
endif()
elseif (Alternative)
set(Alternative FALSE)
elseif (Found)
else()
if (NOT ${Name}_FIND_QUIETLY)
if (Required)
message(STATUS "${Name}: looking for required package: ${ReqPackM}")
@@ -394,139 +368,23 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
endif()
string(TOUPPER ${ReqPack} ReqPACK)
if (NOT (${ReqPack}_FOUND OR ${ReqPACK}_FOUND))
if (NOT ${ReqPack}_TARGET_NAMES)
find_package(${ReqPack} ${Required} ${Quiet} COMPONENTS ${PackComps})
else()
foreach(_target ${ReqPack} ${${ReqPack}_TARGET_NAMES})
# Do not use ${Required} here:
find_package(${_target} NAMES ${_target} ${ReqPack} ${Quiet}
COMPONENTS ${PackComps})
string(TOUPPER ${_target} _TARGET)
if (${_target}_FOUND OR ${_TARGET}_FOUND)
set(${ReqPack}_FOUND TRUE)
break()
endif()
endforeach()
if (${Required} AND NOT ${ReqPack}_FOUND)
message(FATAL_ERROR " *** Required package ${ReqPack} not found."
"Checked target names: ${ReqPack} ${${ReqPack}_TARGET_NAMES}")
endif()
endif()
find_package(${ReqPack} ${Required} ${Quiet} COMPONENTS ${PackComps})
endif()
if (Required AND NOT (${ReqPack}_FOUND OR ${ReqPACK}_FOUND))
message(FATAL_ERROR " --------- INTERNAL ERROR")
endif()
if ("${ReqPack}" STREQUAL "MPI" AND MPI_CXX_FOUND)
if ("${ReqPack}" STREQUAL "MPI")
list(APPEND ${Prefix}_LIBRARIES ${MPI_CXX_LIBRARIES})
list(APPEND ${Prefix}_INCLUDE_DIRS ${MPI_CXX_INCLUDE_PATH})
elseif (${ReqPack}_FOUND OR ${ReqPACK}_FOUND)
else()
if (${ReqPack}_FOUND)
set(_Pack ${ReqPack})
else()
set(_Pack ${ReqPACK})
list(APPEND ${Prefix}_LIBRARIES ${${ReqPack}_LIBRARIES})
list(APPEND ${Prefix}_INCLUDE_DIRS ${${ReqPack}_INCLUDE_DIRS})
elseif (${ReqPACK}_FOUND)
list(APPEND ${Prefix}_LIBRARIES ${${ReqPACK}_LIBRARIES})
list(APPEND ${Prefix}_INCLUDE_DIRS ${${ReqPACK}_INCLUDE_DIRS})
endif()
set(_Pack_LIBS)
set(_Pack_INCS)
# - ${_Pack}_CONFIG is defined by find_package() when a config file was
# loaded
# - If ${ReqPack}_TARGET_NAMES is defined, use target mode
if (NOT ((DEFINED ${_Pack}_CONFIG) OR
(DEFINED ${ReqPack}_TARGET_NAMES)))
# Defined variables expected:
# - ${ReqPack}_LIB_VARS, optional, default: ${_Pack}_LIBRARIES
# - ${ReqPack}_INCLUDE_VARS, optional, default: ${_Pack}_INCLUDE_DIRS
set(_lib_vars ${${ReqPack}_LIB_VARS})
if (NOT _lib_vars)
set(_lib_vars ${_Pack}_LIBRARIES)
endif()
foreach (_var ${_lib_vars})
if (${_var})
list(APPEND _Pack_LIBS ${${_var}})
endif()
endforeach()
# Includes
set(_inc_vars ${${ReqPack}_INCLUDE_VARS})
if (NOT _inc_vars)
set(_inc_vars ${_Pack}_INCLUDE_DIRS)
endif()
foreach (_include ${_inc_vars})
# message(STATUS "${Name}: ${ReqPack}: ${_include}")
if (${_include})
list(APPEND _Pack_INCS ${${_include}})
endif()
endforeach()
else()
# Target mode: check for a valid target:
# - an entry in the variable ${ReqPack}_TARGET_NAMES (optional)
# - ${_Pack}
# Other optional variables:
# - ${ReqPack}_IMPORT_CONFIG, default value: "RELEASE"
# - ${ReqPack}_TARGET_FORCE, default value: "FALSE"
set(TargetName)
foreach (_target ${${ReqPack}_TARGET_NAMES} ${_Pack})
if (TARGET ${_target})
set(TargetName ${_target})
break()
endif()
endforeach()
if ("${TargetName}" STREQUAL "")
message(FATAL_ERROR " *** ${ReqPack}: unknown target. "
"Please set ${ReqPack}_TARGET_NAMES.")
endif()
get_target_property(IsImported ${TargetName} IMPORTED)
if (IsImported)
set(ImportConfig ${${ReqPack}_IMPORT_CONFIG})
if (NOT ImportConfig)
set(ImportConfig RELEASE)
endif()
get_target_property(ImpConfigs ${TargetName} IMPORTED_CONFIGURATIONS)
list(FIND ImpConfigs ${ImportConfig} _Index)
if (_Index EQUAL -1)
message(FATAL_ERROR " *** ${ReqPack}: configuration "
"${ImportConfig} not found. Set ${ReqPack}_IMPORT_CONFIG "
"from the list: ${ImpConfigs}.")
endif()
endif()
# Set _Pack_LIBS
if (NOT IsImported OR ${ReqPack}_TARGET_FORCE)
# Set _Pack_LIBS to be the target itself
set(_Pack_LIBS ${TargetName})
if (NOT ${Name}_FIND_QUIETLY)
message(STATUS "Found ${ReqPack}: ${_Pack_LIBS} (target)")
endif()
else()
# Set _Pack_LIBS from the target properties for ImportConfig
foreach (_prop IMPORTED_LOCATION_${ImportConfig}
IMPORTED_LINK_INTERFACE_LIBRARIES_${ImportConfig})
get_target_property(_value ${TargetName} ${_prop})
if (_value)
list(APPEND _Pack_LIBS ${_value})
endif()
endforeach()
if (NOT ${Name}_FIND_QUIETLY)
message(STATUS
"Imported ${ReqPack}[${ImportConfig}]: ${_Pack_LIBS}")
endif()
endif()
# Set _Pack_INCS
foreach (_prop INCLUDE_DIRECTORIES)
get_target_property(_value ${TargetName} ${_prop})
if (_value)
list(APPEND _Pack_INCS ${_value})
endif()
endforeach()
endif()
# _Pack_LIBS and _Pack_INCS should be fully defined here
list(APPEND ${Prefix}_LIBRARIES ${_Pack_LIBS})
list(APPEND ${Prefix}_INCLUDE_DIRS ${_Pack_INCS})
endif()
endif()
endforeach()
if (Found AND ${Name}_REQUIRED_LIBRARIES)
list(APPEND ${Prefix}_LIBRARIES ${${Name}_REQUIRED_LIBRARIES})
endif()
if (NOT ("${${Prefix}_INCLUDE_DIRS}" STREQUAL ""))
list(INSERT ReqVars 0 ${Prefix}_INCLUDE_DIRS)
set(ReqHeaders 1)
@@ -543,6 +401,12 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
if (ReqHeaders)
list(REMOVE_DUPLICATES ${Prefix}_INCLUDE_DIRS)
endif()
# Write the updated values to the cache.
set(${Prefix}_LIBRARIES ${${Prefix}_LIBRARIES} CACHE STRING
"${LibDoc}" FORCE)
set(${Prefix}_INCLUDE_DIRS ${${Prefix}_INCLUDE_DIRS} CACHE STRING
"${IncDoc}" FORCE)
set(${Prefix}_FOUND TRUE CACHE BOOL "${Name} was found." FORCE)
# Check for optional "CHECK_BUILD" arguments.
set(I 9) # 9 is the number of required arguments
@@ -560,10 +424,6 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
set(CMAKE_REQUIRED_QUIET ${${Name}_FIND_QUIETLY})
check_cxx_source_compiles("${TestSrc}" ${TestVar})
if (TestReq)
if (NOT ${TestVar})
set(Found FALSE)
unset(${TestVar} CACHE)
endif()
list(APPEND ReqVars ${TestVar})
endif()
elseif("${ARGV${I}}" STREQUAL "ADD_COMPONENT")
@@ -574,35 +434,22 @@ function(mfem_find_package Name Prefix DirVar IncSuffixes Header LibSuffixes
endif()
math(EXPR I "${I}+1")
endwhile()
else()
set(${Prefix}_FOUND FALSE CACHE BOOL "${Name} was not found." FORCE)
endif()
if ("_x_${ReqVars}" STREQUAL "_x_")
set(${Prefix}_FOUND ${Found})
set(ReqVars ${Prefix}_FOUND)
endif()
# foreach(ReqVar ${ReqVars})
# message(STATUS " *** ${ReqVar}=${${ReqVar}}")
# get_property(IsCached CACHE ${ReqVar} PROPERTY "VALUE" SET)
# if (IsCached)
# get_property(CachedVal CACHE ${ReqVar} PROPERTY "VALUE")
# message(STATUS " *** ${ReqVar}[cached]=${CachedVal}")
# endif()
# message(STATUS "${ReqVar}=${${ReqVar}}")
# endforeach()
include(FindPackageHandleStandardArgs)
find_package_handle_standard_args(${Name}
" *** ${Name} not found. Please set ${DirVar}." ${ReqVars})
string(TOUPPER ${Name} UName)
if (${UName}_FOUND)
# Write the ${Prefix}_* variables to the cache.
set(${Prefix}_LIBRARIES ${${Prefix}_LIBRARIES} CACHE STRING
"${LibDoc}" FORCE)
set(${Prefix}_INCLUDE_DIRS ${${Prefix}_INCLUDE_DIRS} CACHE STRING
"${IncDoc}" FORCE)
set(${Prefix}_FOUND TRUE CACHE BOOL "${Name} was found." FORCE)
if (ReqHeaders AND (NOT ${Name}_FIND_QUIETLY))
message(STATUS "${Prefix}_INCLUDE_DIRS=${${Prefix}_INCLUDE_DIRS}")
endif()
if (Found AND ReqLibs AND ReqHeaders AND (NOT ${Name}_FIND_QUIETLY))
message(STATUS "${Prefix}_INCLUDE_DIRS=${${Prefix}_INCLUDE_DIRS}")
endif()
endfunction(mfem_find_package)
-11
View File
@@ -30,18 +30,7 @@
#ifdef MFEM_USE_SUPERLU
#error Building with SuperLU_DIST (MFEM_USE_SUPERLU=YES) requires MPI (MFEM_USE_MPI=YES)
#endif
#ifdef MFEM_USE_STRUMPACK
#error Building with STRUMPACK (MFEM_USE_STRUMPACK=YES) requires MPI (MFEM_USE_MPI=YES)
#endif
#ifdef MFEM_USE_PETSC
#error Building with PETSc (MFEM_USE_PETSC=YES) requires MPI (MFEM_USE_MPI=YES)
#endif
#endif // MFEM_USE_MPI not defined
// Macro that returns its first arg when MFEM_USE_BACKENDS is defined, and its
// second arg if it is not defined.
#ifdef MFEM_USE_BACKENDS
#define MFEM_IF_BACKENDS(x,y) (x)
#else
#define MFEM_IF_BACKENDS(x,y) (y)
#endif
+5 -58
View File
@@ -12,33 +12,6 @@
#ifndef MFEM_CONFIG_HEADER
#define MFEM_CONFIG_HEADER
// MFEM version: integer of the form: (major*100 + minor)*100 + patch.
// #define MFEM_VERSION @MFEM_VERSION@
// MFEM version string of the form "3.3" or "3.3.1".
// #define MFEM_VERSION_STRING "@MFEM_VERSION_STRING@"
// MFEM version type, see the MFEM_VERSION_TYPE_* constants below.
#define MFEM_VERSION_TYPE ((MFEM_VERSION)%2)
// MFEM version type constants.
#define MFEM_VERSION_TYPE_RELEASE 0
#define MFEM_VERSION_TYPE_DEVELOPMENT 1
// Separate MFEM version numbers for major, minor, and patch.
#define MFEM_VERSION_MAJOR ((MFEM_VERSION)/10000)
#define MFEM_VERSION_MINOR (((MFEM_VERSION)/100)%100)
#define MFEM_VERSION_PATCH ((MFEM_VERSION)%100)
// Description of the git commit used to build MFEM.
// #define MFEM_GIT_STRING "@MFEM_GIT_STRING@"
// The absolute path of the MFEM source prefix
// #define MFEM_SOURCE_DIR "@MFEM_SOURCE_DIR@"
// The absolute path of the MFEM installation prefix
// #define MFEM_INSTALL_DIR "@MFEM_INSTALL_DIR@"
// Build the parallel MFEM library.
// Requires an MPI compiler, and the libraries HYPRE and METIS.
// #define MFEM_USE_MPI
@@ -46,18 +19,12 @@
// Enable debug checks in MFEM.
// #define MFEM_DEBUG
// Throw an exception on errors.
// #define MFEM_USE_EXCEPTIONS
// Enable gzstream in MFEM.
// #define MFEM_USE_GZSTREAM
// Enable backtraces for mfem_error through libunwind.
// #define MFEM_USE_LIBUNWIND
// Enable MFEM features that use the METIS library (parallel MFEM).
// #define MFEM_USE_METIS
// Enable this option if linking with METIS version 5 (parallel MFEM).
// #define MFEM_USE_METIS_5
@@ -75,7 +42,11 @@
// #define MFEM_USE_MEMALLOC
// Which library functions to use in class StopWatch for measuring time.
// For a list of the available options, see INSTALL.
// The available options are:
// 0 - use std::clock from <ctime>
// 1 - use times from <sys/times.h>
// 2 - use high-resolution POSIX clocks
// 3 - use QueryPerformanceCounter from <windows.h>
// If not defined, an option is selected automatically.
// #define MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@
@@ -91,9 +62,6 @@
// Enable MFEM functionality based on the SuperLU library.
// #define MFEM_USE_SUPERLU
// Enable MFEM functionality based on the STRUMPACK library.
// #define MFEM_USE_STRUMPACK
// Enable functionality based on the Gecko library
// #define MFEM_USE_GECKO
@@ -103,9 +71,6 @@
// Enable Sidre support
// #define MFEM_USE_SIDRE
// Enable Conduit support
// #define MFEM_USE_CONDUIT
// Enable functionality based on the NetCDF library (reading CUBIT files)
// #define MFEM_USE_NETCDF
@@ -115,28 +80,10 @@
// Enable functionality based on the MPFR library.
// #define MFEM_USE_MPFR
// Enable the use of MFEM backends.
// #define MFEM_USE_BACKENDS
// Enable the OCCA backend.
// #define MFEM_USE_OCCA
// Enable the OMP backend.
// #define MFEM_USE_OMP
// Enable use of acrotensor in backends.
// #define MFEM_USE_ACROTENSOR
// Enable use of unified memory.
// #define MFEM_USE_CUDAUM
// Windows specific options
#ifdef _WIN32
// Macro needed to get defines like M_PI from <cmath>. (Visual Studio C++ only?)
#define _USE_MATH_DEFINES
#endif
// Version of HYPRE used for building MFEM.
// #define MFEM_HYPRE_VERSION @MFEM_HYPRE_VERSION@
#endif // MFEM_CONFIG_HEADER
-22
View File
@@ -10,16 +10,9 @@
# Software Foundation) version 2.1 dated February 1999.
# Variables corresponding to defines in config.hpp (YES, NO, or value)
MFEM_VERSION = @MFEM_VERSION@
MFEM_VERSION_STRING = @MFEM_VERSION_STRING@
MFEM_GIT_STRING = @MFEM_GIT_STRING@
MFEM_SOURCE_DIR = @MFEM_SOURCE_DIR@
MFEM_INSTALL_DIR = @MFEM_INSTALL_DIR@
MFEM_USE_MPI = @MFEM_USE_MPI@
MFEM_USE_METIS = @MFEM_USE_METIS@
MFEM_USE_METIS_5 = @MFEM_USE_METIS_5@
MFEM_DEBUG = @MFEM_DEBUG@
MFEM_USE_EXCEPTIONS = @MFEM_USE_EXCEPTIONS@
MFEM_USE_GZSTREAM = @MFEM_USE_GZSTREAM@
MFEM_USE_LIBUNWIND = @MFEM_USE_LIBUNWIND@
MFEM_USE_LAPACK = @MFEM_USE_LAPACK@
@@ -31,19 +24,12 @@ MFEM_USE_SUNDIALS = @MFEM_USE_SUNDIALS@
MFEM_USE_MESQUITE = @MFEM_USE_MESQUITE@
MFEM_USE_SUITESPARSE = @MFEM_USE_SUITESPARSE@
MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
MFEM_USE_GECKO = @MFEM_USE_GECKO@
MFEM_USE_GNUTLS = @MFEM_USE_GNUTLS@
MFEM_USE_NETCDF = @MFEM_USE_NETCDF@
MFEM_USE_PETSC = @MFEM_USE_PETSC@
MFEM_USE_MPFR = @MFEM_USE_MPFR@
MFEM_USE_SIDRE = @MFEM_USE_SIDRE@
MFEM_USE_CONDUIT = @MFEM_USE_CONDUIT@
MFEM_USE_BACKENDS = @MFEM_USE_BACKENDS@
MFEM_USE_OCCA = @MFEM_USE_OCCA@
MFEM_USE_OMP = @MFEM_USE_OMP@
MFEM_USE_ACROTENSOR = @MFEM_USE_ACROTENSOR@
MFEM_USE_CUDAUM = @MFEM_USE_CUDAUM@
# Compiler, compile options, and link options
MFEM_CXX = @MFEM_CXX@
@@ -51,25 +37,17 @@ MFEM_CPPFLAGS = @MFEM_CPPFLAGS@
MFEM_CXXFLAGS = @MFEM_CXXFLAGS@
MFEM_TPLFLAGS = @MFEM_TPLFLAGS@
MFEM_INCFLAGS = @MFEM_INCFLAGS@
MFEM_PICFLAG = @MFEM_PICFLAG@
MFEM_FLAGS = @MFEM_FLAGS@
MFEM_EXT_LIBS = @MFEM_EXT_LIBS@
MFEM_LIBS = @MFEM_LIBS@
MFEM_LIB_FILE = @MFEM_LIB_FILE@
MFEM_STATIC = @MFEM_STATIC@
MFEM_SHARED = @MFEM_SHARED@
MFEM_BUILD_TAG = @MFEM_BUILD_TAG@
MFEM_PREFIX = @MFEM_PREFIX@
MFEM_INC_DIR = @MFEM_INC_DIR@
MFEM_LIB_DIR = @MFEM_LIB_DIR@
# Location of test.mk
MFEM_TEST_MK = @MFEM_TEST_MK@
# Command used to launch MPI jobs
MFEM_MPIEXEC = @MFEM_MPIEXEC@
MFEM_MPIEXEC_NP = @MFEM_MPIEXEC_NP@
MFEM_MPI_NP = @MFEM_MPI_NP@
# Optional extra configuration
@MFEM_CONFIG_EXTRA@
+11 -40
View File
@@ -18,10 +18,8 @@ if (NOT CMAKE_BUILD_TYPE)
"Build type: Debug, Release, RelWithDebInfo, or MinSizeRel." FORCE)
endif()
# MFEM options. Set to mimic the default "defaults.mk" file.
# MFEM options. Set to mimic the default "default.mk" file.
option(MFEM_USE_MPI "Enable MPI parallel build" OFF)
option(MFEM_USE_METIS "Enable METIS usage" ${MFEM_USE_MPI})
option(MFEM_USE_EXCEPTIONS "Enable the use of exceptions" OFF)
option(MFEM_USE_GZSTREAM "Enable gzstream for compressed data streams." OFF)
option(MFEM_USE_LIBUNWIND "Enable backtrace for errors." OFF)
option(MFEM_USE_LAPACK "Enable LAPACK usage" OFF)
@@ -32,20 +30,17 @@ option(MFEM_USE_SUNDIALS "Enable SUNDIALS usage" OFF)
option(MFEM_USE_MESQUITE "Enable MESQUITE usage" OFF)
option(MFEM_USE_SUITESPARSE "Enable SuiteSparse usage" OFF)
option(MFEM_USE_SUPERLU "Enable SuperLU_DIST usage" OFF)
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
option(MFEM_USE_GECKO "Enable GECKO usage" OFF)
option(MFEM_USE_GNUTLS "Enable GNUTLS usage" OFF)
option(MFEM_USE_NETCDF "Enable NETCDF usage" OFF)
option(MFEM_USE_PETSC "Enable PETSc support." OFF)
option(MFEM_USE_MPFR "Enable MPFR usage." OFF)
option(MFEM_USE_SIDRE "Enable Axom/Sidre usage" OFF)
option(MFEM_USE_CONDUIT "Enable Conduit usage" OFF)
option(MFEM_USE_SIDRE "Enable ATK/Sidre usage" OFF)
# Allow a user to disable testing, examples, and/or miniapps at CONFIGURE TIME
# if they don't want/need them (e.g. if MFEM is "just a dependency" and all they
# need is the library, building all that stuff adds unnecessary overhead). Note
# that the examples or miniapps can always be built using the targets 'examples'
# or 'miniapps', respectively.
# need is the library, building all that stuff adds unnecessary overhead). To
# match "makefile" behavior, they are all enabled by default.
option(MFEM_ENABLE_TESTING "Enable the ctest framework for testing" ON)
option(MFEM_ENABLE_EXAMPLES "Build all of the examples" OFF)
option(MFEM_ENABLE_MINIAPPS "Build all of the miniapps" OFF)
@@ -71,7 +66,7 @@ set(METIS_DIR "${MFEM_DIR}/../metis-4.0" CACHE PATH "Path to the METIS library."
set(LIBUNWIND_DIR "" CACHE PATH "Path to Libunwind.")
set(SUNDIALS_DIR "${MFEM_DIR}/../sundials-3.0.0" CACHE PATH
set(SUNDIALS_DIR "${MFEM_DIR}/../sundials-2.7.0" CACHE PATH
"Path to the SUNDIALS library.")
# The following may be necessary, if SUNDIALS was built with KLU:
# set(SUNDIALS_REQUIRED_PACKAGES "SuiteSparse/KLU/AMD/BTF/COLAMD/config"
@@ -96,32 +91,6 @@ set(SuperLUDist_DIR "${MFEM_DIR}/../SuperLU_DIST_5.1.0" CACHE PATH
set(SuperLUDist_REQUIRED_PACKAGES "MPI" "BLAS" "ParMETIS" CACHE STRING
"Additional packages required by SuperLU_DIST.")
set(STRUMPACK_DIR "${MFEM_DIR}/../STRUMPACK-build" CACHE PATH
"Path to the STRUMPACK library.")
# STRUMPACK may also depend on "OpenMP", depending on how it was compiled.
set(STRUMPACK_REQUIRED_PACKAGES "MPI" "MPI_Fortran" "ParMETIS" "METIS"
"ScaLAPACK" "Scotch/ptscotch/ptscotcherr/scotch/scotcherr" CACHE STRING
"Additional packages required by STRUMPACK.")
# If the MPI package does not find all required Fortran libraries:
# set(STRUMPACK_REQUIRED_LIBRARIES "gfortran" "mpi_mpifh" CACHE STRING
# "Additional libraries required by STRUMPACK.")
# The Scotch library, required by STRUMPACK
set(Scotch_DIR "${MFEM_DIR}/../scotch_6.0.4" CACHE PATH
"Path to the Scotch and PT-Scotch libraries.")
set(Scotch_REQUIRED_PACKAGES "Threads" CACHE STRING
"Additional packages required by Scotch.")
# Tell the "Threads" package/module to prefer pthreads.
set(CMAKE_THREAD_PREFER_PTHREAD TRUE)
set(Threads_LIB_VARS CMAKE_THREAD_LIBS_INIT)
# The ScaLAPACK library, required by STRUMPACK
set(ScaLAPACK_DIR "${MFEM_DIR}/../scalapack-2.0.2/lib/cmake/scalapack-2.0.2"
CACHE PATH "Path to the configuration file scalapack-config.cmake")
set(ScaLAPACK_TARGET_NAMES scalapack)
# set(ScaLAPACK_TARGET_FORCE)
# set(ScaLAPACK_IMPORT_CONFIG DEBUG)
set(GECKO_DIR "${MFEM_DIR}/../gecko" CACHE PATH "Path to the Gecko library.")
set(GNUTLS_DIR "" CACHE PATH "Path to the GnuTLS library.")
@@ -133,17 +102,19 @@ set(NetCDF_REQUIRED_PACKAGES "" CACHE STRING
set(PETSC_DIR "${MFEM_DIR}/../petsc" CACHE PATH
"Path to the PETSc main directory.")
set(PETSC_ARCH "arch-linux2-c-debug" CACHE STRING "PETSc build architecture.")
set(PETSC_ARCH "arch-linux2-c-debug" CACHE PATH "PETSc build architecture.")
set(MPFR_DIR "" CACHE PATH "Path to the MPFR library.")
set(CONDUIT_DIR "${MFEM_DIR}/../conduit" CACHE PATH
"Path to the Conduit library.")
set(Conduit_REQUIRED_PACKAGES "HDF5" CACHE STRING
"Additional packages required by Conduit.")
set(AXOM_DIR "${MFEM_DIR}/../axom" CACHE PATH "Path to the Axom library.")
set(ATK_DIR "${MFEM_DIR}/../asctoolkit" CACHE PATH "Path to the ATK library.")
# May need to add "Boost" as requirement.
set(Axom_REQUIRED_PACKAGES "Conduit/relay" CACHE STRING
"Additional packages required by Axom.")
set(ATK_REQUIRED_PACKAGES "Conduit/relay" CACHE STRING
"Additional packages required by ATK.")
set(BLAS_INCLUDE_DIRS "" CACHE STRING "Path to BLAS headers.")
set(BLAS_LIBRARIES "" CACHE STRING "The BLAS library.")
+31 -137
View File
@@ -30,36 +30,15 @@ PREFIX = ./mfem
# Install program
INSTALL = /usr/bin/install
STATIC = YES
SHARED = NO
ifneq ($(NOTMAC),)
AR = ar
ARFLAGS = cruv
RANLIB = ranlib
PICFLAG = -fPIC
SO_EXT = so
SO_VER = so.$(MFEM_VERSION_STRING)
BUILD_SOFLAGS = -shared -Wl,-soname,libmfem.$(SO_VER)
BUILD_RPATH = -Wl,-rpath,$(BUILD_REAL_DIR)
INSTALL_SOFLAGS = $(BUILD_SOFLAGS)
INSTALL_RPATH = -Wl,-rpath,@MFEM_LIB_DIR@
else
# Silence "has no symbols" warnings on Mac OS X
AR = ar
ARFLAGS = Scruv
RANLIB = ranlib -no_warning_for_no_symbols
PICFLAG = -fPIC
SO_EXT = dylib
SO_VER = $(MFEM_VERSION_STRING).dylib
MAKE_SOFLAGS = -Wl,-dylib,-install_name,$(1)/libmfem.$(SO_VER),\
-compatibility_version,$(MFEM_VERSION_STRING),\
-current_version,$(MFEM_VERSION_STRING),\
-undefined,dynamic_lookup
BUILD_SOFLAGS = $(subst $1 ,,$(call MAKE_SOFLAGS,$(BUILD_REAL_DIR)))
BUILD_RPATH = -Wl,-undefined,dynamic_lookup
INSTALL_SOFLAGS = $(subst $1 ,,$(call MAKE_SOFLAGS,$(MFEM_LIB_DIR)))
INSTALL_RPATH = -Wl,-undefined,dynamic_lookup
endif
# Set CXXFLAGS to overwrite the default selection of DEBUG_FLAGS/OPTIM_FLAGS
@@ -75,50 +54,31 @@ endif
# Command used to launch MPI jobs
MFEM_MPIEXEC = mpirun
MFEM_MPIEXEC_NP = -np
# Number of mpi tasks for parallel jobs
MFEM_MPI_NP = 4
# MFEM configuration options: YES/NO values, which are exported to config.mk and
# config.hpp. The values below are the defaults for generating the actual values
# in config.mk and config.hpp.
MFEM_USE_MPI = NO
# FIXME: add MFEM_USE_BACKENDS, MFEM_USE_OCCA to the CMake build system
MFEM_USE_BACKENDS = YES
MFEM_USE_OCCA = YES
MFEM_USE_METIS = $(MFEM_USE_MPI)
MFEM_USE_METIS_5 = NO
MFEM_DEBUG = NO
MFEM_USE_EXCEPTIONS = NO
MFEM_USE_GZSTREAM = NO
MFEM_USE_LIBUNWIND = NO
MFEM_USE_LAPACK = NO
MFEM_THREAD_SAFE = NO
MFEM_USE_OPENMP = NO
MFEM_USE_MEMALLOC = YES
MFEM_TIMER_TYPE = $(if $(NOTMAC),2,4)
MFEM_TIMER_TYPE = $(if $(NOTMAC),2,0)
MFEM_USE_SUNDIALS = NO
MFEM_USE_MESQUITE = NO
MFEM_USE_SUITESPARSE = NO
MFEM_USE_SUPERLU = NO
MFEM_USE_STRUMPACK = NO
MFEM_USE_GECKO = NO
MFEM_USE_GNUTLS = NO
MFEM_USE_NETCDF = NO
MFEM_USE_PETSC = NO
MFEM_USE_MPFR = NO
MFEM_USE_SIDRE = NO
MFEM_USE_CONDUIT = NO
# FIXME: add MFEM_USE_OMP and MFEM_USE_ACROTENSOR to the CMake build system
MFEM_USE_OMP = NO
MFEM_USE_ACROTENSOR = NO
MFEM_USE_CUDAUM = NO
# Compile and link options for zlib.
ZLIB_DIR =
ZLIB_OPT = $(if $(ZLIB_DIR),-I$(ZLIB_DIR)/include)
ZLIB_LIB = $(if $(ZLIB_DIR),$(ZLIB_RPATH) -L$(ZLIB_DIR)/lib ,)-lz
ZLIB_RPATH = -Wl,-rpath,$(ZLIB_DIR)/lib
LIBUNWIND_OPT = -g
LIBUNWIND_LIB = $(if $(NOTMAC),-lunwind -ldl,)
@@ -129,7 +89,7 @@ HYPRE_OPT = -I$(HYPRE_DIR)/include
HYPRE_LIB = -L$(HYPRE_DIR)/lib -lHYPRE
# METIS library configuration
ifeq ($(MFEM_USE_SUPERLU)$(MFEM_USE_STRUMPACK),NONO)
ifeq ($(MFEM_USE_SUPERLU),NO)
ifeq ($(MFEM_USE_METIS_5),NO)
METIS_DIR = @MFEM_DIR@/../metis-4.0
METIS_OPT =
@@ -140,7 +100,7 @@ ifeq ($(MFEM_USE_SUPERLU)$(MFEM_USE_STRUMPACK),NONO)
METIS_LIB = -L$(METIS_DIR)/lib -lmetis
endif
else
# ParMETIS: currently needed by SuperLU or STRUMPACK. We assume that METIS 5
# ParMETIS currently needed only with SuperLU. We assume that METIS 5
# (included with ParMETIS) is installed in the same location.
METIS_DIR = @MFEM_DIR@/../parmetis-4.0.3
METIS_OPT = -I$(METIS_DIR)/include
@@ -160,10 +120,10 @@ OPENMP_LIB =
POSIX_CLOCKS_LIB = -lrt
# SUNDIALS library configuration
SUNDIALS_DIR = @MFEM_DIR@/../sundials-3.0.0
SUNDIALS_DIR = @MFEM_DIR@/../sundials-2.7.0
SUNDIALS_OPT = -I$(SUNDIALS_DIR)/include
SUNDIALS_LIB = -Wl,-rpath,$(SUNDIALS_DIR)/lib -L$(SUNDIALS_DIR)/lib\
-lsundials_arkode -lsundials_cvode -lsundials_nvecserial -lsundials_kinsol
-lsundials_arkode -lsundials_cvode -lsundials_nvecserial -lsundials_kinsol
ifeq ($(MFEM_USE_MPI),YES)
SUNDIALS_LIB += -lsundials_nvecparhyp -lsundials_nvecparallel
@@ -180,41 +140,14 @@ MESQUITE_LIB = -L$(MESQUITE_DIR)/lib -lmesquite
LIB_RT = $(if $(NOTMAC),-lrt,)
SUITESPARSE_DIR = @MFEM_DIR@/../SuiteSparse
SUITESPARSE_OPT = -I$(SUITESPARSE_DIR)/include
SUITESPARSE_LIB = -Wl,-rpath,$(SUITESPARSE_DIR)/lib -L$(SUITESPARSE_DIR)/lib\
-lklu -lbtf -lumfpack -lcholmod -lcolamd -lamd -lcamd -lccolamd\
-lsuitesparseconfig $(LIB_RT) $(METIS_LIB) $(LAPACK_LIB)
SUITESPARSE_LIB = -L$(SUITESPARSE_DIR)/lib -lklu -lbtf -lumfpack -lcholmod\
-lcolamd -lamd -lcamd -lccolamd -lsuitesparseconfig $(LIB_RT) $(METIS_LIB)\
$(LAPACK_LIB)
# SuperLU library configuration
SUPERLU_DIR = @MFEM_DIR@/../SuperLU_DIST_5.1.0
SUPERLU_OPT = -I$(SUPERLU_DIR)/SRC
SUPERLU_LIB = -Wl,-rpath,$(SUPERLU_DIR)/SRC -L$(SUPERLU_DIR)/SRC -lsuperlu_dist
# SCOTCH library configuration (required by STRUMPACK)
SCOTCH_DIR = @MFEM_DIR@/../scotch_6.0.4
SCOTCH_OPT = -I$(SCOTCH_DIR)/include
SCOTCH_LIB = -L$(SCOTCH_DIR)/lib -lptscotch -lptscotcherr -lscotch -lscotcherr\
-lpthread
# SCALAPACK library configuration (required by STRUMPACK)
SCALAPACK_DIR = @MFEM_DIR@/../scalapack-2.0.2
SCALAPACK_OPT = -I$(SCALAPACK_DIR)/SRC
SCALAPACK_LIB = -L$(SCALAPACK_DIR)/lib -lscalapack $(LAPACK_LIB)
# MPI Fortran library, needed e.g. by STRUMPACK
# MPICH:
MPI_FORTRAN_LIB = -lmpifort
# OpenMPI:
# MPI_FORTRAN_LIB = -lmpi_mpifh
# Additional Fortan library:
# MPI_FORTRAN_LIB += -lgfortran
# STRUMPACK library configuration
STRUMPACK_DIR = @MFEM_DIR@/../STRUMPACK-build
STRUMPACK_OPT = -I$(STRUMPACK_DIR)/include $(SCOTCH_OPT)
# If STRUMPACK was build with OpenMP support, the following may be need:
# STRUMPACK_OPT += $(OPENMP_OPT)
STRUMPACK_LIB = -L$(STRUMPACK_DIR)/lib -lstrumpack $(MPI_FORTRAN_LIB)\
$(SCOTCH_LIB) $(SCALAPACK_LIB)
SUPERLU_LIB = -L$(SUPERLU_DIR)/SRC -lsuperlu_dist
# Gecko library configuration
GECKO_DIR = @MFEM_DIR@/../gecko
@@ -226,81 +159,42 @@ GNUTLS_OPT =
GNUTLS_LIB = -lgnutls
# NetCDF library configuration
NETCDF_DIR = $(HOME)/local
HDF5_DIR = $(HOME)/local
NETCDF_OPT = -I$(NETCDF_DIR)/include -I$(HDF5_DIR)/include $(ZLIB_OPT)
NETCDF_LIB = -Wl,-rpath,$(NETCDF_DIR)/lib -L$(NETCDF_DIR)/lib\
-Wl,-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib\
-lnetcdf -lhdf5_hl -lhdf5 $(ZLIB_LIB)
NETCDF_DIR = $(HOME)/local
HDF5_DIR = $(HOME)/local
ZLIB_DIR = $(HOME)/local
NETCDF_OPT = -I$(NETCDF_DIR)/include
NETCDF_LIB = -L$(NETCDF_DIR)/lib -lnetcdf -L$(HDF5_DIR)/lib -lhdf5_hl -lhdf5\
-L$(ZLIB_DIR)/lib -lz
# PETSc library configuration (version greater or equal to 3.8 or the dev branch)
PETSC_ARCH := arch-linux2-c-debug
PETSC_DIR := $(MFEM_DIR)/../petsc/$(PETSC_ARCH)
PETSC_VARS := $(PETSC_DIR)/lib/petsc/conf/petscvariables
PETSC_FOUND := $(if $(wildcard $(PETSC_VARS)),YES,)
PETSC_INC_VAR = PETSC_CC_INCLUDES
PETSC_LIB_VAR = PETSC_EXTERNAL_LIB_BASIC
ifeq ($(PETSC_FOUND),YES)
PETSC_OPT := $(shell sed -n "s/$(PETSC_INC_VAR) = *//p" $(PETSC_VARS))
PETSC_LIB := $(shell sed -n "s/$(PETSC_LIB_VAR) = *//p" $(PETSC_VARS))
PETSC_LIB := -Wl,-rpath,$(abspath $(PETSC_DIR))/lib\
-L$(abspath $(PETSC_DIR))/lib -lpetsc $(PETSC_LIB)
ifeq ($(MFEM_USE_PETSC),YES)
PETSC_DIR := $(MFEM_DIR)/../petsc/arch-linux2-c-debug
PETSC_PC := $(PETSC_DIR)/lib/pkgconfig/PETSc.pc
$(if $(wildcard $(PETSC_PC)),,$(error PETSc config not found - $(PETSC_PC)))
PETSC_OPT := $(shell sed -n "s/Cflags: *//p" $(PETSC_PC))
PETSC_LIB := $(shell sed -n "s/Libs.*: *//p" $(PETSC_PC))
PETSC_LIB := -Wl,-rpath -Wl,$(abspath $(PETSC_DIR))/lib $(PETSC_LIB)
endif
# MPFR library configuration
MPFR_OPT =
MPFR_LIB = -lmpfr
# Conduit and required libraries configuration
CONDUIT_DIR = @MFEM_DIR@/../conduit
CONDUIT_OPT = -I$(CONDUIT_DIR)/include/conduit
CONDUIT_LIB = \
-Wl,-rpath,$(CONDUIT_DIR)/lib -L$(CONDUIT_DIR)/lib \
-lconduit -lconduit_relay -lconduit_blueprint -ldl
# Check if Conduit was built with hdf5 support, by looking
# for the relay hdf5 header
CONDUIT_HDF5_HEADER=$(CONDUIT_DIR)/include/conduit/conduit_relay_hdf5.hpp
ifneq (,$(wildcard $(CONDUIT_HDF5_HEADER)))
CONDUIT_OPT += -I$(HDF5_DIR)/include
CONDUIT_LIB += -Wl,-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib \
-lhdf5 $(ZLIB_LIB)
endif
# Sidre and required libraries configuration
# Be sure to check the HDF5_DIR (set above) is correct
SIDRE_DIR = @MFEM_DIR@/../axom
SIDRE_DIR = @MFEM_DIR@/../asctoolkit
CONDUIT_DIR = @MFEM_DIR@/../conduit
SIDRE_OPT = -I$(SIDRE_DIR)/include -I$(CONDUIT_DIR)/include/conduit\
-I$(HDF5_DIR)/include
SIDRE_LIB = \
-Wl,-rpath,$(SIDRE_DIR)/lib -L$(SIDRE_DIR)/lib \
-Wl,-rpath,$(CONDUIT_DIR)/lib -L$(CONDUIT_DIR)/lib \
-Wl,-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib \
-lsidre -lslic -laxom_utils -lconduit -lconduit_relay -lhdf5 $(ZLIB_LIB) -ldl
SIDRE_LIB = -L$(SIDRE_DIR)/lib \
-L$(CONDUIT_DIR)/lib \
-Wl,-rpath -Wl,$(CONDUIT_DIR)/lib \
-L$(HDF5_DIR)/lib\
-Wl,-rpath -Wl,$(HDF5_DIR)/lib \
-lsidre -lslic -lcommon -lconduit -lconduit_relay -lhdf5 -lz -ldl
OCCA_DIR = @MFEM_DIR@/../occa
OCCA_OPT = -I$(OCCA_DIR)/include
OCCA_LIB = -Wl,-rpath,$(OCCA_DIR)/lib -L$(OCCA_DIR)/lib -locca
CUDA_DIR = /usr/local/cuda
CUDAUM_LIB = -L$(CUDA_DIR)/lib64 -lcudart
CUDAUM_OPT = -I$(CUDA_DIR)/include
OMP_OPT = -qsmp=omp -qoffload
ACROTENSOR_DIR = @MFEM_DIR@/../acrotensor
ACROTENSOR_OPT = -std=c++11 -I$(ACROTENSOR_DIR)/inc
ACROTENSOR_LIB = -Wl,-rpath,$(ACROTENSOR_DIR)/lib/shared -L$(ACROTENSOR_DIR)/lib/shared -lacrotensor
# If Acrotensor was compile with CUDA support, but MFEM_USE_CUDAUM==NO, then uncomment the lines below
# ACROTENSOR_OPT += -I$(CUDA_DIR)/include
# ACROTENSOR_LIB += -L$(CUDA_DIR)/lib64 -lcuda -lcudart -lnvrtc
ifeq ($(MFEM_USE_CUDAUM),YES)
ifeq ($(MFEM_USE_MPI),YES)
# HYPRE needs some extra libraries in parallel on the GPU
# FIXME: We need another solution for compilers other than XL for the
# dlink CUDA step, but fixes need to happen elsewhere as well.
HYPRE_LIB += -qcuda -lcublas -lcusparse -lnvToolsExt
endif
SIDRE_LIB += -lspio -lcommon
endif
# If YES, enable some informational messages
-48
View File
@@ -1,48 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "HYPRE_config.h"
#include <cstdio>
#ifdef HYPRE_RELEASE_VERSION
#define HYPRE_VERSION_STRING HYPRE_RELEASE_VERSION
#elif defined(HYPRE_PACKAGE_VERSION)
#define HYPRE_VERSION_STRING HYPRE_PACKAGE_VERSION
#endif
// Macros to expand a macro as a string
#define STR_EXPAND(s) #s
#define STR(s) STR_EXPAND(s)
// Convert the HYPRE_RELEASE_VERSION macro (string) to integer.
// Examples: "2.10.0b" --> 21000, "2.11.2" --> 21102
int main()
{
#ifdef HYPRE_VERSION_STRING
const char *ptr = STR(HYPRE_VERSION_STRING);
if (*ptr == '"') { ptr++; }
int version = 0;
for (int i = 0; i < 3; i++, ptr++)
{
int pv = 0;
for (char d; d = *ptr, '0' <= d && d <= '9'; ptr++)
{
pv = 10*pv + (d - '0');
if (pv >= 100) { return 1; }
}
version = 100*version + pv;
}
printf("%i\n", version);
return 0;
#else
return 2;
#endif
}

Some files were not shown because too many files have changed in this diff Show More