Compare commits

..
217 changed files with 1358 additions and 72651 deletions
+3 -40
View File
@@ -9,7 +9,6 @@
# Object and library files
*.o
/libmfem.*
/miniapps/common/libmfem-common.*
# CMake generated files
CMakeCache.txt
@@ -54,7 +53,6 @@ examples/displaced.mesh
examples/mesh.*
examples/ex5.mesh
examples/Example5*
examples/PVExample*
examples/Example9*
examples/Example15*
examples/Example16*
@@ -62,8 +60,6 @@ examples/sphere_refined.*
examples/sol.*
examples/sol_u.*
examples/sol_p.*
examples/sol_r.*
examples/sol_i.*
examples/ex9.mesh
examples/ex9-mesh.*
examples/ex9-init.*
@@ -117,7 +113,6 @@ examples/petsc/sol.*
examples/petsc/sol_p.*
examples/petsc/sol_u.*
examples/petsc/Example5*
examples/petsc/ex9.mesh
examples/petsc/ex9-mesh.*
examples/petsc/ex9-init.*
examples/petsc/ex9-final.*
@@ -129,11 +124,6 @@ examples/petsc/elastic_energy.*
examples/pumi/ex1
examples/pumi/ex[126]p
examples/hiop/ex9.mesh
examples/hiop/ex9-mesh.*
examples/hiop/ex9-init.*
examples/hiop/ex9-final.*
examples/pumi/refined.mesh
examples/pumi/sol.gf
examples/pumi/mesh.*
@@ -180,43 +170,16 @@ miniapps/tools/display-basis
miniapps/tools/load-dc
miniapps/tools/convert-dc
miniapps/tools/lor-transfer
miniapps/tools/get-values
miniapps/toys/automata
miniapps/toys/life
miniapps/toys/mandel
miniapps/toys/rubik
miniapps/toys/snake
miniapps/toys/lissajous
miniapps/toys/mondrian
miniapps/toys/snake-init.mesh
miniapps/toys/snake-user.mesh
miniapps/toys/snake-joined.mesh
miniapps/toys/snake-c*.mesh
miniapps/toys/automata.gf
miniapps/toys/automata.mesh
miniapps/toys/rubik-init.mesh
miniapps/toys/mandel.mesh
miniapps/toys/life.gf
miniapps/toys/life.mesh
miniapps/toys/lissajous.mesh
miniapps/toys/lissajous.gf
miniapps/toys/mondrian.mesh
miniapps/nurbs/nurbs_ex1
miniapps/nurbs/nurbs_ex1p
miniapps/nurbs/nurbs_ex11p
miniapps/nurbs/ex1
miniapps/nurbs/ex1p
miniapps/nurbs/ex11p
miniapps/nurbs/refined.mesh
miniapps/nurbs/mesh.*
miniapps/nurbs/sol.*
miniapps/nurbs/mode_*
miniapps/nurbs/Example1*
miniapps/gslib/field-diff
miniapps/gslib/findpts
miniapps/gslib/pfindpts
# Unit test binary and outputs
tests/unit/output_meshes
tests/unit/unit_tests
+15 -112
View File
@@ -1,114 +1,28 @@
language: cpp
sudo: false
stages:
- checks
- tests
- optional
language: cpp
jobs:
matrix:
include:
# ========================
# Checks
# ========================
# - code-style
# - documentation
# - gitignore
- stage: checks
os: linux
name: "code-style"
addons:
apt:
packages:
- astyle=2.05.1-0ubuntu1
script:
- cd ${TRAVIS_BUILD_DIR}
- cd tests/scripts
- ./runtest code-style
- stage: checks
os: linux
name: "documentation"
addons:
apt:
packages:
- doxygen
- graphviz
- mpich
- libmpich-dev
env: MPI=YES
script:
- cd ${TRAVIS_BUILD_DIR}
- cd tests/scripts
- ./runtest documentation
- stage: checks
os: linux
name: "gitignore"
addons:
apt:
packages:
- mpich
- libmpich-dev
env: MPI=YES
script:
- cd ${TRAVIS_BUILD_DIR}
- make config MFEM_USE_MPI=YES MFEM_MPI_NP=2
- make all -j3
- make test-noclean
- cd tests/scripts
- ./runtest gitignore
# ========================
# Optional Checks/Tests
# ========================
# - branch-history
- stage: optional
name: "branch-history"
# need full git history for the binary/big files check
git:
depth: false
script:
- cd ${TRAVIS_BUILD_DIR}
# update master
- git fetch origin master:master
# checkout a branch (otherwise Travis works in detached head)
- git checkout -b travis_tests
- cd tests/scripts
- ./runtest branch-history
# ========================
# Linux tests
# ========================
# - serial + debug
# - serial
# - parallel + debug
# - parallel
- stage: tests
os: linux
#
# Linux
#
- os: linux
compiler: gcc
name: "Linux: Serial + Debug"
env: DEBUG=YES
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=check
#
- os: linux
compiler: gcc
name: "Linux: Serial"
env: DEBUG=NO
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=test
#
- os: linux
compiler: gcc
name: "Linux: Parallel + Debug"
addons:
apt:
# sources:
@@ -135,10 +49,9 @@ jobs:
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
#
- os: linux
compiler: gcc
name: "Linux: Parallel"
addons:
apt:
# sources:
@@ -165,37 +78,28 @@ jobs:
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
# ========================
# Mac OS X tests
# ========================
# - serial + debug
# - serial
# - parallel + debug
# - parallel
#
# Mac OS X
#
- os: osx
# osx_image: xcode7.3
compiler: clang
name: "Mac: Serial + Debug"
env: DEBUG=YES
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=check
#
- os: osx
# osx_image: xcode7.3
compiler: clang
name: "Mac: Serial"
env: DEBUG=NO
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=test
#
- os: osx
# osx_image: xcode7.3
compiler: clang
name: "Mac: Parallel + Debug"
env: DEBUG=YES
MPI=YES
CODECOV=NO
@@ -211,11 +115,10 @@ jobs:
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
#
- os: osx
# osx_image: xcode7.3
compiler: clang
name: "Mac: Parallel"
env: DEBUG=NO
MPI=YES
CODECOV=YES
+13 -150
View File
@@ -13,185 +13,48 @@ Version 4.0.1 (development)
Improved GPU support
--------------------
- Added support for matrix-free diagonal smoothers on GPUs.
- Added initial support for AMD GPUs based on HIP: a C++ runtime API and kernel
language that can run on both AMD and NVIDIA hardware. With this change and
the libCEED addition below, the current list of available backends is:
"ceed-cuda", "occa-cuda", "raja-cuda", "cuda", "hip", "occa-omp", "raja-omp",
"omp", "ceed-cpu", "occa-cpu", "raja-cpu", and "cpu".
language that can run on both AMD and NVIDIA hardware. With this change, the
list of backends is: "occa-cuda", "raja-cuda", "cuda", "hip", "occa-omp",
"raja-omp", "omp", "occa-cpu", "raja-cpu", and "cpu".
- Improved RAJA backend and multi-GPU MPI communications.
libCEED support
---------------
- Added support for libCEED, the portable library for high-order operator
evaluation developed by the Center for Efficient Exascale Discretizations in
the Exascale Computing Project, https://github.com/CEED/libCEED.
- This initial integration includes Mass and Diffusion integrators. libCEED GPU
backends can be used without specific MFEM configuration, however it is highly
recommended to use the "cuda" build option to minimize memory transfers.
- Both CPU and GPU modes are available as MFEM device backends (ceed-cpu and
ceed-cuda), using some of the best performing CPU and GPU backends from
libCEED, see the sample runs in examples 1 and 6.
Discretization improvements
---------------------------
- Added support for non-conforming prism AMR, including coarsening and parallel
load balancing. Anisotropic prism refinement is only available in the serial
version at the moment.
Meshing improvements
--------------------
- Added support for non-conforming AMR on prisms and tetrahedra, including
coarsening and parallel load balancing. Anisotropic prism refinement is only
available in the serial version at the moment.
- The TMOP mesh optimization algorithms were extended to support r-adaptivity.
Target matrices can now be constructed either via a given analytical function
(e.g. spatial dependence of size, aspect ratio, etc., for each element) or via
a (Par)GridFunction specified on the original mesh.
- New method Mesh::GetHilbertElementOrdering for sorting mesh elements along the
Hilbert curve. The ordering can be used to improve caching and parallel
partitioning in non-conforming AMR.
- The TMOP mesh optimization algorithms have been improved to support AMR meshes.
- Added support for creating refined versions of periodic meshes, making use of
the new L2ElementRestriction class. This class also allows for computing
geometric factors on periodic meshes using partial assembly.
- The TMOP mesh optimization algorithms have been improved to support AMR meshes.
- Improved element numbering after uniform mesh refinement.
Discretization improvements
---------------------------
- Added support for GSLIB-FindPoints, a general high-order interpolation utility
that can robustly evaluate a GridFunction in an arbitrary collection of points
in physical space. See INSTALL for details on building MFEM with GSLIB, and
miniapps/gslib for examples of how to use this feature.
- Added support for serendipity elements of arbitrary order on affinely-mapped
square elements. Basis functions for these elements can be visualized using
an option in the display-basis miniapp.
- Added support for complex-valued finite element operators and fields using a
2x2 block structured linear system to mimic complex arithmetic. New classes
include: ComplexGridFunction, SesquilinearForm, ComplexLinearForm, and their
parallel counterparts.
- Two integrators related to Stokes problems, (Q grad u, v) and (Q div v, u),
where u and the components of v are in H1, were added/modified to support full
and partial assembly modes. See the new GradientIntegrator and the updated
VectorDivergenceIntegrator classes in fem/bilininteg.hpp, as well as the PA
kernels in fem/bilininteg_gradient.cpp and fem/bilininteg_divergence.cpp.
- Diagonals of partially assembled operators can now be computed efficiently.
See the new methods AssembleDiagonal in BilinearForm, AssembleDiagonalPA in
BilinearFormIntegrator and the implementations in fem/bilininteg_*.cpp.
- Added second order derivatives of NURBS shape functions.
- Added initial support for NonlinearForms to support the partial assembly mode.
- Added a nonlinear vector valued convection integrator (Q u \cdot grad u, v)
where u_i and v_i are in H1. This form occurs e.g. in the Navier-Stokes
equations. The integrator supports the partial assembly mode for its
action. In full assembly mode we also provide the GetGradient method that
computes the linearized version of the integrator.
- Added a new method, MixedBilinearForm::FormRectangularLinearSystem, that can
be used to impose boundary conditions on the non-square off-diagonal blocks of
a block operator (similar to FormLinearSystem in the square case).
- Extended the support for partial assembly to vector mass and vector diffusion
bilinear integrators.
Linear and nonlinear solvers
----------------------------
- Added a general interface for specifying and solving nonlinear constrained
optimization problems through the new classes OptimizationProblem and
OptimizationSolver, see linalg/solver.hpp
- Added support for HiOp, a lightweight HPC solver for nonlinear optimization
problems see class HiOpNLPOptimizer and the example codes in examples/hiop.
- Added support for Ginkgo, a high-performance linear algebra library for GPU
and manycore nodes, with a focus on sparse solution of linear systems. For
more details see linalg/ginkgo.hpp and the example code in examples/gingko.
- Added Adams-Bashforth and Adams-Moulton time integrators.
- Added a block ILU(0) preconditioner for DG-type discretizations. Example 9
(DG advection) now takes advantage of this for implicit time integration.
- Added a LinearSolve(A,X) convenience method to solve dense linear systems. In
the trivial cases, i.e., square matrices of size 1 or 2, the system is solved
directly, otherwise, LU factorization is employed.
New and updated examples and miniapps
-------------------------------------
- Added two new miniapps: Find Points (serial + parallel) and Field Diff in
miniapps/gslib that show how GSLIB-FindPoints can be used to interpolate a
(Par)GridFunction in an arbitrary number of physical space points in 2D and
3D. The GridFunction must be in H1 and in the same space as the mesh that is
used to find the points.
- Added a new example, Example 22/22p, to demonstrate the use of the new
complex-valued finite element operators. The example defines and solves
a family of time-harmonic PDEs related to damped harmonic oscillators.
- Updated Example 1/1p to use diagonal preconditioning in partial assembly mode.
- The mesh-optimizer and pmesh-optimizer miniapps have been updated to
demonstrate the new r-adaptivity capabilities of TMOP.
- New options to reorder and partition the mesh in the mesh-explorer miniapp.
- The mesh-explorer miniapp now supports visualization of boundary attributes of
3D meshes (key 'b').
- The (p)mesh-optimizer miniapp has been updated to demonstrate mesh
optimization for an AMR mesh.
- Added a modification of Example 1 in examples/ginkgo that demonstrates the use
of the Gingko interface to solve a linear system.
- Added a modification of ex9 in examples/hiop that demonstrates the nonlinear
constrained optimization interface and the use of the SLBQP and HiOp solvers.
- Added a collection of 7 playful miniapps in miniapps/toys that illustrate the
meshing and visualization features of the library in more relaxed settings.
The toys include simulations of cellular automata, Rubik's cube, Mandelbrot
set, a tool to convert any image to mfem mesh, and more.
- Example 9 and 9p now support implicit time integration, using the new block
ILU(0) solvers as preconditioners for the linear system.
Improved testing
----------------
- Added a new directory, tests/scripts, with several shell scripts that perform
simple checks on the code including: code styling, documentation formatting,
proper use of .gitignore, and preventing the accidental commit of large files.
- It is recommended that developers run the above tests scripts (via the runtest
script) before pushing to GitHub. See the README file in tests/scripts.
- The Travis CI settings have been updated to include an initial Checks stage
which currently runs the code-style, documentation and gitignore test scripts,
as well as a final stage for optional checks/tests which currently runs the
branch-history script.
Miscellaneous
-------------
- Upgraded the SUNDIALS interface to utilize version 5.0. This necessitated a
complete rework of the interface and requires changes at the application
level. Example usage of the new interface can be found in examples/sundials.
- Added support for output in the ParaView XML format. See Examples 5/5p, 9/9p
and the new ParaViewDataCollection class.
- Collected object files from the miniapps/common directory into a new library,
libmfem-common for the convenience of application developers. The new library
is now used in several miniapps in the electromagnetic and tools directories.
- Added unit tests for time integrators.
- Upgraded the SUNDIALS interface to utilize SUNDIALS version 5.0. This
necessitated a complete rework of the interface and requires changes at
the application level. Example usage of this new interface can be found
in the examples/sundials directory.
Version 4.0, released on May 24, 2019
+2 -28
View File
@@ -164,15 +164,6 @@ if (MFEM_USE_METIS)
find_package(METIS REQUIRED)
endif()
if (MFEM_USE_GINKGO)
find_package(Ginkgo REQUIRED)
if (Ginkgo_FOUND)
get_target_property(Ginkgo_INCLUDE_DIRS
Ginkgo::ginkgo INTERFACE_INCLUDE_DIRECTORIES)
set(Ginkgo_LIBRARIES Ginkgo::ginkgo)
endif()
endif()
# GZSTREAM -> zlib
if (MFEM_USE_GZSTREAM)
find_package(ZLIB REQUIRED)
@@ -257,10 +248,6 @@ if (MFEM_USE_MPFR)
find_package(MPFR REQUIRED)
endif()
if (MFEM_USE_CEED)
find_package(libCEED REQUIRED)
endif()
if (MFEM_USE_CONDUIT)
find_package(Conduit REQUIRED conduit relay blueprint )
endif()
@@ -286,12 +273,6 @@ if (MFEM_USE_PUMI)
endif()
endif()
# HiOp optimizer
if (MFEM_USE_HIOP)
find_package(HIOP REQUIRED)
# find_package updates HIOP_FOUND, HIOP_INCLUDE_DIRS, HIOP_LIBRARIES
endif()
# CUDA
if (MFEM_USE_CUDA)
set(CMAKE_CUDA_STANDARD 11)
@@ -341,8 +322,8 @@ endif()
# integers, the METIS header (with 32-bit indices, as used by mfem) needs to
# be before SuiteSparse.
set(MFEM_TPLS MPI_CXX OPENMP BLAS LAPACK METIS HYPRE SuiteSparse SUNDIALS PETSC
MESQUITE SuperLUDist STRUMPACK AXOM CONDUIT GECKO Ginkgo GNUTLS NETCDF MPFR
PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA)
MESQUITE SuperLUDist STRUMPACK AXOM CONDUIT GECKO GNUTLS NETCDF MPFR PUMI
POSIXCLOCKS MFEMBacktrace ZLIB OCCA RAJA)
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
set(TPL_LIBRARIES "")
set(TPL_INCLUDE_DIRS "")
@@ -554,13 +535,6 @@ if (MFEM_USE_OCCA)
FILES_MATCHING PATTERN "*.okl")
endif()
# Install the libCEED files
if (MFEM_USE_CEED)
install(DIRECTORY ${MFEM_SOURCE_DIRS}
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem
FILES_MATCHING PATTERN "fem/libceed/*.h")
endif()
# Install ${HEADERS}
# ---
# foreach (HDR ${HEADERS})
-5
View File
@@ -83,8 +83,6 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
│ └── web
│ └── examples
├── examples
│ ├── ginkgo
│ ├── hiop
│ ├── petsc
│ ├── pumi
│ └── sundials
@@ -95,7 +93,6 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
├── miniapps
│ ├── common
│ ├── electromagnetics
│ ├── gslib
│ ├── meshing
│ ├── nurbs
│ ├── performance
@@ -353,7 +350,6 @@ Before a PR can be merged, it should satisfy the following:
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
- [ ] List the new example in `doc/CodeDocumentation.dox`.
- [ ] If new examples directory (e.g.`examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
@@ -368,7 +364,6 @@ Before a PR can be merged, it should satisfy the following:
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
- [ ] Consider adding a new test for the new miniapp.
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
+7 -57
View File
@@ -349,12 +349,6 @@ MFEM_USE_STRUMPACK = YES/NO
classes. When enabled, this option uses the STRUMPACK_* library options, see
below.
MFEM_USE_GINKGO = YES/NO
Enable MFEM functionality based on the Ginkgo library, which provides
iterative linear solvers and preconditioners with OpenMP, CUDA backends, see
https://github.com/ginkgo-project/ginkgo. When enabled, the user can use
Ginkgo's solvers and preconditioners as shown in examples/ginkgo/.
MFEM_USE_GNUTLS = YES/NO
Enable secure socket support in class socketstream, using the auxiliary
GnuTLS_* classes, based on the GnuTLS library. This option may be useful in
@@ -421,10 +415,6 @@ MFEM_USE_PUMI = YES/NO
models and effectively supports automated adaptive analysis. PUMI enables
support for parallel unstructured mesh modifications in MFEM.
MFEM_USE_HIOP = YES/NO
Enable the usage of HiOp (https://github.com/LLNL/hiop) in MFEM. HiOp is an
HPC solver for nonlinear optimization problems.
MFEM_USE_CUDA = YES/NO
Enables support for CUDA devices in MFEM. CUDA is a parallel computing
platform and programming model for general computing on graphical processing
@@ -452,17 +442,6 @@ MFEM_USE_OCCA = YES/NO
backends. In order to use the OCCA CUDA backend, CUDA support must be enabled
in MFEM as well, i.e. MFEM_USE_CUDA=YES must be set.
MFEM_USE_GSLIB = YES/NO
Enables MFEM functionality based on the GSLIB library, and specifically its
FindPoints component, which provides a robust algorithms to evaluate finite
element functions in a collection of points in physical space. When enabled,
the user can use the GSLIB-FindPoints methods as shown in miniapps/gslib.
MFEM_USE_CEED = YES/NO
Enables support for the libCEED library in MFEM. libCEED is a portable
library for performant high-order operator evaluation developed by the Center
for Efficient Exascale Discretizations in the Exascale Computing Project.
MFEM_BUILD_TAG = (any value)
An optional tag to characterize the build. Exported to config/config.mk.
Can be used to identify the MFEM build from other makefiles.
@@ -539,12 +518,6 @@ The specific libraries and their options are:
URL: http://portal.nersc.gov/project/sparse/strumpack
Options: STRUMPACK_OPT, STRUMPACK_LIB.
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Note that Ginkgo needs a
C++ compiler that supports the C++-11 standard. For additional requirements
and dependencies of specific modules see the Ginkgo webpage below.
URL: https://ginkgo-project.github.io
Options: GINKGO_OPT (Not used), GINKGO_LIB.
- GnuTLS (optional), used when MFEM_USE_GNUTLS = YES. On most Linux systems,
GnuTLS is available as a development package, e.g. gnutls-devel. On Mac OS X,
one can get the library through the Homebrew package manager (http://brew.sh).
@@ -576,48 +549,29 @@ The specific libraries and their options are:
https://support.hdfgroup.org/HDF5 (HDF5)
Options: SIDRE_OPT, SIDRE_LIB.
- Conduit (optional), used when MFEM_USE_CONDUIT = YES. Conduit Mesh Blueprint
- Conduit, used when MFEM_USE_CONDUIT = YES. Direct Conduit Mesh Blueprint
support requires Conduit >= v0.3.1 and VisIt >= v2.13.1 to read the output.
URL: https://github.com/LLNL/conduit (Conduit)
https://support.hdfgroup.org/HDF5 (HDF5)
Options: CONDUIT_OPT, CONDUIT_LIB.
- PUMI (optional), used when MFEM_USE_PUMI = YES.
- PUMI, used when MFEM_USE_PUMI = YES.
URL: https://scorec.rpi.edu/pumi
Options: PUMI_OPT, PUMI_LIB.
- HiOp (optional), used when MFEM_USE_HIOP = YES.
URL: https://github.com/LLNL/hiop
Options: HIOP_OPT, HIOP_LIB.
- GSLIB (optional), used when MFEM_USE_GSLIB = YES. The gslib library must be
built prior to the MFEM build, as follows: download gslib-1.0.5, untar it at
the same level as MFEM and create a symbolic link: "ln -s gslib-1.0.5 gslib".
Build gslib in parallel or in serial based on the desired MFEM build: "make
clean; make CC=mpicc" or "make clean; make CC=gcc MPI=0". Build MFEM with
MFEM_USE_GSLIB=YES.
URL: https://github.com/gslib/gslib/archive/v1.0.5.tar.gz
Options: GSLIB_OPT, GSLIB_LIB.
- CUDA (optional), used when MFEM_USE_CUDA = YES.
- CUDA, used when MFEM_USE_CUDA = YES.
URL: https://developer.nvidia.com/cuda-toolkit
Options: CUDA_CXX, CUDA_ARCH, CUDA_OPT, CUDA_LIB.
- HIP (optional), used when MFEM_USE_HIP = YES.
- HIP, used when MFEM_USE_HIP = YES.
URL: https://rocm.github.io/ROCmInstall.html
Options: HIP_CXX, HIP_ARCH, HIP_OPT, HIP_LIB.
- OCCA (optional), used when MFEM_USE_OCCA = YES.
- OCCA, used when MFEM_USE_OCCA = YES.
URL: https://libocca.org
Options: OCCA_DIR, OCCA_OPT, OCCA_LIB.
- libCEED (optional), used when MFEM_USE_CEED = YES. Requires libCEED's master
branch, specifically, git-hash c00ee0d or later.
URL: https://github.com/CEED/libCEED
https://ceed.exascaleproject.org/libceed
Options: CEED_DIR, CEED_OPT, CEED_LIB.
- RAJA (optional), used when MFEM_USE_RAJA = YES.
- RAJA, used when MFEM_USE_RAJA = YES.
Beginning with MFEM v4.1, only RAJA v0.10.0+ is supported.
URL: https://github.com/LLNL/RAJA
Options: RAJA_DIR, RAJA_OPT, RAJA_LIB.
@@ -637,6 +591,7 @@ The specific libraries and their options are:
URL: https://zlib.net
Options: ZLIB_OPT, ZLIB_LIB.
Building with CMake
===================
The MFEM build system consists of two steps: configuration and compilation.
@@ -752,16 +707,13 @@ MFEM_USE_MESQUITE
MFEM_USE_SUITESPARSE
MFEM_USE_SUPERLU
MFEM_USE_STRUMPACK
MFEM_USE_GINKGO
MFEM_USE_GNUTLS
MFEM_USE_NETCDF
MFEM_USE_MPFR
MFEM_USE_GZSTREAM
MFEM_USE_PUMI
MFEM_USE_HIOP
MFEM_USE_CUDA
MFEM_USE_OCCA
MFEM_USE_CEED
MFEM_USE_RAJA
MFEM_USE_SIDRE
@@ -804,14 +756,12 @@ The CMake build system adds auto-detection for the following packages/libraries:
- SuiteSparse
- SuperLUDist, STRUMPACK
- ParMETIS
- Ginkgo
- GNUTLS - Extends the built-in CMake support, to search GNUTLS_DIR as well.
- NETCDF
- MPFR
- LIBUNWIND
- POSIXCLOCKS
- PUMI
- HIOP
- OCCA
- RAJA
- AXOM - Used when MFEM_USE_SIDRE is enabled
-4
View File
@@ -232,10 +232,6 @@ IF (DEFINED TPL_ENABLE_GECKO)
SET(MFEM_USE_GECKO ${TPL_ENABLE_GECKO} CACHE BOOL "Enable GECKO usage" FORCE)
ENDIF()
IF (DEFINED TPL_ENABLE_GINKGO)
SET(MFEM_USE_GINKGO ${TPL_ENABLE_GINKGO} CACHE BOOL "Enable GINKGO usage" FORCE)
ENDIF()
IF (DEFINED TPL_ENABLE_GNUTLS)
SET(MFEM_USE_GNUTLS ${TPL_ENABLE_GNUTLS} CACHE BOOL "Enable GNUTLS usage" FORCE)
ENDIF()
-2
View File
@@ -34,7 +34,6 @@ set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
set(MFEM_USE_GECKO @MFEM_USE_GECKO@)
set(MFEM_USE_GINKGO @MFEM_USE_GINKGO@)
set(MFEM_USE_GNUTLS @MFEM_USE_GNUTLS@)
set(MFEM_USE_NETCDF @MFEM_USE_NETCDF@)
set(MFEM_USE_PETSC @MFEM_USE_PETSC@)
@@ -45,7 +44,6 @@ set(MFEM_USE_PUMI @MFEM_USE_PUMI@)
set(MFEM_USE_CUDA @MFEM_USE_CUDA@)
set(MFEM_USE_OCCA @MFEM_USE_OCCA@)
set(MFEM_USE_RAJA @MFEM_USE_RAJA@)
set(MFEM_USE_CEED @MFEM_USE_CEED@)
set(MFEM_CXX_COMPILER "@CMAKE_CXX_COMPILER@")
set(MFEM_CXX_FLAGS "@CMAKE_CXX_FLAGS@")
-9
View File
@@ -92,9 +92,6 @@
// Enable functionality based on the Gecko library
#cmakedefine MFEM_USE_GECKO
// Enable functionality based on the Ginkgo library
#cmakedefine MFEM_USE_GINKGO
// Enable MFEM functionality based on the GnuTLS library
#cmakedefine MFEM_USE_GNUTLS
@@ -113,9 +110,6 @@
// Enable MFEM functionality based on the PUMI library
#cmakedefine MFEM_USE_PUMI
// Enable MFEM functionality based on the HiOp library
#cmakedefine MFEM_USE_HIOP
// Build the GPU/CUDA-enabled version of the MFEM library.
// Requires a CUDA compiler (nvcc).
#cmakedefine MFEM_USE_CUDA
@@ -126,9 +120,6 @@
// Enable MFEM functionality based on the OCCA library
#cmakedefine MFEM_USE_OCCA
// Enable MFEM functionality based on the libCEED library
#cmakedefine MFEM_USE_CEED
// Which library functions to use in class StopWatch for measuring time.
// For a list of the available options, see INSTALL.
// If not defined, an option is selected automatically.
-36
View File
@@ -1,36 +0,0 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Sets the following variables:
# - HIOP_FOUND
# - HIOP_INCLUDE_DIRS
# - HIOP_LIBRARIES
include(MfemCmakeUtilities)
mfem_find_package(HIOP HIOP HIOP_DIR
"include" "hiopInterface.hpp"
"lib" "hiop"
"Paths to headers required by HIOP."
"Libraries required by HIOP.")
# this test fails with parallel MFEM since mpi.h is not available (cxx compiler is used for some reason)
# CHECK_BUILD HIOP_VERSION_OK TRUE
#"
##include <hiopInterface.hpp>
#using namespace hiop;
#int main(int argc, char *argv[])
#{
# MPI_Init(&argc, &argv);
# MPI_Comm comm = MPI_COMM_WORLD;
#
# return 0;
#}
#")
-19
View File
@@ -1,19 +0,0 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Defines the following variables:
# - CEED_FOUND
# - CEED_LIBRARIES
# - CEED_INCLUDE_DIRS
include(MfemCmakeUtilities)
mfem_find_package(libCEED CEED CEED_DIR "include" ceed.h "lib" ceed
"Paths to headers required by libCEED." "Libraries required by libCEED.")
@@ -814,7 +814,7 @@ function(mfem_export_mk_files)
foreach(lib ${TPL_LIBRARIES})
get_filename_component(suffix ${lib} EXT)
# handle interfaces (e.g., SCOREC::apf)
if ("${lib}" MATCHES "SCOREC::.*" OR "${lib}" MATCHES "Ginkgo::.*")
if ("${lib}" MATCHES "SCOREC::.*")
elseif (NOT "${lib}" MATCHES "SCOREC::.*" AND "${lib}" MATCHES ".*::.*")
message(FATAL_ERROR "***** interface lib found ... exiting *****")
# handle static and shared libs
-12
View File
@@ -100,9 +100,6 @@
// Enable functionality based on the Gecko library
// #define MFEM_USE_GECKO
// Enable MFEM features based on the Ginkgo library
// #define MFEM_USE_GINKGO
// Enable secure socket streams based on the GNUTLS library
// #define MFEM_USE_GNUTLS
@@ -124,12 +121,6 @@
// Enable MFEM functionality based on the PUMI library
// #define MFEM_USE_PUMI
// Enable MFEM functionality based on the HIOP library.
// #define MFEM_USE_HIOP
// Enable MFEM functionality based on the GSLIB library
// #define MFEM_USE_GSLIB
// Build the NVIDIA GPU/CUDA-enabled version of the MFEM library.
// Requires a CUDA compiler (nvcc).
// #define MFEM_USE_CUDA
@@ -144,9 +135,6 @@
// Enable functionality based on the OCCA library.
// #define MFEM_USE_OCCA
// Enable functionality based on the libCEED library.
// #define MFEM_USE_CEED
// Version of HYPRE used for building MFEM.
// #define MFEM_HYPRE_VERSION @MFEM_HYPRE_VERSION@
-4
View File
@@ -34,7 +34,6 @@ MFEM_USE_SUITESPARSE = @MFEM_USE_SUITESPARSE@
MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
MFEM_USE_GECKO = @MFEM_USE_GECKO@
MFEM_USE_GINKGO = @MFEM_USE_GINKGO@
MFEM_USE_GNUTLS = @MFEM_USE_GNUTLS@
MFEM_USE_NETCDF = @MFEM_USE_NETCDF@
MFEM_USE_PETSC = @MFEM_USE_PETSC@
@@ -42,13 +41,10 @@ MFEM_USE_MPFR = @MFEM_USE_MPFR@
MFEM_USE_SIDRE = @MFEM_USE_SIDRE@
MFEM_USE_CONDUIT = @MFEM_USE_CONDUIT@
MFEM_USE_PUMI = @MFEM_USE_PUMI@
MFEM_USE_HIOP = @MFEM_USE_HIOP@
MFEM_USE_GSLIB = @MFEM_USE_GSLIB@
MFEM_USE_CUDA = @MFEM_USE_CUDA@
MFEM_USE_HIP = @MFEM_USE_HIP@
MFEM_USE_RAJA = @MFEM_USE_RAJA@
MFEM_USE_OCCA = @MFEM_USE_OCCA@
MFEM_USE_CEED = @MFEM_USE_CEED@
# Compiler, compile options, and link options
MFEM_CXX = @MFEM_CXX@
-11
View File
@@ -35,7 +35,6 @@ option(MFEM_USE_SUITESPARSE "Enable SuiteSparse usage" OFF)
option(MFEM_USE_SUPERLU "Enable SuperLU_DIST usage" OFF)
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
option(MFEM_USE_GECKO "Enable GECKO usage" OFF)
option(MFEM_USE_GINKGO "Enable Ginkgo usage" OFF)
option(MFEM_USE_GNUTLS "Enable GNUTLS usage" OFF)
option(MFEM_USE_NETCDF "Enable NETCDF usage" OFF)
option(MFEM_USE_PETSC "Enable PETSc support." OFF)
@@ -43,11 +42,9 @@ option(MFEM_USE_MPFR "Enable MPFR usage." OFF)
option(MFEM_USE_SIDRE "Enable Axom/Sidre usage" OFF)
option(MFEM_USE_CONDUIT "Enable Conduit usage" OFF)
option(MFEM_USE_PUMI "Enable PUMI" OFF)
option(MFEM_USE_HIOP "Enable HiOp" OFF)
option(MFEM_USE_CUDA "Enable CUDA" OFF)
option(MFEM_USE_OCCA "Enable OCCA" OFF)
option(MFEM_USE_RAJA "Enable RAJA" OFF)
option(MFEM_USE_CEED "Enable CEED" OFF)
set(MFEM_MPI_NP 4 CACHE STRING "Number of processes used for MPI tests")
@@ -139,8 +136,6 @@ set(ScaLAPACK_TARGET_NAMES scalapack)
set(GECKO_DIR "${MFEM_DIR}/../gecko" CACHE PATH "Path to the Gecko library.")
set(Ginkgo_DIR "${MFEM_DIR}/../ginkgo" CACHE PATH "Path to the Ginkgo library.")
set(GNUTLS_DIR "" CACHE PATH "Path to the GnuTLS library.")
set(NETCDF_DIR "" CACHE PATH "Path to the NetCDF library.")
@@ -165,14 +160,8 @@ set(Axom_REQUIRED_PACKAGES "Conduit/relay/blueprint" CACHE STRING
set(PUMI_DIR "${MFEM_DIR}/../pumi-2.1.0" CACHE STRING
"Directory where PUMI is installed")
set(HIOP_DIR "${MFEM_DIR}/../hiop/install" CACHE STRING
"Directory where HiOp is installed")
set(HIOP_REQUIRED_PACKAGES "BLAS" "LAPACK" CACHE STRING
"Packages that HiOp depends on.")
set(OCCA_DIR "${MFEM_DIR}/../occa" CACHE PATH "Path to OCCA")
set(RAJA_DIR "${MFEM_DIR}/../raja" CACHE PATH "Path to RAJA")
set(CEED_DIR "${MFEM_DIR}/../libCEED" CACHE PATH "Path to libCEED")
set(BLAS_INCLUDE_DIRS "" CACHE STRING "Path to BLAS headers.")
set(BLAS_LIBRARIES "" CACHE STRING "The BLAS library.")
-24
View File
@@ -122,7 +122,6 @@ MFEM_USE_SUITESPARSE = NO
MFEM_USE_SUPERLU = NO
MFEM_USE_STRUMPACK = NO
MFEM_USE_GECKO = NO
MFEM_USE_GINKGO = NO
MFEM_USE_GNUTLS = NO
MFEM_USE_NETCDF = NO
MFEM_USE_PETSC = NO
@@ -130,13 +129,10 @@ MFEM_USE_MPFR = NO
MFEM_USE_SIDRE = NO
MFEM_USE_CONDUIT = NO
MFEM_USE_PUMI = NO
MFEM_USE_HIOP = NO
MFEM_USE_GSLIB = NO
MFEM_USE_CUDA = NO
MFEM_USE_HIP = NO
MFEM_USE_RAJA = NO
MFEM_USE_OCCA = NO
MFEM_USE_CEED = NO
# Compile and link options for zlib.
ZLIB_DIR =
@@ -248,11 +244,6 @@ GECKO_DIR = @MFEM_DIR@/../gecko
GECKO_OPT = -I$(GECKO_DIR)/inc
GECKO_LIB = -L$(GECKO_DIR)/lib -lgecko
# Ginkgo library configuration (currently not needed)
GINKGO_DIR = @MFEM_DIR@/../ginkgo/install
GINKGO_OPT = -isystem $(GINKGO_DIR)/include
GINKGO_LIB = $(XLINKER)-rpath,$(GINKGO_DIR)/lib -L$(GINKGO_DIR)/lib -lginkgo -lginkgo_omp -lginkgo_cuda -lginkgo_reference
# GnuTLS library configuration
GNUTLS_OPT =
GNUTLS_LIB = -lgnutls
@@ -317,16 +308,6 @@ PUMI_OPT = -I$(PUMI_DIR)/include
PUMI_LIB = -L$(PUMI_DIR)/lib -lpumi -lcrv -lma -lmds -lapf -lpcu -lgmi -lparma\
-llion -lmth -lapf_zoltan -lspr
# HIOP
HIOP_DIR = @MFEM_DIR@/../hiop/install
HIOP_OPT = -I$(HIOP_DIR)/include
HIOP_LIB = -L$(HIOP_DIR)/lib -lhiop $(LAPACK_LIB)
# GSLIB library
GSLIB_DIR = @MFEM_DIR@/../gslib/build
GSLIB_OPT = -I$(GSLIB_DIR)/include
GSLIB_LIB = -L$(GSLIB_DIR)/lib -lgs
# CUDA library configuration (currently not needed)
CUDA_OPT =
CUDA_LIB =
@@ -340,11 +321,6 @@ OCCA_DIR = @MFEM_DIR@/../occa
OCCA_OPT = -I$(OCCA_DIR)/include
OCCA_LIB = $(XLINKER)-rpath,$(OCCA_DIR)/lib -L$(OCCA_DIR)/lib -locca
# libCEED library configuration
CEED_DIR ?= @MFEM_DIR@/../libCEED
CEED_OPT = -I$(CEED_DIR)/include
CEED_LIB = $(XLINKER)-rpath,$(CEED_DIR)/lib -L$(CEED_DIR)/lib -lceed
# RAJA library configuration
RAJA_DIR = @MFEM_DIR@/../raja
RAJA_OPT = -I$(RAJA_DIR)/include
+1 -35
View File
@@ -152,15 +152,6 @@ function extract_sample_runs()
fi
if [ "$device_runs" == "yes" ]; then
runs=`printf "%s" "$runs" | grep ".* -d .*"`
if [ "$have_occa" == "no" ]; then
runs=`printf "%s" "$runs" | grep -v ".* -d occa-.*"`
fi
if [ "$have_raja" == "no" ]; then
runs=`printf "%s" "$runs" | grep -v ".* -d raja-.*"`
fi
if [ "$have_ceed" == "no" ]; then
runs=`printf "%s" "$runs" | grep -v ".* -d ceed-.*"`
fi
else
runs=`printf "%s" "$runs" | grep -v ".* -d .*"`
fi
@@ -285,8 +276,7 @@ case "$1" in
;;
-dev)
device_runs="yes"
mfem_config+=" MFEM_USE_CUDA=YES MFEM_USE_OPENMP=YES"
# OCCA, RAJA, libCEED are enabled below, if available
mfem_config+=" MFEM_USE_CUDA=YES MFEM_USE_OCCA=YES MFEM_USE_RAJA=YES MFEM_USE_OPENMP=YES"
;;
-v)
valgrind="yes"
@@ -467,30 +457,6 @@ fi
TIMEFORMAT="${base_timeformat}"
# Setup optional libraries when not using externally built MFEM:
if [ "${built}" == "no" ]; then
have_occa="no"
have_raja="no"
have_ceed="no"
if [ "${device_runs}" == "yes" ]; then
if [ -n "${CUDA_ARCH}" ]; then
mfem_config+=" CUDA_ARCH=${CUDA_ARCH}"
fi
if [ -d "${mfem_dir}/../occa" ]; then
mfem_config+=" MFEM_USE_OCCA=YES"
have_occa="yes"
fi
if [ -d "${mfem_dir}/../raja" ]; then
mfem_config+=" MFEM_USE_RAJA=YES"
have_raja="yes"
fi
if [ -d "${mfem_dir}/../libCEED" ]; then
mfem_config+=" MFEM_USE_CEED=YES"
have_ceed="yes"
fi
fi
fi
function set_echo_log()
{
local dirname=`dirname "$1"`
-1
View File
@@ -82,7 +82,6 @@ test-par-NO: $(SEQ_$(MFEM_TESTS):=-test-seq)
test-ser: test-par-NO
test-par: test-par-YES
test: all test-par-$(MFEM_USE_MPI) clean-exec
test-noclean: all test-par-$(MFEM_USE_MPI)
test-clean: ; @rm -f *.stderr
test-print: mfem-test=printf " $(3) [$(2) ./$(1) -no-vis $(if $(4),$(4) )]\n"
test-print: mfem-test-file=printf " $(3) [$(2) ./$(1) -no-vis ]\n"
-1409
View File
File diff suppressed because it is too large Load Diff
+3 -6
View File
@@ -768,16 +768,13 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
@MFEM_SOURCE_DIR@/examples \
@MFEM_SOURCE_DIR@/examples/petsc \
@MFEM_SOURCE_DIR@/examples/pumi \
@MFEM_SOURCE_DIR@/examples/hiop \
@MFEM_SOURCE_DIR@/examples/sundials \
@MFEM_SOURCE_DIR@/miniapps/common \
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
@MFEM_SOURCE_DIR@/miniapps/gslib \
@MFEM_SOURCE_DIR@/miniapps/meshing \
@MFEM_SOURCE_DIR@/miniapps/nurbs \
@MFEM_SOURCE_DIR@/miniapps/performance \
@MFEM_SOURCE_DIR@/miniapps/tools \
@MFEM_SOURCE_DIR@/miniapps/toys
@MFEM_SOURCE_DIR@/miniapps/nurbs \
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
@MFEM_SOURCE_DIR@/miniapps/performance
# This tag can be used to specify the character encoding of the source files
# that doxygen parses. Internally doxygen uses the UTF-8 encoding. Doxygen uses
+1 -12
View File
@@ -81,8 +81,6 @@ namespace mfem {
* - <a class="el" href="ex20p_8cpp_source.html">Example 20p</a>: parallel symplectic ODE integration
* - <a class="el" href="ex21_8cpp_source.html">Example 21</a>: adaptive mesh refinement for linear elasticity
* - <a class="el" href="ex21p_8cpp_source.html">Example 21p</a>: parallel adaptive mesh refinement for linear elasticity
* - <a class="el" href="ex22_8cpp_source.html">Example 22</a>: complex-valued linear systems for damped harmonic oscillators
* - <a class="el" href="ex22p_8cpp_source.html">Example 22p</a>: parallel complex-valued linear systems for damped harmonic oscillators
*
* <H4>SUNDIALS Examples</H4>
* - Variants of Examples
@@ -117,12 +115,6 @@ namespace mfem {
* <a class="el" href="pumi_2ex6p_8cpp_source.html">6p</a>
* demonstrating the use of MFEM's \link pumi.hpp PUMI classes\endlink
*
* <H4>HiOp Examples</H4>
* - Variants of Examples
* <a class="el" href="hiop_2ex9_8cpp_source.html">9</a> and
* <a class="el" href="hiop_2ex9p_8cpp_source.html">9p</a>,
* demonstrating the use of MFEM's \link hiop.hpp HiOp classes\endlink
*
* <H3>Miniapps</H3>
* - <a class="el" href="volta_8cpp_source.html">Volta</a>: simple electrostatics simulation code
* - <a class="el" href="tesla_8cpp_source.html">Tesla</a>: simple magnetostatics simulation code
@@ -136,12 +128,9 @@ namespace mfem {
* - <a class="el" href="mesh-explorer_8cpp_source.html">Mesh Explorer</a>: visualize and manipulate meshes
* - <a class="el" href="mesh-optimizer_8cpp_source.html">Mesh Optimizer</a>: optimize high-order meshes, <a class="el" href="mesh-optimizer_8cpp_source.html">serial</a> and <a class="el" href="pmesh-optimizer_8cpp_source.html">parallel</a> versions
* - <a class="el" href="display-basis_8cpp_source.html">Display Basis</a>: visualize finite element basis functions
* - <a class="el" href="get-values_8cpp_source.html">Get Values</a>: extract field values via DataCollection classes
* - <a class="el" href="load-dc_8cpp_source.html">Load DC</a>: visualize fields saved via DataCollection classes
* - <a class="el" href="convert-dc_8cpp_source.html">Convert DC</a>: convert between different DataCollection formats
* - <a class="el" href="convert-dc_8cpp_source.html">Convert DC</a>: convert between diffirent DataCollection formats
* - <a class="el" href="lor-transfer_8cpp_source.html">LOR Transfer</a>: map functions between high-order and low-order refined spaces
* - <a class="el" href="findpts_8cpp_source.html">Find Points</a>: evaluate grid function in physical space, <a class="el" href="findpts_8cpp_source.html">serial</a> and <a class="el" href="pfindpts_8cpp_source.html">parallel</a> versions
* - <a class="el" href="field-diff_8cpp_source.html">Field Diff</a>: compare grid functions on different meshes
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Laplace problem
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Laplace problem
*
-12
View File
@@ -28,7 +28,6 @@ list(APPEND ALL_EXE_SRCS
ex19.cpp
ex20.cpp
ex21.cpp
ex22.cpp
)
if (MFEM_USE_MPI)
@@ -54,7 +53,6 @@ if (MFEM_USE_MPI)
ex19p.cpp
ex20p.cpp
ex21p.cpp
ex22p.cpp
)
endif()
@@ -102,11 +100,6 @@ if (MFEM_USE_SUNDIALS)
add_subdirectory(sundials)
endif()
# Include the examples/sundials directory if SUNDIALS is enabled.
if (MFEM_USE_GINKGO)
add_subdirectory(ginkgo)
endif()
# Include the examples/petsc directory if PETSc is enabled.
if (MFEM_USE_PETSC)
add_subdirectory(petsc)
@@ -116,8 +109,3 @@ endif()
if (MFEM_USE_PUMI)
add_subdirectory(pumi)
endif()
if (MFEM_USE_HIOP)
add_subdirectory(hiop)
endif()
+2 -5
View File
@@ -31,8 +31,6 @@
// ex1 -pa -d occa-cuda
// ex1 -pa -d raja-omp
// ex1 -pa -d occa-omp
// ex1 -pa -d ceed-cpu
// ex1 -pa -d ceed-cuda
// ex1 -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code demonstrates the use of MFEM to define a
@@ -196,10 +194,9 @@ int main(int argc, char *argv[])
umf_solver.Mult(B, X);
#endif
}
else // Jacobi preconditioning in partial assembly mode
else // No preconditioning for now in partial assembly mode.
{
OperatorJacobiSmoother M(*a, ess_tdof_list);
PCG(*A, M, B, X, 1, 400, 1e-12, 0.0);
CG(*A, B, X, 1, 2000, 1e-12, 0.0);
}
// 12. Recover the solution as a finite element grid function.
+1 -3
View File
@@ -179,9 +179,7 @@ int main(int argc, char *argv[])
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
"ODE solver: 1 - Backward Euler, 2 - SDIRK2, 3 - SDIRK3,\n\t"
" 11 - Forward Euler, 12 - RK2,\n\t"
" 13 - RK3 SSP, 14 - RK4."
" 22 - Implicit Midpoint Method,\n\t"
" 23 - SDIRK23 (A-stable), 24 - SDIRK34");
" 13 - RK3 SSP, 14 - RK4.");
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
+1 -3
View File
@@ -193,9 +193,7 @@ int main(int argc, char *argv[])
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
"ODE solver: 1 - Backward Euler, 2 - SDIRK2, 3 - SDIRK3,\n\t"
" 11 - Forward Euler, 12 - RK2,\n\t"
" 13 - RK3 SSP, 14 - RK4."
" 22 - Implicit Midpoint Method,\n\t"
" 23 - SDIRK23 (A-stable), 24 - SDIRK34");
" 13 - RK3 SSP, 14 - RK4.");
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
+2 -2
View File
@@ -8,7 +8,7 @@
// mpirun -np 4 ex12p -m ../data/beam-tet.mesh -s 462 -n 10 -o 2 -elast
// mpirun -np 4 ex12p -m ../data/beam-hex.mesh -s 3878
// mpirun -np 4 ex12p -m ../data/beam-wedge.mesh -s 81
// mpirun -np 4 ex12p -m ../data/beam-tri.mesh -s 3877 -o 2 -sys
// mpirun -np 4 ex12p -m ../data/beam-tri.mesh -s 3876 -o 2 -sys
// mpirun -np 4 ex12p -m ../data/beam-quad.mesh -s 4544 -n 6 -o 3 -elast
// mpirun -np 4 ex12p -m ../data/beam-quad-nurbs.mesh
// mpirun -np 4 ex12p -m ../data/beam-hex-nurbs.mesh
@@ -57,7 +57,7 @@ int main(int argc, char *argv[])
const char *mesh_file = "../data/beam-tri.mesh";
int order = 1;
int nev = 5;
int seed = 66;
int seed = 75;
bool visualization = 1;
bool amg_elast = 0;
+1 -1
View File
@@ -3,7 +3,7 @@
// Compile with: make ex13p
//
// Sample runs: mpirun -np 4 ex13p -m ../data/star.mesh
// mpirun -np 4 ex13p -m ../data/square-disc.mesh -o 2 -n 4
// mpirun -np 4 ex13p -m ../data/square-disc.mesh -o 2
// mpirun -np 4 ex13p -m ../data/beam-tet.mesh
// mpirun -np 4 ex13p -m ../data/beam-hex.mesh
// mpirun -np 4 ex13p -m ../data/escher.mesh
+4 -1
View File
@@ -16,6 +16,9 @@
// ex15 -m ../data/ball-nurbs.mesh -tf 0.3
// ex15 -m ../data/mobius-strip.mesh
// ex15 -m ../data/amr-quad.mesh
//
// Conforming meshes (no derefinement):
//
// ex15 -m ../data/square-disc.mesh
// ex15 -m ../data/escher.mesh -r 2 -tf 0.3
//
@@ -127,7 +130,7 @@ int main(int argc, char *argv[])
if (ref_levels > 0) { ref_levels--; }
mesh.SetCurvature(2);
}
mesh.EnsureNCMesh(true);
mesh.EnsureNCMesh();
for (int l = 0; l < ref_levels; l++)
{
mesh.UniformRefinement();
+4 -1
View File
@@ -16,6 +16,9 @@
// mpirun -np 4 ex15p -m ../data/ball-nurbs.mesh -tf 0.5
// mpirun -np 4 ex15p -m ../data/mobius-strip.mesh
// mpirun -np 4 ex15p -m ../data/amr-quad.mesh
//
// Conforming meshes (no load balancing and derefinement):
//
// mpirun -np 4 ex15p -m ../data/square-disc.mesh
// mpirun -np 4 ex15p -m ../data/escher.mesh -r 2 -tf 0.3
//
@@ -143,7 +146,7 @@ int main(int argc, char *argv[])
if (ref_levels > 0) { ref_levels--; }
mesh->SetCurvature(2);
}
mesh->EnsureNCMesh(true);
mesh->EnsureNCMesh();
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
+2 -11
View File
@@ -29,8 +29,6 @@
// mpirun -np 4 ex1p -pa -d cuda
// mpirun -np 4 ex1p -pa -d occa-cuda
// mpirun -np 4 ex1p -pa -d raja-omp
// mpirun -np 4 ex1p -pa -d ceed-cpu
// mpirun -np 4 ex1p -pa -d ceed-cuda
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Laplace problem
@@ -210,16 +208,9 @@ int main(int argc, char *argv[])
// 13. Solve the linear system A X = B.
// * With full assembly, use the BoomerAMG preconditioner from hypre.
// * With partial assembly, use Jacobi smoothing, for now.
// * With partial assembly, use no preconditioner, for now.
Solver *prec = NULL;
if (pa)
{
prec = new OperatorJacobiSmoother(*a, ess_tdof_list);
}
else
{
prec = new HypreBoomerAMG;
}
if (!pa) { prec = new HypreBoomerAMG; }
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
-561
View File
@@ -1,561 +0,0 @@
// MFEM Example 22
//
// Compile with: make ex22
//
// Sample runs: ex22 -m ../data/inline-segment.mesh -o 3
// ex22 -m ../data/inline-tri.mesh -o 3
// ex22 -m ../data/inline-quad.mesh -o 3
// ex22 -m ../data/inline-quad.mesh -o 3 -p 1
// ex22 -m ../data/inline-quad.mesh -o 3 -p 2
// ex22 -m ../data/inline-tet.mesh -o 2
// ex22 -m ../data/inline-hex.mesh -o 2
// ex22 -m ../data/inline-hex.mesh -o 2 -p 1
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0
//
// Description: This example code demonstrates the use of MFEM to define and
// solve simple complex-valued linear systems. It implements three
// variants of a damped harmonic oscillator:
//
// 1) A scalar H1 field
// -Div(a Grad u) - omega^2 b u + i omega c u = 0
//
// 2) A vector H(Curl) field
// Curl(a Curl u) - omega^2 b u + i omega c u = 0
//
// 3) A vector H(Div) field
// -Grad(a Div u) - omega^2 b u + i omega c u = 0
//
// In each case the field is driven by a forced oscillation, with
// angular frequency omega, imposed at the boundary or a portion
// of the boundary.
//
// In electromagnetics, the coefficients are typically named the
// permeability, mu = 1/a, permittivity, epsilon = b, and
// conductivity, sigma = c. The user can specify these constants
// using either set of names.
//
// The example also demonstrates how to display a time-varying
// solution as a sequence of fields sent to a single GLVis socket.
//
// We recommend viewing examples 1, 3 and 4 before viewing this
// example.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
static double mu_ = 1.0;
static double epsilon_ = 1.0;
static double sigma_ = 20.0;
static double omega_ = 10.0;
double u0_real_exact(const Vector &);
double u0_imag_exact(const Vector &);
void u1_real_exact(const Vector &, Vector &);
void u1_imag_exact(const Vector &, Vector &);
void u2_real_exact(const Vector &, Vector &);
void u2_imag_exact(const Vector &, Vector &);
bool check_for_inline_mesh(const char * mesh_file);
int main(int argc, char *argv[])
{
// 1. Parse command-line options.
const char *mesh_file = "../data/inline-quad.mesh";
int ref_levels = 0;
int order = 1;
int prob = 0;
double freq = -1.0;
double a_coef = 0.0;
bool visualization = 1;
bool herm_conv = true;
bool exact_sol = true;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&ref_levels, "-r", "--refine",
"Number of times to refine the mesh uniformly.");
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree).");
args.AddOption(&prob, "-p", "--problem-type",
"Choose between 0: H_1, 1: H(Curl), or 2: H(Div) "
"damped harmonic oscillator.");
args.AddOption(&a_coef, "-a", "--stiffness-coef",
"Stiffness coefficient (spring constant or 1/mu).");
args.AddOption(&epsilon_, "-b", "--mass-coef",
"Mass coefficient (or epsilon).");
args.AddOption(&sigma_, "-c", "--damping-coef",
"Damping coefficient (or sigma).");
args.AddOption(&mu_, "-mu", "--permeability",
"Permeability of free space (or 1/(spring constant)).");
args.AddOption(&epsilon_, "-eps", "--permittivity",
"Permittivity of free space (or mass constant).");
args.AddOption(&sigma_, "-sigma", "--conductivity",
"Conductivity (or damping constant).");
args.AddOption(&freq, "-f", "--frequency",
"Frequency (in Hz).");
args.AddOption(&herm_conv, "-herm", "--hermitian", "-no-herm",
"--no-hermitian", "Use convention for Hermitian operators.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.Parse();
if (!args.Good())
{
args.PrintUsage(cout);
return 1;
}
args.PrintOptions(cout);
MFEM_VERIFY(prob >= 0 && prob <=2,
"Unrecognized problem type: " << prob);
if ( a_coef != 0.0 )
{
mu_ = 1.0 / a_coef;
}
if ( freq > 0.0 )
{
omega_ = 2.0 * M_PI * freq;
}
exact_sol = check_for_inline_mesh(mesh_file);
if (exact_sol)
{
cout << "Identified a mesh with known exact solution" << endl;
}
ComplexOperator::Convention conv =
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
// 2. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes
// with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 3. Refine the mesh to increase resolution. In this example we do
// 'ref_levels' of uniform refinement where the user specifies
// the number of levels with the '-r' option.
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
}
// 4. Define a finite element space on the mesh. Here we use continuous
// Lagrange, Nedelec, or Raviart-Thomas finite elements of the specified
// order.
if (dim == 1 && prob != 0 )
{
cout << "Switching to problem type 0, H1 basis functions, "
<< "for 1 dimensional mesh." << endl;
prob = 0;
}
FiniteElementCollection *fec = NULL;
switch (prob)
{
case 0: fec = new H1_FECollection(order, dim); break;
case 1: fec = new ND_FECollection(order, dim); break;
case 2: fec = new RT_FECollection(order - 1, dim); break;
default: break; // This should be unreachable
}
FiniteElementSpace *fespace = new FiniteElementSpace(mesh, fec);
cout << "Number of finite element unknowns: " << fespace->GetTrueVSize()
<< endl;
// 5. Determine the list of true (i.e. conforming) essential boundary dofs.
// In this example, the boundary conditions are defined based on the type
// of mesh and the problem type.
Array<int> ess_tdof_list;
Array<int> ess_bdr;
if (mesh->bdr_attributes.Size())
{
ess_bdr.SetSize(mesh->bdr_attributes.Max());
ess_bdr = 1;
if (exact_sol)
{
switch (prob)
{
case 0: ess_bdr = 0; ess_bdr[0] = 1; break;
default: ess_bdr = 1; ess_bdr[2] = 0; break;
}
}
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 6. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system.
ComplexLinearForm b(fespace, conv);
b.Vector::operator=(0.0);
// 7. Define the solution vector u as a complex finite element grid function
// corresponding to fespace. Initialize u with initial guess of 1+0i or
// the exact solution if it is known.
ComplexGridFunction u(fespace);
ComplexGridFunction * u_exact = NULL;
if (exact_sol) { u_exact = new ComplexGridFunction(fespace); }
FunctionCoefficient u0_r(u0_real_exact);
FunctionCoefficient u0_i(u0_imag_exact);
VectorFunctionCoefficient u1_r(dim, u1_real_exact);
VectorFunctionCoefficient u1_i(dim, u1_imag_exact);
VectorFunctionCoefficient u2_r(dim, u2_real_exact);
VectorFunctionCoefficient u2_i(dim, u2_imag_exact);
ConstantCoefficient zeroCoef(0.0);
ConstantCoefficient oneCoef(1.0);
Vector zeroVec(dim); zeroVec = 0.0;
Vector oneVec(dim); oneVec = 0.0; oneVec[(prob==2)?(dim-1):0] = 1.0;
VectorConstantCoefficient zeroVecCoef(zeroVec);
VectorConstantCoefficient oneVecCoef(oneVec);
u = 0.0;
switch (prob)
{
case 0:
u.ProjectBdrCoefficient(oneCoef, zeroCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u0_r, u0_i); }
break;
case 1:
u.ProjectBdrCoefficientTangent(oneVecCoef, zeroVecCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u1_r, u1_i); }
break;
case 2:
u.ProjectBdrCoefficientNormal(oneVecCoef, zeroVecCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u2_r, u2_i); }
break;
default: break; // This should be unreachable
}
if (visualization && exact_sol)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *mesh << u_exact->real()
<< "window_title 'Exact: Real Part'" << flush;
sol_sock_i << "solution\n" << *mesh << u_exact->imag()
<< "window_title 'Exact: Imaginary Part'" << flush;
}
// 8. Set up the sesquilinear form a(.,.) on the finite element space
// corresponding to the damped harmonic oscillator operator of the
// appropriate type:
//
// 0) A scalar H1 field
// -Div(a Grad) - omega^2 b + i omega c
//
// 1) A vector H(Curl) field
// Curl(a Curl) - omega^2 b + i omega c
//
// 2) A vector H(Div) field
// -Grad(a Div) - omega^2 b + i omega c
//
ConstantCoefficient stiffnessCoef(1.0/mu_);
ConstantCoefficient massCoef(-omega_ * omega_ * epsilon_);
ConstantCoefficient lossCoef(omega_ * sigma_);
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
SesquilinearForm *a = new SesquilinearForm(fespace, conv);
switch (prob)
{
case 0:
a->AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new MassIntegrator(massCoef),
new MassIntegrator(lossCoef));
break;
case 1:
a->AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
break;
case 2:
a->AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
break;
default: break; // This should be unreachable
}
// 8a. Set up the bilinear form for the preconditioner corresponding to the
// appropriate operator
//
// 0) A scalar H1 field
// -Div(a Grad) - omega^2 b + omega c
//
// 1) A vector H(Curl) field
// Curl(a Curl) + omega^2 b + omega c
//
// 2) A vector H(Div) field
// -Grad(a Div) - omega^2 b + omega c
//
BilinearForm *pcOp = new BilinearForm(fespace);
switch (prob)
{
case 0:
pcOp->AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new MassIntegrator(massCoef));
pcOp->AddDomainIntegrator(new MassIntegrator(lossCoef));
break;
case 1:
pcOp->AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(negMassCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(lossCoef));
break;
case 2:
pcOp->AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(lossCoef));
break;
default: break; // This should be unreachable
}
// 9. Assemble the form and the corresponding linear system, applying any
// necessary transformations such as: assembly, eliminating boundary
// conditions, conforming constraints for non-conforming AMR, etc.
a->Assemble();
pcOp->Assemble();
OperatorHandle A;
Vector B, U;
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
u = 0.0;
U = 0.0;
OperatorHandle PCOp;
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
{
ComplexSparseMatrix * Asp =
dynamic_cast<ComplexSparseMatrix*>(A.Ptr());
cout << "Size of linear system: "
<< 2 * Asp->real().Width() << endl << endl;
}
// 10. Define and apply a GMRES solver for AU=B with a block diagonal
// preconditioner based on the appropriate sparse smoother.
{
Array<int> blockOffsets;
blockOffsets.SetSize(3);
blockOffsets[0] = 0;
blockOffsets[1] = PCOp.Ptr()->Height();
blockOffsets[2] = PCOp.Ptr()->Height();
blockOffsets.PartialSum();
BlockDiagonalPreconditioner BDP(blockOffsets);
Operator * pc_r = NULL;
Operator * pc_i = NULL;
switch (prob)
{
case 0: // fallthrough to case 2
case 2:
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
break;
case 1:
pc_r = new GSSmoother(*PCOp.As<SparseMatrix>());
break;
default: break; // This should be unreachable
}
pc_i = new ScaledOperator(pc_r,
(conv == ComplexOperator::HERMITIAN) ?
1.0:-1.0);
BDP.SetDiagonalBlock(0, pc_r);
BDP.SetDiagonalBlock(1, pc_i);
BDP.owns_blocks = 1;
GMRESSolver gmres;
gmres.SetPreconditioner(BDP);
gmres.SetOperator(*A.Ptr());
gmres.SetRelTol(1e-12);
gmres.SetMaxIter(1000);
gmres.SetPrintLevel(1);
gmres.Mult(B, U);
}
// 11. Recover the solution as a finite element grid function and compute the
// errors if the exact solution is known.
a->RecoverFEMSolution(U, b, u);
if (exact_sol)
{
double err_r = -1.0;
double err_i = -1.0;
switch (prob)
{
case 0:
err_r = u.real().ComputeL2Error(u0_r);
err_i = u.imag().ComputeL2Error(u0_i);
break;
case 1:
err_r = u.real().ComputeL2Error(u1_r);
err_i = u.imag().ComputeL2Error(u1_i);
break;
case 2:
err_r = u.real().ComputeL2Error(u2_r);
err_i = u.imag().ComputeL2Error(u2_i);
break;
default: break; // This should be unreachable
}
cout << endl;
cout << "|| Re (u_h - u) ||_{L^2} = " << err_r << endl;
cout << "|| Im (u_h - u) ||_{L^2} = " << err_i << endl;
cout << endl;
}
// 12. Save the refined mesh and the solution. This output can be viewed
// later using GLVis: "glvis -m mesh -g sol".
{
ofstream mesh_ofs("refined.mesh");
mesh_ofs.precision(8);
mesh->Print(mesh_ofs);
ofstream sol_r_ofs("sol_r.gf");
ofstream sol_i_ofs("sol_i.gf");
sol_r_ofs.precision(8);
sol_i_ofs.precision(8);
u.real().Save(sol_r_ofs);
u.imag().Save(sol_i_ofs);
}
// 13. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *mesh << u.real()
<< "window_title 'Solution: Real Part'" << flush;
sol_sock_i << "solution\n" << *mesh << u.imag()
<< "window_title 'Solution: Imaginary Part'" << flush;
}
if (visualization && exact_sol)
{
*u_exact -= u;
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *mesh << u_exact->real()
<< "window_title 'Error: Real Part'" << flush;
sol_sock_i << "solution\n" << *mesh << u_exact->imag()
<< "window_title 'Error: Imaginary Part'" << flush;
}
if (visualization)
{
GridFunction u_t(fespace);
u_t = u.real();
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "solution\n" << *mesh << u_t
<< "window_title 'Harmonic Solution (t = 0.0 T)'"
<< "pause\n" << flush;
cout << "GLVis visualization paused."
<< " Press space (in the GLVis window) to resume it.\n";
int num_frames = 32;
int i = 0;
while (sol_sock)
{
double t = (double)(i % num_frames) / num_frames;
ostringstream oss;
oss << "Harmonic Solution (t = " << t << " T)";
add(cos( 2.0 * M_PI * t), u.real(),
sin(-2.0 * M_PI * t), u.imag(), u_t);
sol_sock << "solution\n" << *mesh << u_t
<< "window_title '" << oss.str() << "'" << flush;
i++;
}
}
// 14. Free the used memory.
delete a;
delete u_exact;
delete pcOp;
delete fespace;
delete fec;
delete mesh;
return 0;
}
bool check_for_inline_mesh(const char * mesh_file)
{
string file(mesh_file);
size_t p0 = file.find_last_of("/");
string s0 = file.substr((p0==string::npos)?0:(p0+1),7);
return s0 == "inline-";
}
complex<double> u0_exact(const Vector &x)
{
int dim = x.Size();
complex<double> i(0.0, 1.0);
complex<double> alpha = (epsilon_ * omega_ - i * sigma_);
complex<double> kappa = std::sqrt(mu_ * omega_* alpha);
return std::exp(-i * kappa * x[dim - 1]);
}
double u0_real_exact(const Vector &x)
{
return u0_exact(x).real();
}
double u0_imag_exact(const Vector &x)
{
return u0_exact(x).imag();
}
void u1_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_real_exact(x);
}
void u1_imag_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_imag_exact(x);
}
void u2_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_real_exact(x);
}
void u2_imag_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_imag_exact(x);
}
-622
View File
@@ -1,622 +0,0 @@
// MFEM Example 22 - Parallel Version
//
// Compile with: make ex22p
//
// Sample runs: mpirun -np 4 ex22p -m ../data/inline-segment.mesh -o 3
// mpirun -np 4 ex22p -m ../data/inline-tri.mesh -o 3
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 3
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 3 -p 1
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 3 -p 2
// mpirun -np 4 ex22p -m ../data/inline-tet.mesh -o 2
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2 -p 1
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2 -p 2
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0
//
// Description: This example code demonstrates the use of MFEM to define and
// solve simple complex-valued linear systems. It implements three
// variants of a damped harmonic oscillator:
//
// 1) A scalar H1 field
// -Div(a Grad u) - omega^2 b u + i omega c u = 0
//
// 2) A vector H(Curl) field
// Curl(a Curl u) - omega^2 b u + i omega c u = 0
//
// 3) A vector H(Div) field
// -Grad(a Div u) - omega^2 b u + i omega c u = 0
//
// In each case the field is driven by a forced oscillation, with
// angular frequency omega, imposed at the boundary or a portion
// of the boundary.
//
// In electromagnetics the coefficients are typically named the
// permeability, mu = 1/a, permittivity, epsilon = b, and
// conductivity, sigma = c. The user can specify these constants
// using either set of names.
//
// The example also demonstrates how to display a time-varying
// solution as a sequence of fields sent to a single GLVis socket.
//
// We recommend viewing examples 1, 3 and 4 before viewing this
// example.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
static double mu_ = 1.0;
static double epsilon_ = 1.0;
static double sigma_ = 20.0;
static double omega_ = 10.0;
double u0_real_exact(const Vector &);
double u0_imag_exact(const Vector &);
void u1_real_exact(const Vector &, Vector &);
void u1_imag_exact(const Vector &, Vector &);
void u2_real_exact(const Vector &, Vector &);
void u2_imag_exact(const Vector &, Vector &);
bool check_for_inline_mesh(const char * mesh_file);
int main(int argc, char *argv[])
{
// 1. Initialize MPI.
int num_procs, myid;
MPI_Init(&argc, &argv);
MPI_Comm_size(MPI_COMM_WORLD, &num_procs);
MPI_Comm_rank(MPI_COMM_WORLD, &myid);
// 2. Parse command-line options.
const char *mesh_file = "../data/inline-quad.mesh";
int ser_ref_levels = 1;
int par_ref_levels = 1;
int order = 1;
int prob = 0;
double freq = -1.0;
double a_coef = 0.0;
bool visualization = 1;
bool herm_conv = true;
bool exact_sol = true;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&ser_ref_levels, "-rs", "--refine-serial",
"Number of times to refine the mesh uniformly in serial.");
args.AddOption(&par_ref_levels, "-rp", "--refine-parallel",
"Number of times to refine the mesh uniformly in parallel.");
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree).");
args.AddOption(&prob, "-p", "--problem-type",
"Choose between 0: H_1, 1: H(Curl), or 2: H(Div) "
"damped harmonic oscillator.");
args.AddOption(&a_coef, "-a", "--stiffness-coef",
"Stiffness coefficient (spring constant or 1/mu).");
args.AddOption(&epsilon_, "-b", "--mass-coef",
"Mass coefficient (or epsilon).");
args.AddOption(&sigma_, "-c", "--damping-coef",
"Damping coefficient (or sigma).");
args.AddOption(&mu_, "-mu", "--permeability",
"Permeability of free space (or 1/(spring constant)).");
args.AddOption(&epsilon_, "-eps", "--permittivity",
"Permittivity of free space (or mass constant).");
args.AddOption(&sigma_, "-sigma", "--conductivity",
"Conductivity (or damping constant).");
args.AddOption(&freq, "-f", "--frequency",
"Frequency (in Hz).");
args.AddOption(&herm_conv, "-herm", "--hermitian", "-no-herm",
"--no-hermitian", "Use convention for Hermitian operators.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.Parse();
if (!args.Good())
{
if (myid == 0)
{
args.PrintUsage(cout);
}
MPI_Finalize();
return 1;
}
if (myid == 0)
{
args.PrintOptions(cout);
}
MFEM_VERIFY(prob >= 0 && prob <=2,
"Unrecognized problem type: " << prob);
if ( a_coef != 0.0 )
{
mu_ = 1.0 / a_coef;
}
if ( freq > 0.0 )
{
omega_ = 2.0 * M_PI * freq;
}
exact_sol = check_for_inline_mesh(mesh_file);
if (myid == 0 && exact_sol)
{
cout << "Identified a mesh with known exact solution" << endl;
}
ComplexOperator::Convention conv =
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
// 3. Read the (serial) mesh from the given mesh file on all processors. We
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
// and volume meshes with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Refine the serial mesh on all processors to increase the resolution.
for (int l = 0; l < ser_ref_levels; l++)
{
mesh->UniformRefinement();
}
// 5. Define a parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
for (int l = 0; l < par_ref_levels; l++)
{
pmesh->UniformRefinement();
}
// 6. Define a parallel finite element space on the parallel mesh. Here we
// use continuous Lagrange, Nedelec, or Raviart-Thomas finite elements of
// the specified order.
if (dim == 1 && prob != 0 )
{
if (myid == 0)
{
cout << "Switching to problem type 0, H1 basis functions, "
<< "for 1 dimensional mesh." << endl;
}
prob = 0;
}
FiniteElementCollection *fec = NULL;
switch (prob)
{
case 0: fec = new H1_FECollection(order, dim); break;
case 1: fec = new ND_FECollection(order, dim); break;
case 2: fec = new RT_FECollection(order - 1, dim); break;
default: break; // This should be unreachable
}
ParFiniteElementSpace *fespace = new ParFiniteElementSpace(pmesh, fec);
HYPRE_Int size = fespace->GlobalTrueVSize();
if (myid == 0)
{
cout << "Number of finite element unknowns: " << size << endl;
}
// 7. Determine the list of true (i.e. parallel conforming) essential
// boundary dofs. In this example, the boundary conditions are defined
// based on the type of mesh and the problem type.
Array<int> ess_tdof_list;
Array<int> ess_bdr;
if (pmesh->bdr_attributes.Size())
{
ess_bdr.SetSize(pmesh->bdr_attributes.Max());
ess_bdr = 1;
if (exact_sol)
{
switch (prob)
{
case 0: ess_bdr = 0; ess_bdr[0] = 1; break;
default: ess_bdr = 1; ess_bdr[2] = 0; break;
}
}
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 8. Set up the parallel linear form b(.) which corresponds to the
// right-hand side of the FEM linear system.
ParComplexLinearForm b(fespace, conv);
b.Vector::operator=(0.0);
// 9. Define the solution vector u as a parallel complex finite element grid
// function corresponding to fespace. Initialize u with initial guess of
// 1+0i or the exact solution if it is known.
ParComplexGridFunction u(fespace);
ParComplexGridFunction * u_exact = NULL;
if (exact_sol) { u_exact = new ParComplexGridFunction(fespace); }
FunctionCoefficient u0_r(u0_real_exact);
FunctionCoefficient u0_i(u0_imag_exact);
VectorFunctionCoefficient u1_r(dim, u1_real_exact);
VectorFunctionCoefficient u1_i(dim, u1_imag_exact);
VectorFunctionCoefficient u2_r(dim, u2_real_exact);
VectorFunctionCoefficient u2_i(dim, u2_imag_exact);
ConstantCoefficient zeroCoef(0.0);
ConstantCoefficient oneCoef(1.0);
Vector zeroVec(dim); zeroVec = 0.0;
Vector oneVec(dim); oneVec = 0.0; oneVec[(prob==2)?(dim-1):0] = 1.0;
VectorConstantCoefficient zeroVecCoef(zeroVec);
VectorConstantCoefficient oneVecCoef(oneVec);
u = 0.0;
switch (prob)
{
case 0:
u.ProjectBdrCoefficient(oneCoef, zeroCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u0_r, u0_i); }
break;
case 1:
u.ProjectBdrCoefficientTangent(oneVecCoef, zeroVecCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u1_r, u1_i); }
break;
case 2:
u.ProjectBdrCoefficientNormal(oneVecCoef, zeroVecCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u2_r, u2_i); }
break;
default: break; // This should be unreachable
}
if (visualization && exact_sol)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *pmesh << u_exact->real()
<< "window_title 'Exact: Real Part'" << flush;
sol_sock_i << "solution\n" << *pmesh << u_exact->imag()
<< "window_title 'Exact: Imaginary Part'" << flush;
}
// 10. Set up the parallel sesquilinear form a(.,.) on the finite element
// space corresponding to the damped harmonic oscillator operator of the
// appropriate type:
//
// 0) A scalar H1 field
// -Div(a Grad) - omega^2 b + i omega c
//
// 1) A vector H(Curl) field
// Curl(a Curl) - omega^2 b + i omega c
//
// 2) A vector H(Div) field
// -Grad(a Div) - omega^2 b + i omega c
//
ConstantCoefficient stiffnessCoef(1.0/mu_);
ConstantCoefficient massCoef(-omega_ * omega_ * epsilon_);
ConstantCoefficient lossCoef(omega_ * sigma_);
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
ParSesquilinearForm *a = new ParSesquilinearForm(fespace, conv);
switch (prob)
{
case 0:
a->AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new MassIntegrator(massCoef),
new MassIntegrator(lossCoef));
break;
case 1:
a->AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
break;
case 2:
a->AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
break;
default: break; // This should be unreachable
}
// 10a. Set up the parallel bilinear form for the preconditioner
// corresponding to the appropriate operator
//
// 0) A scalar H1 field
// -Div(a Grad) - omega^2 b + omega c
//
// 1) A vector H(Curl) field
// Curl(a Curl) + omega^2 b + omega c
//
// 2) A vector H(Div) field
// -Grad(a Div) - omega^2 b + omega c
//
ParBilinearForm *pcOp = new ParBilinearForm(fespace);
switch (prob)
{
case 0:
pcOp->AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new MassIntegrator(massCoef));
pcOp->AddDomainIntegrator(new MassIntegrator(lossCoef));
break;
case 1:
pcOp->AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(negMassCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(lossCoef));
break;
case 2:
pcOp->AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(lossCoef));
break;
default: break; // This should be unreachable
}
// 11. Assemble the parallel bilinear form and the corresponding linear
// system, applying any necessary transformations such as: parallel
// assembly, eliminating boundary conditions, applying conforming
// constraints for non-conforming AMR, etc.
a->Assemble();
pcOp->Assemble();
OperatorHandle A;
Vector B, U;
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
u = 0.0;
U = 0.0;
OperatorHandle PCOp;
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
if (myid == 0)
{
ComplexHypreParMatrix * Ahyp =
dynamic_cast<ComplexHypreParMatrix*>(A.Ptr());
cout << "Size of linear system: "
<< 2 * Ahyp->real().GetGlobalNumRows() << endl << endl;
}
// 12. Define and apply a parallel FGMRES solver for AU=B with a block
// diagonal preconditioner based on the appropriate multigrid
// preconditioner from hypre.
{
Array<HYPRE_Int> blockTrueOffsets;
blockTrueOffsets.SetSize(3);
blockTrueOffsets[0] = 0;
blockTrueOffsets[1] = PCOp.Ptr()->Height();
blockTrueOffsets[2] = PCOp.Ptr()->Height();
blockTrueOffsets.PartialSum();
BlockDiagonalPreconditioner BDP(blockTrueOffsets);
Operator * pc_r = NULL;
Operator * pc_i = NULL;
switch (prob)
{
case 0:
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
break;
case 1:
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
break;
case 2:
if (dim == 2 )
{
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
}
else
{
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), fespace);
}
break;
default: break; // This should be unreachable
}
pc_i = new ScaledOperator(pc_r,
(conv == ComplexOperator::HERMITIAN) ?
1.0:-1.0);
BDP.SetDiagonalBlock(0, pc_r);
BDP.SetDiagonalBlock(1, pc_i);
BDP.owns_blocks = 1;
FGMRESSolver fgmres(MPI_COMM_WORLD);
fgmres.SetPreconditioner(BDP);
fgmres.SetOperator(*A.Ptr());
fgmres.SetRelTol(1e-12);
fgmres.SetMaxIter(1000);
fgmres.SetPrintLevel(1);
fgmres.Mult(B, U);
}
// 13. Recover the parallel grid function corresponding to U. This is the
// local finite element solution on each processor.
a->RecoverFEMSolution(U, b, u);
if (exact_sol)
{
double err_r = -1.0;
double err_i = -1.0;
switch (prob)
{
case 0:
err_r = u.real().ComputeL2Error(u0_r);
err_i = u.imag().ComputeL2Error(u0_i);
break;
case 1:
err_r = u.real().ComputeL2Error(u1_r);
err_i = u.imag().ComputeL2Error(u1_i);
break;
case 2:
err_r = u.real().ComputeL2Error(u2_r);
err_i = u.imag().ComputeL2Error(u2_i);
break;
default: break; // This should be unreachable
}
if ( myid == 0 )
{
cout << endl;
cout << "|| Re (u_h - u) ||_{L^2} = " << err_r << endl;
cout << "|| Im (u_h - u) ||_{L^2} = " << err_i << endl;
cout << endl;
}
}
// 14. Save the refined mesh and the solution in parallel. This output can be
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
{
ostringstream mesh_name, sol_r_name, sol_i_name;
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
sol_r_name << "sol_r." << setfill('0') << setw(6) << myid;
sol_i_name << "sol_i." << setfill('0') << setw(6) << myid;
ofstream mesh_ofs(mesh_name.str().c_str());
mesh_ofs.precision(8);
pmesh->Print(mesh_ofs);
ofstream sol_r_ofs(sol_r_name.str().c_str());
ofstream sol_i_ofs(sol_i_name.str().c_str());
sol_r_ofs.precision(8);
sol_i_ofs.precision(8);
u.real().Save(sol_r_ofs);
u.imag().Save(sol_i_ofs);
}
// 15. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *pmesh << u.real()
<< "window_title 'Solution: Real Part'" << flush;
sol_sock_i << "solution\n" << *pmesh << u.imag()
<< "window_title 'Solution: Imaginary Part'" << flush;
}
if (visualization && exact_sol)
{
*u_exact -= u;
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *pmesh << u_exact->real()
<< "window_title 'Error: Real Part'" << flush;
sol_sock_i << "solution\n" << *pmesh << u_exact->imag()
<< "window_title 'Error: Imaginary Part'" << flush;
}
if (visualization)
{
ParGridFunction u_t(fespace);
u_t = u.real();
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock.precision(8);
sol_sock << "solution\n" << *pmesh << u_t
<< "window_title 'Harmonic Solution (t = 0.0 T)'"
<< "pause\n" << flush;
if (myid == 0)
cout << "GLVis visualization paused."
<< " Press space (in the GLVis window) to resume it.\n";
int num_frames = 32;
int i = 0;
while (sol_sock)
{
double t = (double)(i % num_frames) / num_frames;
ostringstream oss;
oss << "Harmonic Solution (t = " << t << " T)";
add(cos( 2.0 * M_PI * t), u.real(),
sin(-2.0 * M_PI * t), u.imag(), u_t);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock << "solution\n" << *pmesh << u_t
<< "window_title '" << oss.str() << "'" << flush;
i++;
}
}
// 16. Free the used memory.
delete a;
delete u_exact;
delete pcOp;
delete fespace;
delete fec;
delete pmesh;
MPI_Finalize();
return 0;
}
bool check_for_inline_mesh(const char * mesh_file)
{
string file(mesh_file);
size_t p0 = file.find_last_of("/");
string s0 = file.substr((p0==string::npos)?0:(p0+1),7);
return s0 == "inline-";
}
complex<double> u0_exact(const Vector &x)
{
int dim = x.Size();
complex<double> i(0.0, 1.0);
complex<double> alpha = (epsilon_ * omega_ - i * sigma_);
complex<double> kappa = std::sqrt(mu_ * omega_* alpha);
return std::exp(-i * kappa * x[dim - 1]);
}
double u0_real_exact(const Vector &x)
{
return u0_exact(x).real();
}
double u0_imag_exact(const Vector &x)
{
return u0_exact(x).imag();
}
void u1_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_real_exact(x);
}
void u1_imag_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_imag_exact(x);
}
void u2_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_real_exact(x);
}
void u2_imag_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_imag_exact(x);
}
+3 -12
View File
@@ -21,7 +21,7 @@
//
// The example demonstrates the use of the BlockMatrix class, as
// well as the collective saving of several grid functions in a
// VisIt (visit.llnl.gov) and ParaView (paraview.org) formats.
// VisIt (visit.llnl.gov) visualization format.
//
// We recommend viewing examples 1-4 before viewing this example.
@@ -264,16 +264,7 @@ int main(int argc, char *argv[])
visit_dc.RegisterField("pressure", &p);
visit_dc.Save();
// 14. Save data in the ParaView format
ParaViewDataCollection paraview_dc("PVExample5S", mesh);
paraview_dc.SetLevelsOfDetail(2);
paraview_dc.SetCycle(1);
paraview_dc.SetTime(0.0); // set the time
paraview_dc.RegisterField("velocity",&u);
paraview_dc.RegisterField("pressure",&p);
paraview_dc.Save();
// 15. Send the solution by socket to a GLVis server.
// 14. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
@@ -286,7 +277,7 @@ int main(int argc, char *argv[])
p_sock << "solution\n" << *mesh << p << "window_title 'Pressure'" << endl;
}
// 16. Free the used memory.
// 15. Free the used memory.
delete fform;
delete gform;
delete invM;
+4 -12
View File
@@ -21,7 +21,7 @@
//
// The example demonstrates the use of the BlockMatrix class, as
// well as the collective saving of several grid functions in a
// VisIt (visit.llnl.gov) and ParaView (paraview.org) formats.
// VisIt (visit.llnl.gov) visualization format.
//
// We recommend viewing examples 1-4 before viewing this example.
@@ -239,6 +239,7 @@ int main(int argc, char *argv[])
// 12. Solve the linear system with MINRES.
// Check the norm of the unpreconditioned residual.
int maxIter(500);
double rtol(1.e-6);
double atol(1.e-10);
@@ -325,16 +326,7 @@ int main(int argc, char *argv[])
DataCollection::PARALLEL_FORMAT);
visit_dc.Save();
// 16. Save data in the ParaView format
ParaViewDataCollection paraview_dc("PVExample5P", pmesh);
paraview_dc.SetLevelsOfDetail(1);
paraview_dc.SetCycle(1);
paraview_dc.SetTime(0.0);
paraview_dc.RegisterField("velocity",u);
paraview_dc.RegisterField("pressure",p);
paraview_dc.Save();
// 17. Send the solution by socket to a GLVis server.
// 16. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
@@ -354,7 +346,7 @@ int main(int argc, char *argv[])
<< endl;
}
// 18. Free the used memory.
// 17. Free the used memory.
delete fform;
delete gform;
delete u;
-2
View File
@@ -19,8 +19,6 @@
// ex6 -pa -d cuda
// ex6 -pa -d occa-cuda
// ex6 -pa -d raja-omp
// ex6 -pa -d ceed-cpu
// ex6 -pa -d ceed-cuda
//
// Description: This is a version of Example 1 with a simple adaptive mesh
// refinement loop. The problem being solved is again the Laplace
-2
View File
@@ -19,8 +19,6 @@
// mpirun -np 4 ex6p -pa -d cuda
// mpirun -np 4 ex6p -pa -d occa-cuda
// mpirun -np 4 ex6p -pa -d raja-omp
// mpirun -np 4 ex6p -pa -d ceed-cpu
// mpirun -np 4 ex6p -pa -d ceed-cuda
//
// Description: This is a version of Example 1 with a simple adaptive mesh
// refinement loop. The problem being solved is again the Laplace
+11 -109
View File
@@ -9,7 +9,6 @@
// ex9 -m ../data/periodic-square.mesh -p 1 -r 2 -dt 0.005 -tf 9
// ex9 -m ../data/periodic-hexagon.mesh -p 1 -r 2 -dt 0.005 -tf 9
// ex9 -m ../data/amr-quad.mesh -p 1 -r 2 -dt 0.002 -tf 9
// ex9 -m ../data/amr-quad.mesh -p 1 -r 2 -dt 0.02 -s 13 -tf 9
// ex9 -m ../data/star-q3.mesh -p 1 -r 2 -dt 0.005 -tf 9
// ex9 -m ../data/star-mixed.mesh -p 1 -r 2 -dt 0.005 -tf 9
// ex9 -m ../data/disc-nurbs.mesh -p 1 -r 3 -dt 0.005 -tf 9
@@ -22,13 +21,12 @@
// u0(x)=u(0,x) is a given initial condition.
//
// The example demonstrates the use of Discontinuous Galerkin (DG)
// bilinear forms in MFEM (face integrators), the use of implicit
// and explicit ODE time integrators, the definition of periodic
// boundary conditions through periodic meshes, as well as the use
// of GLVis for persistent visualization of a time-evolving
// solution. The saving of time-dependent data files for external
// visualization with VisIt (visit.llnl.gov) and ParaView
// (paraview.org) is also illustrated.
// bilinear forms in MFEM (face integrators), the use of explicit
// ODE time integrators, the definition of periodic boundary
// conditions through periodic meshes, as well as the use of GLVis
// for persistent visualization of a time-evolving solution. The
// saving of time-dependent data files for external visualization
// with VisIt (visit.llnl.gov) is also illustrated.
#include "mfem.hpp"
#include <fstream>
@@ -54,54 +52,6 @@ double inflow_function(const Vector &x);
// Mesh bounding box
Vector bb_min, bb_max;
class DG_Solver : public Solver
{
private:
SparseMatrix &M, &K, A;
GMRESSolver linear_solver;
BlockILU prec;
double dt;
public:
DG_Solver(SparseMatrix &M_, SparseMatrix &K_, const FiniteElementSpace &fes)
: M(M_),
K(K_),
prec(fes.GetFE(0)->GetDof(),
BlockILU::Reordering::MINIMUM_DISCARDED_FILL),
dt(-1.0)
{
linear_solver.iterative_mode = false;
linear_solver.SetRelTol(1e-9);
linear_solver.SetAbsTol(0.0);
linear_solver.SetMaxIter(100);
linear_solver.SetPrintLevel(0);
linear_solver.SetPreconditioner(prec);
}
void SetTimeStep(double dt_)
{
if (dt_ != dt)
{
dt = dt_;
// Form operator A = M - dt*K
A = K;
A *= -dt;
A += M;
// this will also call SetOperator on the preconditioner
linear_solver.SetOperator(A);
}
}
void SetOperator(const Operator &op)
{
linear_solver.SetOperator(op);
}
virtual void Mult(const Vector &x, Vector &y) const
{
linear_solver.Mult(x, y);
}
};
/** A time-dependent operator for the right-hand side of the ODE. The DG weak
form of du/dt = -v.grad(u) is M du/dt = K u + b, where M and K are the mass
@@ -115,16 +65,13 @@ private:
const Vector &b;
DSmoother M_prec;
CGSolver M_solver;
DG_Solver dg_solver;
mutable Vector z;
public:
FE_Evolution(SparseMatrix &_M, SparseMatrix &_K, const Vector &_b,
const FiniteElementSpace &fes);
FE_Evolution(SparseMatrix &_M, SparseMatrix &_K, const Vector &_b);
virtual void Mult(const Vector &x, Vector &y) const;
virtual void ImplicitSolve(const double dt, const Vector &x, Vector &k);
virtual ~FE_Evolution() { }
};
@@ -142,7 +89,6 @@ int main(int argc, char *argv[])
double dt = 0.01;
bool visualization = true;
bool visit = false;
bool paraview = false;
bool binary = false;
int vis_steps = 5;
@@ -160,11 +106,7 @@ int main(int argc, char *argv[])
"Order (degree) of the finite elements.");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
"ODE solver: 1 - Forward Euler,\n\t"
" 2 - RK2 SSP, 3 - RK3 SSP, 4 - RK4, 6 - RK6,\n\t"
" 11 - Backward Euler,\n\t"
" 12 - SDIRK23 (L-stable), 13 - SDIRK33,\n\t"
" 22 - Implicit Midpoint Method,\n\t"
" 23 - SDIRK23 (A-stable), 24 - SDIRK34");
" 2 - RK2 SSP, 3 - RK3 SSP, 4 - RK4, 6 - RK6.");
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
@@ -175,9 +117,6 @@ int main(int argc, char *argv[])
args.AddOption(&visit, "-visit", "--visit-datafiles", "-no-visit",
"--no-visit-datafiles",
"Save data files for VisIt (visit.llnl.gov) visualization.");
args.AddOption(&paraview, "-paraview", "--paraview-datafiles", "-no-paraview",
"--no-paraview-datafiles",
"Save data files for ParaView (paraview.org) visualization.");
args.AddOption(&binary, "-binary", "--binary-datafiles", "-ascii",
"--ascii-datafiles",
"Use binary (Sidre) or ascii format for VisIt data files.");
@@ -201,21 +140,11 @@ int main(int argc, char *argv[])
ODESolver *ode_solver = NULL;
switch (ode_solver_type)
{
// Explicit methods
case 1: ode_solver = new ForwardEulerSolver; break;
case 2: ode_solver = new RK2Solver(1.0); break;
case 3: ode_solver = new RK3SSPSolver; break;
case 4: ode_solver = new RK4Solver; break;
case 6: ode_solver = new RK6Solver; break;
// Implicit (L-stable) methods
case 11: ode_solver = new BackwardEulerSolver; break;
case 12: ode_solver = new SDIRK23Solver(2); break;
case 13: ode_solver = new SDIRK33Solver; break;
// Implicit A-stable methods (not L-stable)
case 22: ode_solver = new ImplicitMidpointSolver; break;
case 23: ode_solver = new SDIRK23Solver; break;
case 24: ode_solver = new SDIRK34Solver; break;
default:
cout << "Unknown ODE solver type: " << ode_solver_type << '\n';
return 3;
@@ -308,16 +237,6 @@ int main(int argc, char *argv[])
dc->Save();
}
ParaViewDataCollection *pd = NULL;
if (paraview)
{
pd = new ParaViewDataCollection("PVExample9S", &mesh);
pd->RegisterField("solution", &u);
pd->SetLevelsOfDetail(2);
pd->SetCycle(0);
pd->SetTime(0.0);
}
socketstream sout;
if (visualization)
{
@@ -345,7 +264,7 @@ int main(int argc, char *argv[])
// 8. Define the time-dependent evolution operator describing the ODE
// right-hand side, and perform time-integration (looping over the time
// iterations, ti, with a time-step dt).
FE_Evolution adv(m.SpMat(), k.SpMat(), b, fes);
FE_Evolution adv(m.SpMat(), k.SpMat(), b);
double t = 0.0;
adv.SetTime(t);
@@ -375,13 +294,6 @@ int main(int argc, char *argv[])
dc->SetTime(t);
dc->Save();
}
if (paraview)
{
pd->SetCycle(ti);
pd->SetTime(t);
pd->Save();
}
}
}
@@ -395,7 +307,6 @@ int main(int argc, char *argv[])
// 10. Free the used memory.
delete ode_solver;
delete pd;
delete dc;
return 0;
@@ -403,10 +314,8 @@ int main(int argc, char *argv[])
// Implementation of class FE_Evolution
FE_Evolution::FE_Evolution(SparseMatrix &_M, SparseMatrix &_K, const Vector &_b,
const FiniteElementSpace &fes)
: TimeDependentOperator(_M.Size()), M(_M), K(_K), b(_b), dg_solver(M, K, fes),
z(_M.Size())
FE_Evolution::FE_Evolution(SparseMatrix &_M, SparseMatrix &_K, const Vector &_b)
: TimeDependentOperator(_M.Size()), M(_M), K(_K), b(_b), z(_M.Size())
{
M_solver.SetPreconditioner(M_prec);
M_solver.SetOperator(M);
@@ -426,13 +335,6 @@ void FE_Evolution::Mult(const Vector &x, Vector &y) const
M_solver.Mult(z, y);
}
void FE_Evolution::ImplicitSolve(const double dt, const Vector &x, Vector &k)
{
K.Mult(x, z);
z += b;
dg_solver.SetTimeStep(dt);
dg_solver.Mult(z, k);
}
// Velocity coefficient
void velocity_function(const Vector &x, Vector &v)
+11 -121
View File
@@ -9,7 +9,6 @@
// mpirun -np 4 ex9p -m ../data/periodic-square.mesh -p 1 -dt 0.005 -tf 9
// mpirun -np 4 ex9p -m ../data/periodic-hexagon.mesh -p 1 -dt 0.005 -tf 9
// mpirun -np 4 ex9p -m ../data/amr-quad.mesh -p 1 -rp 1 -dt 0.002 -tf 9
// mpirun -np 4 ex9p -m ../data/amr-quad.mesh -p 1 -rp 1 -dt 0.02 -s 13 -tf 9
// mpirun -np 4 ex9p -m ../data/star-q3.mesh -p 1 -rp 1 -dt 0.004 -tf 9
// mpirun -np 4 ex9p -m ../data/star-mixed.mesh -p 1 -rp 1 -dt 0.004 -tf 9
// mpirun -np 4 ex9p -m ../data/disc-nurbs.mesh -p 1 -rp 1 -dt 0.005 -tf 9
@@ -22,13 +21,12 @@
// u0(x)=u(0,x) is a given initial condition.
//
// The example demonstrates the use of Discontinuous Galerkin (DG)
// bilinear forms in MFEM (face integrators), the use of implicit
// and explicit ODE time integrators, the definition of periodic
// boundary conditions through periodic meshes, as well as the use
// of GLVis for persistent visualization of a time-evolving
// solution. The saving of time-dependent data files for external
// visualization with VisIt (visit.llnl.gov) and ParaView
// (paraview.org) is also illustrated.
// bilinear forms in MFEM (face integrators), the use of explicit
// ODE time integrators, the definition of periodic boundary
// conditions through periodic meshes, as well as the use of GLVis
// for persistent visualization of a time-evolving solution. The
// saving of time-dependent data files for external visualization
// with VisIt (visit.llnl.gov) is also illustrated.
#include "mfem.hpp"
#include <fstream>
@@ -53,66 +51,6 @@ double inflow_function(const Vector &x);
// Mesh bounding box
Vector bb_min, bb_max;
class DG_Solver : public Solver
{
private:
HypreParMatrix &M, &K;
SparseMatrix M_diag;
HypreParMatrix *A;
GMRESSolver linear_solver;
BlockILU prec;
double dt;
public:
DG_Solver(HypreParMatrix &M_, HypreParMatrix &K_, const FiniteElementSpace &fes)
: M(M_),
K(K_),
A(NULL),
linear_solver(M.GetComm()),
prec(fes.GetFE(0)->GetDof(),
BlockILU::Reordering::MINIMUM_DISCARDED_FILL),
dt(-1.0)
{
linear_solver.iterative_mode = false;
linear_solver.SetRelTol(1e-9);
linear_solver.SetAbsTol(0.0);
linear_solver.SetMaxIter(100);
linear_solver.SetPrintLevel(0);
linear_solver.SetPreconditioner(prec);
M.GetDiag(M_diag);
}
void SetTimeStep(double dt_)
{
if (dt_ != dt)
{
dt = dt_;
// Form operator A = M - dt*K
delete A;
A = Add(-dt, K, 0.0, K);
SparseMatrix A_diag;
A->GetDiag(A_diag);
A_diag.Add(1.0, M_diag);
// this will also call SetOperator on the preconditioner
linear_solver.SetOperator(*A);
}
}
void SetOperator(const Operator &op)
{
linear_solver.SetOperator(op);
}
virtual void Mult(const Vector &x, Vector &y) const
{
linear_solver.Mult(x, y);
}
~DG_Solver()
{
delete A;
}
};
/** A time-dependent operator for the right-hand side of the ODE. The DG weak
form of du/dt = -v.grad(u) is M du/dt = K u + b, where M and K are the mass
@@ -126,16 +64,13 @@ private:
const Vector &b;
HypreSmoother M_prec;
CGSolver M_solver;
DG_Solver dg_solver;
mutable Vector z;
public:
FE_Evolution(HypreParMatrix &_M, HypreParMatrix &_K, const Vector &_b,
const FiniteElementSpace &fes);
FE_Evolution(HypreParMatrix &_M, HypreParMatrix &_K, const Vector &_b);
virtual void Mult(const Vector &x, Vector &y) const;
virtual void ImplicitSolve(const double dt, const Vector &x, Vector &k);
virtual ~FE_Evolution() { }
};
@@ -160,7 +95,6 @@ int main(int argc, char *argv[])
double dt = 0.01;
bool visualization = true;
bool visit = false;
bool paraview = false;
bool binary = false;
int vis_steps = 5;
@@ -180,11 +114,7 @@ int main(int argc, char *argv[])
"Order (degree) of the finite elements.");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
"ODE solver: 1 - Forward Euler,\n\t"
" 2 - RK2 SSP, 3 - RK3 SSP, 4 - RK4, 6 - RK6,\n\t"
" 11 - Backward Euler,\n\t"
" 12 - SDIRK23 (L-stable), 13 - SDIRK33,\n\t"
" 22 - Implicit Midpoint Method,\n\t"
" 23 - SDIRK23 (A-stable), 24 - SDIRK34");
" 2 - RK2 SSP, 3 - RK3 SSP, 4 - RK4, 6 - RK6.");
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
@@ -195,9 +125,6 @@ int main(int argc, char *argv[])
args.AddOption(&visit, "-visit", "--visit-datafiles", "-no-visit",
"--no-visit-datafiles",
"Save data files for VisIt (visit.llnl.gov) visualization.");
args.AddOption(&paraview, "-paraview", "--paraview-datafiles", "-no-paraview",
"--no-paraview-datafiles",
"Save data files for ParaView (paraview.org) visualization.");
args.AddOption(&binary, "-binary", "--binary-datafiles", "-ascii",
"--ascii-datafiles",
"Use binary (Sidre) or ascii format for VisIt data files.");
@@ -228,20 +155,11 @@ int main(int argc, char *argv[])
ODESolver *ode_solver = NULL;
switch (ode_solver_type)
{
// Explicit methods
case 1: ode_solver = new ForwardEulerSolver; break;
case 2: ode_solver = new RK2Solver(1.0); break;
case 3: ode_solver = new RK3SSPSolver; break;
case 4: ode_solver = new RK4Solver; break;
case 6: ode_solver = new RK6Solver; break;
// Implicit (L-stable) methods
case 11: ode_solver = new BackwardEulerSolver; break;
case 12: ode_solver = new SDIRK23Solver(2); break;
case 13: ode_solver = new SDIRK33Solver; break;
// Implicit A-stable methods (not L-stable)
case 22: ode_solver = new ImplicitMidpointSolver; break;
case 23: ode_solver = new SDIRK23Solver; break;
case 24: ode_solver = new SDIRK34Solver; break;
default:
if (myid == 0)
{
@@ -363,17 +281,6 @@ int main(int argc, char *argv[])
dc->Save();
}
ParaViewDataCollection *pd = NULL;
if (paraview)
{
pd = new ParaViewDataCollection("PVExample9P", pmesh);
pd->RegisterField("solution", u);
pd->SetLevelsOfDetail(2);
pd->SetCycle(0);
pd->SetTime(0.0);
pd->Save();
}
socketstream sout;
if (visualization)
{
@@ -407,7 +314,7 @@ int main(int argc, char *argv[])
// 10. Define the time-dependent evolution operator describing the ODE
// right-hand side, and perform time-integration (looping over the time
// iterations, ti, with a time-step dt).
FE_Evolution adv(*M, *K, *B, *fes);
FE_Evolution adv(*M, *K, *B);
double t = 0.0;
adv.SetTime(t);
@@ -445,13 +352,6 @@ int main(int argc, char *argv[])
dc->SetTime(t);
dc->Save();
}
if (paraview)
{
pd->SetCycle(ti);
pd->SetTime(t);
pd->Save();
}
}
}
@@ -478,7 +378,6 @@ int main(int argc, char *argv[])
delete fes;
delete pmesh;
delete ode_solver;
delete pd;
delete dc;
MPI_Finalize();
@@ -488,10 +387,9 @@ int main(int argc, char *argv[])
// Implementation of class FE_Evolution
FE_Evolution::FE_Evolution(HypreParMatrix &_M, HypreParMatrix &_K,
const Vector &_b, const FiniteElementSpace &fes)
const Vector &_b)
: TimeDependentOperator(_M.Height()),
M(_M), K(_K), b(_b), M_solver(M.GetComm()),
dg_solver(M, K, fes), z(_M.Height())
M(_M), K(_K), b(_b), M_solver(M.GetComm()), z(_M.Height())
{
M_prec.SetType(HypreSmoother::Jacobi);
M_solver.SetPreconditioner(M_prec);
@@ -504,14 +402,6 @@ FE_Evolution::FE_Evolution(HypreParMatrix &_M, HypreParMatrix &_K,
M_solver.SetPrintLevel(0);
}
void FE_Evolution::ImplicitSolve(const double dt, const Vector &x, Vector &k)
{
K.Mult(x, z);
z += b;
dg_solver.SetTimeStep(dt);
dg_solver.Mult(z, k);
}
void FE_Evolution::Mult(const Vector &x, Vector &y) const
{
// y = M^{-1} (K x + b)
-59
View File
@@ -1,59 +0,0 @@
# Copyright (c) 2019, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
set(GINKGO_EXAMPLES_SRCS)
list(APPEND GINKGO_EXAMPLES_SRCS
ex1.cpp
)
# Include the source directory where mfem.hpp and mfem-performance.hpp are.
include_directories(BEFORE ${PROJECT_BINARY_DIR})
# Add "test_ginkgo" target, see below.
add_custom_target(test_ginkgo
${CMAKE_CTEST_COMMAND} -R ginkgo USES_TERMINAL)
# Add one executable per cpp file, adding "ginkgo_" as prefix. Sets
# "test_ginkgo" as a target that depends on the given examples.
set(PFX ginkgo_)
add_mfem_examples(GINKGO_EXAMPLES_SRCS ${PFX} "" test_ginkgo)
# Testing.
# The GINKGO tests can be run separately using the target "test_ginkgo"
# which builds the examples and runs:
# ctest -R ginkgo
# Command line options for the tests.
set(EX1_COMMON_OPTS ex1 -m ../data/star.mesh --use_gko_solver)
set(EX1_TEST_OPTS ${EX9_COMMON_OPTS})
# Add the tests: one test per source file.
foreach(SRC_FILE ${GINKGO_EXAMPLES_SRCS})
get_filename_component(SRC_FILENAME ${SRC_FILE} NAME)
string(REPLACE ".cpp" "" TEST_NAME ${SRC_FILENAME})
string(TOUPPER ${TEST_NAME} UP_TEST_NAME)
set(TEST_NAME ${PFX}${TEST_NAME})
set(THIS_TEST_OPTIONS "-no-vis")
list(APPEND THIS_TEST_OPTIONS ${${UP_TEST_NAME}_TEST_OPTS})
# message(STATUS "Test ${TEST_NAME} options: ${THIS_TEST_OPTIONS}")
if (NOT (${TEST_NAME} MATCHES ".*p$"))
add_test(NAME ${TEST_NAME}_ser
COMMAND ${TEST_NAME} ${THIS_TEST_OPTIONS})
else()
add_test(NAME ${TEST_NAME}_np=4
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} ${MFEM_MPI_NP}
${MPIEXEC_PREFLAGS}
$<TARGET_FILE:${TEST_NAME}> ${THIS_TEST_OPTIONS}
${MPIEXEC_POSTFLAGS})
endif()
endforeach()
-259
View File
@@ -1,259 +0,0 @@
// MFEM Example 1
// GINKGO Modification
//
// Compile with: make ex1
//
// Sample runs: ex1 -m ../data/square-disc.mesh
// ex1 -m ../data/star.mesh
// ex1 -m ../data/star-mixed.mesh
// ex1 -m ../data/escher.mesh
// ex1 -m ../data/fichera.mesh
// ex1 -m ../data/fichera-mixed.mesh
// ex1 -m ../data/toroid-wedge.mesh
// ex1 -m ../data/square-disc-p2.vtk -o 2
// ex1 -m ../data/square-disc-p3.mesh -o 3
// ex1 -m ../data/square-disc-nurbs.mesh -o -1
// ex1 -m ../data/star-mixed-p2.mesh -o 2
// ex1 -m ../data/disc-nurbs.mesh -o -1
// ex1 -m ../data/pipe-nurbs.mesh -o -1
// ex1 -m ../data/fichera-mixed-p2.mesh -o 2
// ex1 -m ../data/star-surf.mesh
// ex1 -m ../data/square-disc-surf.mesh
// ex1 -m ../data/inline-segment.mesh
// ex1 -m ../data/amr-quad.mesh
// ex1 -m ../data/amr-hex.mesh
// ex1 -m ../data/fichera-amr.mesh
// ex1 -m ../data/mobius-strip.mesh
// ex1 -m ../data/mobius-strip.mesh -o -1 -sc
//
// Device sample runs:
// ex1 -pa -d cuda
// ex1 -pa -d raja-cuda
// ex1 -pa -d occa-cuda
// ex1 -pa -d raja-omp
// ex1 -pa -d occa-omp
// ex1 -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Laplace problem
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
// Specifically, we discretize using a FE space of the specified
// order, or if order < 1 using an isoparametric/isogeometric
// space (i.e. quadratic for quadratic curvilinear mesh, NURBS for
// NURBS mesh, etc.)
//
// The example highlights the use of mesh refinement, finite
// element grid functions, as well as linear and bilinear forms
// corresponding to the left-hand side and right-hand side of the
// discrete linear system. We also cover the explicit elimination
// of essential boundary conditions, static condensation, and the
// optional connection to the GLVis tool for visualization.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
#ifndef MFEM_USE_GINKGO
#error This example requires that MFEM is built with MFEM_USE_GINKGO=YES
#endif
using namespace std;
using namespace mfem;
int main(int argc, char *argv[])
{
// 1. Parse command-line options.
const char *mesh_file = "../../data/star.mesh";
int order = 1;
bool static_cond = false;
bool pa = false;
const char *device_config = "cpu";
bool visualization = true;
bool use_ginkgo_solver= true;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree) or -1 for"
" isoparametric space.");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&use_ginkgo_solver, "-gko", "--use_gko_solver", "-no-gko",
"--no-gko-solver",
"Solve using ginkgo.");
args.Parse();
if (!args.Good())
{
args.PrintUsage(cout);
return 1;
}
args.PrintOptions(cout);
// 2. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
device.Print();
// 3. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes with
// the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement. We choose 'ref_levels' to be the
// largest number that gives a final mesh with no more than 50,000
// elements.
{
int ref_levels =
(int)floor(log(50000./mesh->GetNE())/log(2.)/dim);
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
}
}
// 5. Define a finite element space on the mesh. Here we use continuous
// Lagrange finite elements of the specified order. If order < 1, we
// instead use an isoparametric/isogeometric space.
FiniteElementCollection *fec;
if (order > 0)
{
fec = new H1_FECollection(order, dim);
}
else if (mesh->GetNodes())
{
fec = mesh->GetNodes()->OwnFEC();
cout << "Using isoparametric FEs: " << fec->Name() << endl;
}
else
{
fec = new H1_FECollection(order = 1, dim);
}
FiniteElementSpace *fespace = new FiniteElementSpace(mesh, fec);
cout << "Number of finite element unknowns: "
<< fespace->GetTrueVSize() << endl;
// 6. Determine the list of true (i.e. conforming) essential boundary dofs.
// In this example, the boundary conditions are defined by marking all
// the boundary attributes from the mesh as essential (Dirichlet) and
// converting them to a list of true dofs.
Array<int> ess_tdof_list;
if (mesh->bdr_attributes.Size())
{
Array<int> ess_bdr(mesh->bdr_attributes.Max());
ess_bdr = 1;
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 7. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system, which in this case is (1,phi_i) where phi_i are
// the basis functions in the finite element fespace.
LinearForm *b = new LinearForm(fespace);
ConstantCoefficient one(1.0);
b->AddDomainIntegrator(new DomainLFIntegrator(one));
b->Assemble();
// 8. Define the solution vector x as a finite element grid function
// corresponding to fespace. Initialize x with initial guess of zero,
// which satisfies the boundary conditions.
GridFunction x(fespace);
x = 0.0;
// 9. Set up the bilinear form a(.,.) on the finite element space
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
// domain integrator.
BilinearForm *a = new BilinearForm(fespace);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
a->AddDomainIntegrator(new DiffusionIntegrator(one));
// 10. Assemble the bilinear form and the corresponding linear system,
// applying any necessary transformations such as: eliminating boundary
// conditions, applying conforming constraints for non-conforming AMR,
// static condensation, etc.
if (static_cond) { a->EnableStaticCondensation(); }
a->Assemble();
OperatorPtr A;
Vector B, X;
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
cout << "Size of linear system: " << A->Height() << endl;
// 11. Solve the linear system A X = B.
if (!pa)
{
if (use_ginkgo_solver)
{
#ifdef MFEM_USE_GINKGO
// Solve the linear system with CG + ILU from Ginkgo.
std::string executor = "reference";
auto exec = gko::ReferenceExecutor::create();
auto ilu_precond =
gko::preconditioner::Ilu<gko::solver::LowerTrs<>,
gko::solver::UpperTrs<>, false>::build()
.on(exec);
GinkgoWrappers::CGSolver ginkgo_solver(executor, 1, 2000, 1e-12, 0.0,
ilu_precond.release() );
ginkgo_solver.solve(&((SparseMatrix&)(*A)), X, B);
#endif
}
else
{
#ifndef MFEM_USE_SUITESPARSE
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
GSSmoother M((SparseMatrix&)(*A));
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
#else
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
UMFPackSolver umf_solver;
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
umf_solver.SetOperator(*A);
umf_solver.Mult(B, X);
#endif
}
}
else // No preconditioning for now in partial assembly mode.
{
CG(*A, B, X, 1, 2000, 1e-12, 0.0);
}
// 12. Recover the solution as a finite element grid function.
a->RecoverFEMSolution(X, *b, x);
// 13. Save the refined mesh and the solution. This output can be viewed later
// using GLVis: "glvis -m refined.mesh -g sol.gf".
ofstream mesh_ofs("refined.mesh");
mesh_ofs.precision(8);
mesh->Print(mesh_ofs);
ofstream sol_ofs("sol.gf");
sol_ofs.precision(8);
x.Save(sol_ofs);
// 14. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "solution\n" << *mesh << x << flush;
}
// 15. Free the used memory.
delete a;
delete b;
delete fespace;
if (order > 0) { delete fec; }
delete mesh;
return 0;
}
-81
View File
@@ -1,81 +0,0 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/ginkgo/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
# Currently there are only serial Ginkgo examples
SEQ_EXAMPLES = ex1
PAR_EXAMPLES =
ifeq ($(MFEM_USE_MPI),NO)
EXAMPLES = $(SEQ_EXAMPLES)
else
EXAMPLES = $(PAR_EXAMPLES) $(SEQ_EXAMPLES)
endif
.SUFFIXES:
.SUFFIXES: .o .cpp .mk
.PHONY: all clean clean-build clean-exec
# Remove built-in rule
%: %.cpp
# Replace the default implicit rule for *.cpp files
%: $(SRC)%.cpp $(MFEM_LIB_FILE) $(CONFIG_MK)
$(MFEM_CXX) $(MFEM_FLAGS) $< -o $@ $(MFEM_LIBS)
all: $(EXAMPLES)
ifeq ($(MFEM_USE_GINKGO),NO)
$(EXAMPLES):
$(error MFEM is not configured with GINKO)
endif
MFEM_TESTS = EXAMPLES
include $(MFEM_TEST_MK)
# Testing: Parallel vs. serial runs
RUN_MPI_NP = $(MFEM_MPIEXEC) $(MFEM_MPIEXEC_NP)
RUN_MPI = $(RUN_MPI_NP) $(MFEM_MPI_NP)
SERIAL_NAME := Serial GINKGO example
PARALLEL_NAME := Parallel GINKGO example
%-test-par: %
@$(call mfem-test,$<, $(RUN_MPI), $(PARALLEL_NAME))
%-test-seq: %
@$(call mfem-test,$<,, $(SERIAL_NAME))
# Testing: Example-specific execution options:
ex1-test-par: ex1
@$(call mfem-test,$<, $(RUN_MPI_NP) 1, $(PARALLEL_NAME))
# Testing: "test" target and mfem-test* variables are defined in config/test.mk
# Generate an error message if the MFEM library is not built and exit
$(MFEM_LIB_FILE):
$(error The MFEM library is not built)
clean: clean-build clean-exec
clean-build:
rm -f *.o *~ $(SEQ_EXAMPLES) $(PAR_EXAMPLES)
rm -rf *.dSYM *.TVD.*breakpoints
clean-exec:
@rm -f refined.mesh sol.gf mesh.* sol.*
-63
View File
@@ -1,63 +0,0 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
set(HIOP_EXAMPLES_SRCS)
list(APPEND HIOP_EXAMPLES_SRCS ex9.cpp)
if (MFEM_USE_MPI)
list(APPEND HIOP_EXAMPLES_SRCS ex9p.cpp)
endif()
# Include the source directory where mfem.hpp and mfem-performance.hpp are.
include_directories(BEFORE ${PROJECT_BINARY_DIR})
# Add "test_hiop" target, see below.
add_custom_target(test_hiop
${CMAKE_CTEST_COMMAND} -R hiop USES_TERMINAL)
# Add one executable per cpp file, adding "hiop_" as prefix. Sets
# "test_hiop" as a target that depends on the given examples.
set(PFX hiop_)
add_mfem_examples(HIOP_EXAMPLES_SRCS ${PFX} "" test_hiop)
# Testing.
# The HIOP tests can be run separately using the target "test_hiop"
# which builds the examples and runs:
# ctest -R hiop
# Command line options for the tests.
# Example 9:
set(EX9_COMMON_OPTS -m ../../data/periodic-segment.mesh -p 0 -dt 0.005)
set(EX9_TEST_OPTS ${EX9_COMMON_OPTS} -r 2 )
set(EX9P_TEST_OPTS ${EX9_COMMON_OPTS})
# Add the tests: one test per source file.
foreach(SRC_FILE ${HIOP_EXAMPLES_SRCS})
get_filename_component(SRC_FILENAME ${SRC_FILE} NAME)
string(REPLACE ".cpp" "" TEST_NAME ${SRC_FILENAME})
string(TOUPPER ${TEST_NAME} UP_TEST_NAME)
set(TEST_NAME ${PFX}${TEST_NAME})
set(THIS_TEST_OPTIONS "-no-vis")
list(APPEND THIS_TEST_OPTIONS ${${UP_TEST_NAME}_TEST_OPTS})
# message(STATUS "Test ${TEST_NAME} options: ${THIS_TEST_OPTIONS}")
if (NOT (${TEST_NAME} MATCHES ".*p$"))
add_test(NAME ${TEST_NAME}_ser
COMMAND ${TEST_NAME} ${THIS_TEST_OPTIONS})
else()
add_test(NAME ${TEST_NAME}_np=4
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} 4
${MPIEXEC_PREFLAGS}
$<TARGET_FILE:${TEST_NAME}> ${THIS_TEST_OPTIONS}
${MPIEXEC_POSTFLAGS})
endif()
endforeach()
-684
View File
@@ -1,684 +0,0 @@
// MFEM Example 9 with Nonlinear Constrained Optimization
//
// Compile with: make ex9
//
// Sample runs:
//
// ex9 -m ../../data/periodic-segment.mesh -r 3 -p 0 -o 2 -dt 0.002 -opt 1
// ex9 -m ../../data/periodic-segment.mesh -r 3 -p 0 -o 2 -dt 0.002 -opt 2
//
// ex9 -m ../../data/periodic-square.mesh -p 0 -r 2 -dt 0.01 -tf 10 -opt 1
// ex9 -m ../../data/periodic-square.mesh -p 0 -r 2 -dt 0.01 -tf 10 -opt 2
//
// ex9 -m ../../data/periodic-square.mesh -p 1 -r 2 -dt 0.005 -tf 9 -opt 1
// ex9 -m ../../data/periodic-square.mesh -p 1 -r 2 -dt 0.005 -tf 9 -opt 2
//
// ex9 -m ../../data/amr-quad.mesh -p 1 -r 1 -dt 0.002 -tf 9 -opt 1
// ex9 -m ../../data/amr-quad.mesh -p 1 -r 1 -dt 0.002 -tf 9 -opt 2
//
// ex9 -m ../../data/disc-nurbs.mesh -p 1 -r 2 -dt 0.005 -tf 9 -opt 1
// ex9 -m ../../data/disc-nurbs.mesh -p 1 -r 2 -dt 0.005 -tf 9 -opt 2
//
// ex9 -m ../../data/disc-nurbs.mesh -p 2 -r 2 -dt 0.01 -tf 9 -opt 1
// ex9 -m ../../data/disc-nurbs.mesh -p 2 -r 2 -dt 0.01 -tf 9 -opt 2
//
// ex9 -m ../../data/periodic-square.mesh -p 3 -r 3 -dt 0.0025 -tf 9 -opt 1
// ex9 -m ../../data/periodic-square.mesh -p 3 -r 3 -dt 0.0025 -tf 9 -opt 2
//
// ex9 -m ../../data/periodic-cube.mesh -p 0 -r 2 -o 2 -dt 0.02 -tf 8 -opt 1
// ex9 -m ../../data/periodic-cube.mesh -p 0 -r 2 -o 2 -dt 0.02 -tf 8 -opt 2
// Description: This example modifies the standard MFEM ex9 by adding nonlinear
// constrained optimization capabilities through the SLBQP and
// HIOP solvers. It demonstrates how a user can define a custom
// class OptimizationProblem that includes linear/nonlinear
// equality/inequality constraints. This optimization is applied
// as post-processing to the solution of the transport equation.
//
// Description of ex9:
// This example code solves the time-dependent advection equation
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
// u0(x)=u(0,x) is a given initial condition.
//
// The example demonstrates the use of Discontinuous Galerkin (DG)
// bilinear forms in MFEM (face integrators), the use of explicit
// ODE time integrators, the definition of periodic boundary
// conditions through periodic meshes, as well as the use of GLVis
// for persistent visualization of a time-evolving solution. The
// saving of time-dependent data files for external visualization
// with VisIt (visit.llnl.gov) is also illustrated.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
// Choice for the problem setup. The fluid velocity, initial condition and
// inflow boundary condition are chosen based on this parameter.
int problem;
// Nonlinear optimizer.
int optimizer_type;
// Velocity coefficient
bool invert_velocity = false;
void velocity_function(const Vector &x, Vector &v);
// Initial condition
double u0_function(const Vector &x);
// Inflow boundary condition
double inflow_function(const Vector &x);
// Mesh bounding box
Vector bb_min, bb_max;
/// Computes C(x) = sum w_i x_i, where w is a given Vector.
class LinearScaleOperator : public Operator
{
private:
const Vector &w;
mutable DenseMatrix grad;
public:
LinearScaleOperator(const Vector &weight)
: Operator(1, weight.Size()), w(weight), grad(1, width)
{
for (int i = 0; i < width; i++) { grad(0, i) = w(i); }
}
virtual void Mult(const Vector &x, Vector &y) const
{
y(0) = w * x;
}
virtual Operator &GetGradient(const Vector &x) const
{
return grad;
}
};
/// Nonlinear monotone bounded operator to test nonlinear ineq constraints.
/// Computes D(x) = tanh(sum(x_i)).
class TanhSumOperator : public Operator
{
private:
mutable DenseMatrix grad;
public:
TanhSumOperator(int size) : Operator(1, size), grad(1, width) { }
virtual void Mult(const Vector &x, Vector &y) const
{
y(0) = std::tanh(x.Sum());
}
virtual Operator &GetGradient(const Vector &x) const
{
const double ts = std::tanh(x.Sum());
const double dtanh = 1.0 - ts * ts;
for (int i = 0; i < width; i++) { grad(0, i) = dtanh; }
return grad;
}
};
/** Monotone and conservative a-posteriori correction for transport solutions:
* Find x that minimizes 0.5 || x - x_HO ||^2, subject to
* sum w_i x_i = mass,
* tanh(sum(x_i_min)) <= tanh(sum(x_i)) <= tanh(sum(x_i_max)),
* x_i_min <= x_i <= x_i_max,
*/
class OptimizedTransportProblem : public OptimizationProblem
{
private:
const Vector &x_HO;
Vector massvec, d_lo, d_hi;
const LinearScaleOperator LSoper;
const TanhSumOperator TSoper;
public:
OptimizedTransportProblem(const Vector &xho, const Vector &w, double mass,
const Vector &xmin, const Vector &xmax)
: OptimizationProblem(xho.Size(), NULL, NULL),
x_HO(xho), massvec(1), d_lo(1), d_hi(1),
LSoper(w), TSoper(w.Size())
{
C = &LSoper;
massvec(0) = mass;
SetEqualityConstraint(massvec);
D = &TSoper;
d_lo(0) = std::tanh(xmin.Sum());
d_hi(0) = std::tanh(xmax.Sum());
MFEM_ASSERT(d_lo(0) < d_hi(0),
"The bounds produce an infeasible optimization problem");
SetInequalityConstraint(d_lo, d_hi);
SetSolutionBounds(xmin, xmax);
}
virtual double CalcObjective(const Vector &x) const
{
double res = 0.0;
for (int i = 0; i < input_size; i++)
{
const double d = x(i) - x_HO(i);
res += d * d;
}
return 0.5 * res;
}
virtual void CalcObjectiveGrad(const Vector &x, Vector &grad) const
{
for (int i = 0; i < input_size; i++) { grad(i) = x(i) - x_HO(i); }
}
};
/** A time-dependent operator for the right-hand side of the ODE. The DG weak
form of du/dt = -v.grad(u) is M du/dt = K u + b, where M and K are the mass
and advection matrices, and b describes the flow on the boundary. This can
be written as a general ODE, du/dt = M^{-1} (K u + b), and this class is
used to evaluate the right-hand side. */
class FE_Evolution : public TimeDependentOperator
{
private:
SparseMatrix &M, &K;
const Vector &b;
DSmoother M_prec;
CGSolver M_solver;
mutable Vector z;
double dt;
BilinearForm &bf;
Vector &M_rowsums;
public:
FE_Evolution(SparseMatrix &_M, SparseMatrix &_K, const Vector &_b,
BilinearForm &_bf, Vector &M_rs);
void SetTimeStep(double _dt) { dt = _dt; }
void SetK(SparseMatrix &_K) { K = _K; }
virtual void Mult(const Vector &x, Vector &y) const;
virtual ~FE_Evolution() { }
};
int main(int argc, char *argv[])
{
// 1. Parse command-line options.
problem = 0;
optimizer_type = 1;
const char *mesh_file = "../../data/periodic-hexagon.mesh";
int ref_levels = 2;
int order = 3;
int ode_solver_type = 3;
double t_final = 1.0;
double dt = 0.01;
bool visualization = true;
bool visit = false;
bool binary = false;
int vis_steps = 5;
int precision = 8;
cout.precision(precision);
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&problem, "-p", "--problem",
"Problem setup to use. See options in velocity_function().");
args.AddOption(&ref_levels, "-r", "--refine",
"Number of times to refine the mesh uniformly.");
args.AddOption(&order, "-o", "--order",
"Order (degree) of the finite elements.");
args.AddOption(&optimizer_type, "-opt", "--optimizer",
"Nonlinear optimizer: 1 - SLBQP,\n\t"
" 2 - HIOP.");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
"ODE solver: 1 - Forward Euler,\n\t"
" 2 - RK2 SSP, 3 - RK3 SSP, 4 - RK4, 6 - RK6.");
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
"Time step.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&visit, "-visit", "--visit-datafiles", "-no-visit",
"--no-visit-datafiles",
"Save data files for VisIt (visit.llnl.gov) visualization.");
args.AddOption(&binary, "-binary", "--binary-datafiles", "-ascii",
"--ascii-datafiles",
"Use binary (Sidre) or ascii format for VisIt data files.");
args.AddOption(&vis_steps, "-vs", "--visualization-steps",
"Visualize every n-th timestep.");
args.Parse();
if (!args.Good())
{
args.PrintUsage(cout);
return 1;
}
args.PrintOptions(cout);
// 2. Read the mesh from the given mesh file. We can handle geometrically
// periodic meshes in this code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 3. Define the ODE solver used for time integration. Several explicit
// Runge-Kutta methods are available.
ODESolver *ode_solver = NULL;
switch (ode_solver_type)
{
case 1: ode_solver = new ForwardEulerSolver; break;
case 2: ode_solver = new RK2Solver(1.0); break;
case 3: ode_solver = new RK3SSPSolver; break;
case 4: ode_solver = new RK4Solver; break;
case 6: ode_solver = new RK6Solver; break;
default:
cout << "Unknown ODE solver type: " << ode_solver_type << '\n';
delete mesh;
return 3;
}
// 4. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement, where 'ref_levels' is a
// command-line parameter. If the mesh is of NURBS type, we convert it to
// a (piecewise-polynomial) high-order mesh.
for (int lev = 0; lev < ref_levels; lev++)
{
mesh->UniformRefinement();
}
if (mesh->NURBSext)
{
mesh->SetCurvature(max(order, 1));
}
mesh->GetBoundingBox(bb_min, bb_max, max(order, 1));
// 5. Define the discontinuous DG finite element space of the given
// polynomial order on the refined mesh.
DG_FECollection fec(order, dim, BasisType::Positive);
FiniteElementSpace fes(mesh, &fec);
cout << "Number of unknowns: " << fes.GetVSize() << endl;
// 6. Set up and assemble the bilinear and linear forms corresponding to the
// DG discretization. The DGTraceIntegrator involves integrals over mesh
// interior faces.
VectorFunctionCoefficient velocity(dim, velocity_function);
FunctionCoefficient inflow(inflow_function);
FunctionCoefficient u0(u0_function);
BilinearForm m(&fes);
m.AddDomainIntegrator(new MassIntegrator);
BilinearForm k(&fes);
k.AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
k.AddInteriorFaceIntegrator(
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
k.AddBdrFaceIntegrator(
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
LinearForm b(&fes);
b.AddBdrFaceIntegrator(
new BoundaryFlowIntegrator(inflow, velocity, -1.0, -0.5));
m.Assemble();
m.Finalize();
int skip_zeros = 0;
k.Assemble(skip_zeros);
k.Finalize(skip_zeros);
b.Assemble();
// 7. Define the initial conditions, save the corresponding grid function to
// a file and (optionally) save data in the VisIt format and initialize
// GLVis visualization.
GridFunction u(&fes);
u.ProjectCoefficient(u0);
{
ofstream omesh("ex9.mesh");
omesh.precision(precision);
mesh->Print(omesh);
ofstream osol("ex9-init.gf");
osol.precision(precision);
u.Save(osol);
}
// Create data collection for solution output: either VisItDataCollection for
// ascii data files, or SidreDataCollection for binary data files.
DataCollection *dc = NULL;
if (visit)
{
if (binary)
{
#ifdef MFEM_USE_SIDRE
dc = new SidreDataCollection("Example9", mesh);
#else
MFEM_ABORT("Must build with MFEM_USE_SIDRE=YES for binary output.");
#endif
}
else
{
dc = new VisItDataCollection("Example9", mesh);
dc->SetPrecision(precision);
}
dc->RegisterField("solution", &u);
dc->SetCycle(0);
dc->SetTime(0.0);
dc->Save();
}
socketstream sout;
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
sout.open(vishost, visport);
if (!sout)
{
cout << "Unable to connect to GLVis server at "
<< vishost << ':' << visport << endl;
visualization = false;
cout << "GLVis visualization disabled.\n";
}
else
{
sout.precision(precision);
sout << "solution\n" << *mesh << u;
sout << "pause\n";
sout << flush;
cout << "GLVis visualization paused."
<< " Press space (in the GLVis window) to resume it.\n";
}
}
Vector M_rowsums(m.Size());
m.SpMat().GetRowSums(M_rowsums);
// 8. Define the time-dependent evolution operator describing the ODE
// right-hand side, and perform time-integration (looping over the time
// iterations, ti, with a time-step dt).
FE_Evolution adv(m.SpMat(), k.SpMat(), b, k, M_rowsums);
double t = 0.0;
adv.SetTime(t);
ode_solver->Init(adv);
// Compute initial volume.
const double vol0 = M_rowsums * u;
bool done = false;
for (int ti = 0; !done; )
{
double dt_real = min(dt, t_final - t);
adv.SetTimeStep(dt_real);
ode_solver->Step(u, t, dt_real);
ti++;
done = (t >= t_final - 1e-8*dt);
if (done || ti % vis_steps == 0)
{
cout << "time step: " << ti << ", time: " << t << endl;
if (visualization)
{
sout << "solution\n" << *mesh << u << flush;
}
if (visit)
{
dc->SetCycle(ti);
dc->SetTime(t);
dc->Save();
}
}
}
// Print the error vs exact solution.
const double max_error = u.ComputeMaxError(u0),
l1_error = u.ComputeL1Error(u0),
l2_error = u.ComputeL2Error(u0);
std::cout << "Linf error = " << max_error << endl
<< "L1 error = " << l1_error << endl
<< "L2 error = " << l2_error << endl;
// Print error in volume.
const double vol = M_rowsums * u;
std::cout << "Vol error = " << vol - vol0 << endl;
// 9. Save the final solution. This output can be viewed later using GLVis:
// "glvis -m ex9.mesh -g ex9-final.gf".
{
ofstream osol("ex9-final.gf");
osol.precision(precision);
u.Save(osol);
}
// 10. Free the used memory.
delete ode_solver;
delete dc;
delete mesh;
return 0;
}
// Implementation of class FE_Evolution
FE_Evolution::FE_Evolution(SparseMatrix &_M, SparseMatrix &_K,
const Vector &_b, BilinearForm &_bf, Vector &M_rs)
: TimeDependentOperator(_M.Size()),
M(_M), K(_K), b(_b), M_prec(), M_solver(), z(_M.Size()),
bf(_bf), M_rowsums(M_rs)
{
M_solver.SetPreconditioner(M_prec);
M_solver.SetOperator(M);
M_solver.iterative_mode = false;
M_solver.SetRelTol(1e-9);
M_solver.SetAbsTol(0.0);
M_solver.SetMaxIter(100);
M_solver.SetPrintLevel(0);
}
void FE_Evolution::Mult(const Vector &x, Vector &y) const
{
// Compute bounds y_min, y_max for y from x on the ldofs.
const int dofs = x.Size();
Vector y_min(dofs), y_max(dofs);
const int *In = bf.SpMat().GetI(), *Jn = bf.SpMat().GetJ();
for (int i = 0, k = 0; i < dofs; i++)
{
double x_i_min = +std::numeric_limits<double>::infinity();
double x_i_max = -std::numeric_limits<double>::infinity();
for (int end = In[i+1]; k < end; k++)
{
const int j = Jn[k];
if (x(j) > x_i_max) { x_i_max = x(j); }
if (x(j) < x_i_min) { x_i_min = x(j); }
}
y_min(i) = x_i_min;
y_max(i) = x_i_max;
}
for (int i = 0; i < dofs; i++)
{
y_min(i) = (y_min(i) - x(i) ) / dt;
y_max(i) = (y_max(i) - x(i) ) / dt;
}
// Compute the high-order solution y = M^{-1} (K x + b).
K.Mult(x, z);
z += b;
M_solver.Mult(z, y);
// The solution y is an increment; it should not introduce new mass.
const double mass_y = 0.0;
// Perform optimization.
Vector y_out(dofs);
const int max_iter = 500;
const double rtol = 1.e-7;
double atol = 1.e-7;
OptimizationSolver *optsolver = NULL;
if (optimizer_type == 2)
{
#ifdef MFEM_USE_HIOP
HiopNlpOptimizer *tmp_opt_ptr = new HiopNlpOptimizer();
optsolver = tmp_opt_ptr;
#else
MFEM_ABORT("MFEM is not built with HiOp support!");
#endif
}
else
{
SLBQPOptimizer *slbqp = new SLBQPOptimizer();
slbqp->SetBounds(y_min, y_max);
slbqp->SetLinearConstraint(M_rowsums, mass_y);
atol = 1.e-15;
optsolver = slbqp;
}
OptimizedTransportProblem ot_prob(y, M_rowsums, mass_y, y_min, y_max);
optsolver->SetOptimizationProblem(ot_prob);
optsolver->SetMaxIter(max_iter);
optsolver->SetAbsTol(atol);
optsolver->SetRelTol(rtol);
optsolver->SetPrintLevel(0);
optsolver->Mult(y, y_out);
y = y_out;
delete optsolver;
}
// Velocity coefficient
void velocity_function(const Vector &x, Vector &v)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
double center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
{
// Translations in 1D, 2D, and 3D
switch (dim)
{
case 1: v(0) = (invert_velocity) ? -1.0 : 1.0; break;
case 2: v(0) = sqrt(2./3.); v(1) = sqrt(1./3.); break;
case 3: v(0) = sqrt(3./6.); v(1) = sqrt(2./6.); v(2) = sqrt(1./6.);
break;
}
break;
}
case 1:
case 2:
{
// Clockwise rotation in 2D around the origin
const double w = M_PI/2;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = w*X(1); v(1) = -w*X(0); break;
case 3: v(0) = w*X(1); v(1) = -w*X(0); v(2) = 0.0; break;
}
break;
}
case 3:
{
// Clockwise twisting rotation in 2D around the origin
const double w = M_PI/2;
double d = max((X(0)+1.)*(1.-X(0)),0.) * max((X(1)+1.)*(1.-X(1)),0.);
d = d*d;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = d*w*X(1); v(1) = -d*w*X(0); break;
case 3: v(0) = d*w*X(1); v(1) = -d*w*X(0); v(2) = 0.0; break;
}
break;
}
}
}
// Initial condition
double u0_function(const Vector &x)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
double center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
case 1:
{
switch (dim)
{
case 1:
return (X(0) > -0.15 && X(0) < 0.15) ? 1.0 : 0.0;
//return exp(-40.*pow(X(0)-0.0,2));
case 2:
case 3:
{
double rx = 0.45, ry = 0.25, cx = 0., cy = -0.2, w = 10.;
if (dim == 3)
{
const double s = (1. + 0.25*cos(2*M_PI*X(2)));
rx *= s;
ry *= s;
}
return ( erfc(w*(X(0)-cx-rx))*erfc(-w*(X(0)-cx+rx)) *
erfc(w*(X(1)-cy-ry))*erfc(-w*(X(1)-cy+ry)) )/16;
}
}
}
case 2:
{
double x_ = X(0), y_ = X(1), rho, phi;
rho = hypot(x_, y_);
phi = atan2(y_, x_);
return pow(sin(M_PI*rho),2)*sin(3*phi);
}
case 3:
{
const double f = M_PI;
return sin(f*X(0))*sin(f*X(1));
}
}
return 0.0;
}
// Inflow boundary condition (zero for the problems considered in this example)
double inflow_function(const Vector &x)
{
switch (problem)
{
case 0:
case 1:
case 2:
case 3: return 0.0;
}
return 0.0;
}
-802
View File
@@ -1,802 +0,0 @@
// MFEM Example 9 with Nonlinear Constrained Optimization - Parallel Version
//
// Compile with: make ex9p
//
// Sample runs:
//
// mpirun -np 4 ex9p -m ../../data/periodic-segment.mesh -rs 3 -p 0 -o 2 -dt 0.002 -opt 1
// mpirun -np 4 ex9p -m ../../data/periodic-segment.mesh -rs 3 -p 0 -o 2 -dt 0.002 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 0 -rs 2 -dt 0.01 -tf 10 -opt 1
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 0 -rs 2 -dt 0.01 -tf 10 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 1 -rs 2 -dt 0.005 -tf 9 -opt 1
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 1 -rs 2 -dt 0.005 -tf 9 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/amr-quad.mesh -p 1 -rs 1 -dt 0.002 -tf 9 -opt 1
// mpirun -np 4 ex9p -m ../../data/amr-quad.mesh -p 1 -rs 1 -dt 0.002 -tf 9 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/disc-nurbs.mesh -p 1 -rs 2 -dt 0.005 -tf 9 -opt 1
// mpirun -np 4 ex9p -m ../../data/disc-nurbs.mesh -p 1 -rs 2 -dt 0.005 -tf 9 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/disc-nurbs.mesh -p 2 -rs 2 -dt 0.01 -tf 9 -opt 1
// mpirun -np 4 ex9p -m ../../data/disc-nurbs.mesh -p 2 -rs 2 -dt 0.01 -tf 9 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 3 -rs 3 -dt 0.0025 -tf 9 -opt 1
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 3 -rs 3 -dt 0.0025 -tf 9 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/periodic-cube.mesh -p 0 -rs 2 -o 2 -dt 0.02 -tf 8 -opt 1
// mpirun -np 4 ex9p -m ../../data/periodic-cube.mesh -p 0 -rs 2 -o 2 -dt 0.02 -tf 8 -opt 2
// Description: This example modifies the standard MFEM ex9 by adding nonlinear
// constrained optimization capabilities through the SLBQP and
// HIOP solvers. It demonstrates how a user can define a custom
// class OptimizationProblem that includes linear/nonlinear
// equality/inequality constraints. This optimization is applied
// as post-processing to the solution of the transport equation.
//
// Description of ex9:
// This example code solves the time-dependent advection equation
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
// u0(x)=u(0,x) is a given initial condition.
//
// The example demonstrates the use of Discontinuous Galerkin (DG)
// bilinear forms in MFEM (face integrators), the use of explicit
// ODE time integrators, the definition of periodic boundary
// conditions through periodic meshes, as well as the use of GLVis
// for persistent visualization of a time-evolving solution. The
// saving of time-dependent data files for external visualization
// with VisIt (visit.llnl.gov) is also illustrated.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
// Choice for the problem setup. The fluid velocity, initial condition and
// inflow boundary condition are chosen based on this parameter.
int problem;
// Nonlinear optimizer.
int optimizer_type;
// Velocity coefficient
bool invert_velocity = false;
void velocity_function(const Vector &x, Vector &v);
// Initial condition
double u0_function(const Vector &x);
// Inflow boundary condition
double inflow_function(const Vector &x);
// Mesh bounding box
Vector bb_min, bb_max;
/// Computes C(x) = sum w_i x_i, where w is a given Vector.
class LinearScaleOperator : public Operator
{
private:
ParFiniteElementSpace &pfes;
// Local weights.
const Vector &w;
// Gradient for the tdofs.
mutable DenseMatrix grad;
public:
LinearScaleOperator(ParFiniteElementSpace &space, const Vector &weight)
: Operator(1, space.TrueVSize()),
pfes(space), w(weight), grad(1, width)
{
Vector w_glob(width);
pfes.Dof_TrueDof_Matrix()->MultTranspose(w, w_glob);
for (int i = 0; i < width; i++) { grad(0, i) = w_glob(i); }
}
virtual void Mult(const Vector &x, Vector &y) const
{
Vector x_loc(w.Size());
pfes.GetProlongationMatrix()->Mult(x, x_loc);
const double loc_res = w * x_loc;
MPI_Allreduce(&loc_res, &y(0), 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
}
virtual Operator &GetGradient(const Vector &x) const
{
return grad;
}
};
/// Nonlinear monotone bounded operator to test nonlinear ineq constraints.
/// Computes D(x) = tanh(sum(x_i)).
class TanhSumOperator : public Operator
{
private:
// Gradient for the tdofs.
mutable DenseMatrix grad;
public:
TanhSumOperator(ParFiniteElementSpace &space)
: Operator(1, space.TrueVSize()), grad(1, width) { }
virtual void Mult(const Vector &x, Vector &y) const
{
double sum_loc = x.Sum();
MPI_Allreduce(&sum_loc, &y(0), 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
y(0) = std::tanh(y(0));
}
virtual Operator &GetGradient(const Vector &x) const
{
double sum_loc = x.Sum();
double dtanh;
MPI_Allreduce(&sum_loc, &dtanh, 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
dtanh = 1.0 - pow(std::tanh(dtanh), 2);
for (int i = 0; i < width; i++) { grad(0, i) = dtanh; }
return grad;
}
};
/** Monotone and conservative a-posteriori correction for transport solutions:
* Find x that minimizes 0.5 || x - x_HO ||^2, subject to
* sum w_i x_i = mass,
* tanh(sum(x_i_min)) <= tanh(sum(x_i)) <= tanh(sum(x_i_max)),
* x_i_min <= x_i <= x_i_max,
*/
class OptimizedTransportProblem : public OptimizationProblem
{
private:
const Vector &x_HO;
Vector massvec, d_lo, d_hi;
const LinearScaleOperator LSoper;
const TanhSumOperator TSoper;
public:
OptimizedTransportProblem(ParFiniteElementSpace &space,
const Vector &xho, const Vector &w, double mass,
const Vector &xmin, const Vector &xmax)
: OptimizationProblem(xho.Size(), NULL, NULL),
x_HO(xho), massvec(1), d_lo(1), d_hi(1),
LSoper(space, w), TSoper(space)
{
C = &LSoper;
massvec(0) = mass;
SetEqualityConstraint(massvec);
D = &TSoper;
double lsums[2], gsums[2];
lsums[0] = xmin.Sum();
lsums[1] = xmax.Sum();
MPI_Allreduce(lsums, gsums, 2, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
d_lo(0) = std::tanh(gsums[0]);
d_hi(0) = std::tanh(gsums[1]);
MFEM_ASSERT(d_lo(0) < d_hi(0),
"The bounds produce an infeasible optimization problem");
SetInequalityConstraint(d_lo, d_hi);
SetSolutionBounds(xmin, xmax);
}
virtual double CalcObjective(const Vector &x) const
{
double loc_res = 0.0;
for (int i = 0; i < input_size; i++)
{
const double d = x(i) - x_HO(i);
loc_res += d * d;
}
loc_res *= 0.5;
double res;
MPI_Allreduce(&loc_res, &res, 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
return res;
}
virtual void CalcObjectiveGrad(const Vector &x, Vector &grad) const
{
for (int i = 0; i < input_size; i++) { grad(i) = x(i) - x_HO(i); }
}
};
/** A time-dependent operator for the right-hand side of the ODE. The DG weak
form of du/dt = -v.grad(u) is M du/dt = K u + b, where M and K are the mass
and advection matrices, and b describes the flow on the boundary. This can
be written as a general ODE, du/dt = M^{-1} (K u + b), and this class is
used to evaluate the right-hand side. */
class FE_Evolution : public TimeDependentOperator
{
private:
HypreParMatrix &M, &K;
const Vector &b;
HypreSmoother M_prec;
CGSolver M_solver;
mutable Vector z;
double dt;
ParBilinearForm &pbf;
Vector &M_rowsums;
public:
FE_Evolution(HypreParMatrix &_M, HypreParMatrix &_K,
const Vector &_b, ParBilinearForm &_pbf, Vector &M_rs);
void SetTimeStep(double _dt) { dt = _dt; }
void SetK(HypreParMatrix &_K) { K = _K; }
virtual void Mult(const Vector &x, Vector &y) const;
virtual ~FE_Evolution() { }
};
int main(int argc, char *argv[])
{
// 1. Initialize MPI.
int num_procs, myid;
MPI_Init(&argc, &argv);
MPI_Comm_size(MPI_COMM_WORLD, &num_procs);
MPI_Comm_rank(MPI_COMM_WORLD, &myid);
// 2. Parse command-line options.
problem = 0;
optimizer_type = 1;
const char *mesh_file = "../../data/periodic-hexagon.mesh";
int ser_ref_levels = 2;
int par_ref_levels = 0;
int order = 3;
int ode_solver_type = 3;
double t_final = 1.0;
double dt = 0.01;
bool visualization = true;
bool visit = false;
bool binary = false;
int vis_steps = 5;
int precision = 8;
cout.precision(precision);
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&problem, "-p", "--problem",
"Problem setup to use. See options in velocity_function().");
args.AddOption(&ser_ref_levels, "-rs", "--refine-serial",
"Number of times to refine the mesh uniformly in serial.");
args.AddOption(&par_ref_levels, "-rp", "--refine-parallel",
"Number of times to refine the mesh uniformly in parallel.");
args.AddOption(&order, "-o", "--order",
"Order (degree) of the finite elements.");
args.AddOption(&optimizer_type, "-opt", "--optimizer",
"Nonlinear optimizer: 1 - SLBQP,\n\t"
" 2 - HIOP.");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
"ODE solver: 1 - Forward Euler,\n\t"
" 2 - RK2 SSP, 3 - RK3 SSP.");
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
"Time step.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&visit, "-visit", "--visit-datafiles", "-no-visit",
"--no-visit-datafiles",
"Save data files for VisIt (visit.llnl.gov) visualization.");
args.AddOption(&binary, "-binary", "--binary-datafiles", "-ascii",
"--ascii-datafiles",
"Use binary (Sidre) or ascii format for VisIt data files.");
args.AddOption(&vis_steps, "-vs", "--visualization-steps",
"Visualize every n-th timestep.");
args.Parse();
if (!args.Good())
{
if (myid == 0) { args.PrintUsage(cout); }
MPI_Finalize();
return 1;
}
if (myid == 0) { args.PrintOptions(cout); }
// 3. Read the serial mesh from the given mesh file on all processors. We can
// handle geometrically periodic meshes in this code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Define the ODE solver used for time integration. Several explicit
// Runge-Kutta methods are available.
ODESolver *ode_solver = NULL;
switch (ode_solver_type)
{
case 1: ode_solver = new ForwardEulerSolver; break;
case 2: ode_solver = new RK2Solver(1.0); break;
case 3: ode_solver = new RK3SSPSolver; break;
case 4: ode_solver = new RK4Solver; break;
case 6: ode_solver = new RK6Solver; break;
default:
if (myid == 0)
{
cout << "Unknown ODE solver type: " << ode_solver_type << '\n';
}
delete mesh;
MPI_Finalize();
return 3;
}
// 5. Refine the mesh in serial to increase the resolution. In this example
// we do 'ser_ref_levels' of uniform refinement, where 'ser_ref_levels' is
// a command-line parameter. If the mesh is of NURBS type, we convert it
// to a (piecewise-polynomial) high-order mesh.
for (int lev = 0; lev < ser_ref_levels; lev++)
{
mesh->UniformRefinement();
}
if (mesh->NURBSext)
{
mesh->SetCurvature(max(order, 1));
}
mesh->GetBoundingBox(bb_min, bb_max, max(order, 1));
// 6. Define the parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
for (int lev = 0; lev < par_ref_levels; lev++)
{
pmesh->UniformRefinement();
}
// 7. Define the parallel discontinuous DG finite element space on the
// parallel refined mesh of the given polynomial order.
DG_FECollection fec(order, dim, BasisType::Positive);
ParFiniteElementSpace *fes = new ParFiniteElementSpace(pmesh, &fec);
HYPRE_Int global_vSize = fes->GlobalTrueVSize();
if (myid == 0)
{
cout << "Number of unknowns: " << global_vSize << endl;
}
// 8. Set up and assemble the parallel bilinear and linear forms (and the
// parallel hypre matrices) corresponding to the DG discretization. The
// DGTraceIntegrator involves integrals over mesh interior faces.
VectorFunctionCoefficient velocity(dim, velocity_function);
FunctionCoefficient inflow(inflow_function);
FunctionCoefficient u0(u0_function);
ParBilinearForm *m = new ParBilinearForm(fes);
m->AddDomainIntegrator(new MassIntegrator);
ParBilinearForm *k = new ParBilinearForm(fes);
k->AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
k->AddInteriorFaceIntegrator(
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
k->AddBdrFaceIntegrator(
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
ParLinearForm *b = new ParLinearForm(fes);
b->AddBdrFaceIntegrator(
new BoundaryFlowIntegrator(inflow, velocity, -1.0, -0.5));
m->Assemble();
m->Finalize();
int skip_zeros = 0;
k->Assemble(skip_zeros);
k->Finalize(skip_zeros);
b->Assemble();
HypreParMatrix *M = m->ParallelAssemble();
HypreParMatrix *K = k->ParallelAssemble();
HypreParVector *B = b->ParallelAssemble();
// 9. Define the initial conditions, save the corresponding grid function to
// a file and (optionally) save data in the VisIt format and initialize
// GLVis visualization.
ParGridFunction *u = new ParGridFunction(fes);
u->ProjectCoefficient(u0);
HypreParVector *U = u->GetTrueDofs();
{
ostringstream mesh_name, sol_name;
mesh_name << "ex9-mesh." << setfill('0') << setw(6) << myid;
sol_name << "ex9-init." << setfill('0') << setw(6) << myid;
ofstream omesh(mesh_name.str().c_str());
omesh.precision(precision);
pmesh->Print(omesh);
ofstream osol(sol_name.str().c_str());
osol.precision(precision);
u->Save(osol);
}
// Create data collection for solution output: either VisItDataCollection for
// ascii data files, or SidreDataCollection for binary data files.
DataCollection *dc = NULL;
if (visit)
{
if (binary)
{
#ifdef MFEM_USE_SIDRE
dc = new SidreDataCollection("Example9-Parallel", pmesh);
#else
MFEM_ABORT("Must build with MFEM_USE_SIDRE=YES for binary output.");
#endif
}
else
{
dc = new VisItDataCollection("Example9-Parallel", pmesh);
dc->SetPrecision(precision);
// To save the mesh using MFEM's parallel mesh format:
// dc->SetFormat(DataCollection::PARALLEL_FORMAT);
}
dc->RegisterField("solution", u);
dc->SetCycle(0);
dc->SetTime(0.0);
dc->Save();
}
socketstream sout;
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
sout.open(vishost, visport);
if (!sout)
{
if (myid == 0)
cout << "Unable to connect to GLVis server at "
<< vishost << ':' << visport << endl;
visualization = false;
if (myid == 0)
{
cout << "GLVis visualization disabled.\n";
}
}
else
{
sout << "parallel " << num_procs << " " << myid << "\n";
sout.precision(precision);
sout << "solution\n" << *pmesh << *u;
sout << "pause\n";
sout << flush;
if (myid == 0)
cout << "GLVis visualization paused."
<< " Press space (in the GLVis window) to resume it.\n";
}
}
Vector M_rowsums(m->Size());
m->SpMat().GetRowSums(M_rowsums);
// 10. Define the time-dependent evolution operator describing the ODE
// right-hand side, and perform time-integration (looping over the time
// iterations, ti, with a time-step dt).
FE_Evolution adv(*M, *K, *B, *k, M_rowsums);
double t = 0.0;
adv.SetTime(t);
ode_solver->Init(adv);
*u = *U;
// Compute initial volume.
const double vol0_loc = M_rowsums * (*u);
double vol0;
MPI_Allreduce(&vol0_loc, &vol0, 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
bool done = false;
for (int ti = 0; !done; )
{
double dt_real = min(dt, t_final - t);
adv.SetTimeStep(dt_real);
ode_solver->Step(*U, t, dt_real);
ti++;
done = (t >= t_final - 1e-8*dt);
if (done || ti % vis_steps == 0)
{
if (myid == 0)
{
cout << "time step: " << ti << ", time: " << t << endl;
}
// 11. Extract the parallel grid function corresponding to the finite
// element approximation U (the local solution on each processor).
*u = *U;
if (visualization)
{
sout << "parallel " << num_procs << " " << myid << "\n";
sout << "solution\n" << *pmesh << *u << flush;
}
if (visit)
{
dc->SetCycle(ti);
dc->SetTime(t);
dc->Save();
}
}
}
// Print the error vs exact solution.
const double max_error = u->ComputeMaxError(u0),
l1_error = u->ComputeL1Error(u0),
l2_error = u->ComputeL2Error(u0);
if (myid == 0)
{
std::cout << "Linf error = " << max_error << endl
<< "L1 error = " << l1_error << endl
<< "L2 error = " << l2_error << endl;
}
// Print error in volume.
const double vol_loc = M_rowsums * (*u);
double vol;
MPI_Allreduce(&vol_loc, &vol, 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
if (myid == 0)
{
std::cout << "Vol error = " << vol - vol0 << endl;
}
// 12. Save the final solution in parallel. This output can be viewed later
// using GLVis: "glvis -np <np> -m ex9-mesh -g ex9-final".
{
*u = *U;
ostringstream sol_name;
sol_name << "ex9-final." << setfill('0') << setw(6) << myid;
ofstream osol(sol_name.str().c_str());
osol.precision(precision);
u->Save(osol);
}
// 13. Free the used memory.
delete U;
delete u;
delete B;
delete b;
delete K;
delete k;
delete M;
delete m;
delete fes;
delete pmesh;
delete ode_solver;
delete dc;
MPI_Finalize();
return 0;
}
// Implementation of class FE_Evolution
FE_Evolution::FE_Evolution(HypreParMatrix &_M, HypreParMatrix &_K,
const Vector &_b, ParBilinearForm &_pbf,
Vector &M_rs)
: TimeDependentOperator(_M.Height()),
M(_M), K(_K), b(_b), M_solver(M.GetComm()), z(_M.Height()),
pbf(_pbf), M_rowsums(M_rs)
{
M_prec.SetType(HypreSmoother::Jacobi);
M_solver.SetPreconditioner(M_prec);
M_solver.SetOperator(M);
M_solver.iterative_mode = false;
M_solver.SetRelTol(1e-9);
M_solver.SetAbsTol(0.0);
M_solver.SetMaxIter(100);
M_solver.SetPrintLevel(0);
}
void FE_Evolution::Mult(const Vector &x, Vector &y) const
{
// Get values on the ldofs.
ParFiniteElementSpace *pfes = pbf.ParFESpace();
ParGridFunction x_gf(pfes);
pfes->GetProlongationMatrix()->Mult(x, x_gf);
// Compute bounds y_min, y_max for y from from x on the ldofs.
const int ldofs = x_gf.Size();
Vector y_min(ldofs), y_max(ldofs);
x_gf.ExchangeFaceNbrData();
Vector &x_nd = x_gf.FaceNbrData();
const int *In = pbf.SpMat().GetI(), *Jn = pbf.SpMat().GetJ();
for (int i = 0, k = 0; i < ldofs; i++)
{
double x_i_min = +std::numeric_limits<double>::infinity();
double x_i_max = -std::numeric_limits<double>::infinity();
for (int end = In[i+1]; k < end; k++)
{
const int j = Jn[k];
const double x_j = (j < ldofs) ? x(j): x_nd(j-ldofs);
if (x_j > x_i_max) { x_i_max = x_j; }
if (x_j < x_i_min) { x_i_min = x_j; }
}
y_min(i) = x_i_min;
y_max(i) = x_i_max;
}
for (int i = 0; i < ldofs; i++)
{
y_min(i) = (y_min(i) - x_gf(i) ) / dt;
y_max(i) = (y_max(i) - x_gf(i) ) / dt;
}
Vector y_min_tdofs(y.Size()), y_max_tdofs(y.Size());
// Move the bounds to the tdofs.
pfes->GetRestrictionMatrix()->Mult(y_min, y_min_tdofs);
pfes->GetRestrictionMatrix()->Mult(y_max, y_max_tdofs);
// Compute the high-order solution y = M^{-1} (K x + b) on the tdofs.
K.Mult(x, z);
z += b;
M_solver.Mult(z, y);
// The solution y is an increment; it should not introduce new mass.
const double mass_y = 0.0;
// Perform optimization on the tdofs.
Vector y_out(y.Size());
const int max_iter = 500;
const double rtol = 1.e-7;
double atol = 1.e-7;
OptimizationSolver* optsolver = NULL;
if (optimizer_type == 2)
{
#ifdef MFEM_USE_HIOP
HiopNlpOptimizer *tmp_opt_ptr = new HiopNlpOptimizer(MPI_COMM_WORLD);
optsolver = tmp_opt_ptr;
#else
MFEM_ABORT("MFEM is not built with HiOp support!");
#endif
}
else
{
SLBQPOptimizer *slbqp = new SLBQPOptimizer(MPI_COMM_WORLD);
slbqp->SetBounds(y_min_tdofs, y_max_tdofs);
slbqp->SetLinearConstraint(M_rowsums, mass_y);
atol = 1.e-15;
optsolver = slbqp;
}
OptimizedTransportProblem ot_prob(*pfes, y, M_rowsums, mass_y,
y_min_tdofs, y_max_tdofs);
optsolver->SetOptimizationProblem(ot_prob);
optsolver->SetMaxIter(max_iter);
optsolver->SetAbsTol(atol);
optsolver->SetRelTol(rtol);
optsolver->SetPrintLevel(0);
optsolver->Mult(y, y_out);
y = y_out;
delete optsolver;
}
// Velocity coefficient
void velocity_function(const Vector &x, Vector &v)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
double center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
{
// Translations in 1D, 2D, and 3D
switch (dim)
{
case 1: v(0) = (invert_velocity) ? -1.0 : 1.0; break;
case 2: v(0) = sqrt(2./3.); v(1) = sqrt(1./3.); break;
case 3: v(0) = sqrt(3./6.); v(1) = sqrt(2./6.); v(2) = sqrt(1./6.);
break;
}
break;
}
case 1:
case 2:
{
// Clockwise rotation in 2D around the origin
const double w = M_PI/2;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = w*X(1); v(1) = -w*X(0); break;
case 3: v(0) = w*X(1); v(1) = -w*X(0); v(2) = 0.0; break;
}
break;
}
case 3:
{
// Clockwise twisting rotation in 2D around the origin
const double w = M_PI/2;
double d = max((X(0)+1.)*(1.-X(0)),0.) * max((X(1)+1.)*(1.-X(1)),0.);
d = d*d;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = d*w*X(1); v(1) = -d*w*X(0); break;
case 3: v(0) = d*w*X(1); v(1) = -d*w*X(0); v(2) = 0.0; break;
}
break;
}
}
}
// Initial condition
double u0_function(const Vector &x)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
double center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
case 1:
{
switch (dim)
{
case 1:
return (X(0) > -0.15 && X(0) < 0.15) ? 1.0 : 0.0;
//return exp(-40.*pow(X(0)-0.0,2));
case 2:
case 3:
{
double rx = 0.45, ry = 0.25, cx = 0., cy = -0.2, w = 10.;
if (dim == 3)
{
const double s = (1. + 0.25*cos(2*M_PI*X(2)));
rx *= s;
ry *= s;
}
return ( erfc(w*(X(0)-cx-rx))*erfc(-w*(X(0)-cx+rx)) *
erfc(w*(X(1)-cy-ry))*erfc(-w*(X(1)-cy+ry)) )/16;
}
}
}
case 2:
{
double x_ = X(0), y_ = X(1), rho, phi;
rho = hypot(x_, y_);
phi = atan2(y_, x_);
return pow(sin(M_PI*rho),2)*sin(3*phi);
}
case 3:
{
const double f = M_PI;
return sin(f*X(0))*sin(f*X(1));
}
}
return 0.0;
}
// Inflow boundary condition (zero for the problems considered in this example)
double inflow_function(const Vector &x)
{
switch (problem)
{
case 0:
case 1:
case 2:
case 3: return 0.0;
}
return 0.0;
}
-66
View File
@@ -1,66 +0,0 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/hiop/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
SEQ_EXAMPLES = ex9
PAR_EXAMPLES = ex9p
ifeq ($(MFEM_USE_MPI),NO)
EXAMPLES = $(SEQ_EXAMPLES)
else
EXAMPLES = $(PAR_EXAMPLES) $(SEQ_EXAMPLES)
endif
.SUFFIXES:
.SUFFIXES: .o .cpp .mk
.PHONY: all clean clean-build clean-exec
# Remove built-in rule
%: %.cpp
# Replace the default implicit rule for *.cpp files
%: $(SRC)%.cpp $(MFEM_LIB_FILE) $(CONFIG_MK)
$(MFEM_CXX) $(MFEM_FLAGS) $< -o $@ $(MFEM_LIBS)
all: $(EXAMPLES)
ifeq ($(MFEM_USE_HIOP),NO)
$(EXAMPLES):
$(error MFEM is not configured with HIOP)
endif
MFEM_TESTS = EXAMPLES
include $(MFEM_TEST_MK)
# Testing: "test" target and mfem-test* variables are defined in config/test.mk
# Generate an error message if the MFEM library is not built and exit
$(MFEM_LIB_FILE):
$(error The MFEM library is not built)
clean: clean-build clean-exec
clean-build:
rm -f *.o *~ $(SEQ_EXAMPLES) $(PAR_EXAMPLES)
rm -rf *.dSYM *.TVD.*breakpoints
clean-exec:
@rm -f ex9.mesh ex9-mesh.* ex9-init.* ex9-final.* Example9*
+4 -11
View File
@@ -22,9 +22,9 @@ MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
SEQ_EXAMPLES = ex1 ex2 ex3 ex4 ex5 ex6 ex7 ex8 ex9 ex10 ex14 ex15 ex16 ex17\
ex18 ex19 ex20 ex21 ex22
ex18 ex19 ex20 ex21
PAR_EXAMPLES = ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex8p ex9p ex10p ex11p ex12p\
ex13p ex14p ex15p ex16p ex17p ex18p ex19p ex20p ex21p ex22p
ex13p ex14p ex15p ex16p ex17p ex18p ex19p ex20p ex21p
ifeq ($(MFEM_USE_MPI),NO)
EXAMPLES = $(SEQ_EXAMPLES)
@@ -41,13 +41,6 @@ endif
ifeq ($(MFEM_USE_PUMI),YES)
SUBDIRS += pumi
endif
ifeq ($(MFEM_USE_HIOP),YES)
SUBDIRS += hiop
endif
ifeq ($(MFEM_USE_GINKGO),YES)
SUBDIRS += ginkgo
endif
SUBDIRS_ALL = $(addsuffix /all,$(SUBDIRS))
SUBDIRS_TEST = $(addsuffix /test,$(SUBDIRS))
SUBDIRS_CLEAN = $(addsuffix /clean,$(SUBDIRS))
@@ -124,8 +117,8 @@ clean-build:
clean-exec:
@rm -f refined.mesh displaced.mesh mesh.* ex5.mesh
@rm -rf Example5* Example9* Example15* Example16* PVExample*
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.*
@rm -rf Example5* Example9* Example15* Example16*
@rm -f sphere_refined.* sol.* sol_u.* sol_p.*
@rm -f ex9.mesh ex9-mesh.* ex9-init.* ex9-final.*
@rm -f deformed.* velocity.* elastic_energy.* mode_*
@rm -f ex16.mesh ex16-mesh.* ex16-init.* ex16-final.*
-25
View File
@@ -14,16 +14,8 @@ set(SRCS
bilinearform_ext.cpp
bilininteg.cpp
bilininteg_diffusion.cpp
bilininteg_divergence.cpp
bilininteg_gradient.cpp
bilininteg_mass.cpp
bilininteg_divergence.cpp
bilininteg_gradient.cpp
bilininteg_convection.cpp
bilininteg_vecdiffusion.cpp
bilininteg_vecmass.cpp
coefficient.cpp
complex_fem.cpp
datacollection.cpp
eltrans.cpp
estimators.cpp
@@ -37,13 +29,10 @@ set(SRCS
linearform.cpp
lininteg.cpp
nonlinearform.cpp
nonlinearform_ext.cpp
nonlininteg.cpp
nonlininteg_vectorconvection.cpp
staticcond.cpp
tmop.cpp
tmop_tools.cpp
gslib.cpp
)
set(HDRS
@@ -51,7 +40,6 @@ set(HDRS
bilinearform_ext.hpp
bilininteg.hpp
coefficient.hpp
complex_fem.hpp
datacollection.hpp
eltrans.hpp
estimators.hpp
@@ -66,7 +54,6 @@ set(HDRS
linearform.hpp
lininteg.hpp
nonlinearform.hpp
nonlinearform_ext.hpp
nonlininteg.hpp
staticcond.hpp
tbilinearform.hpp
@@ -79,7 +66,6 @@ set(HDRS
tintrules.hpp
tmop.hpp
tmop_tools.hpp
gslib.hpp
)
if (MFEM_USE_SIDRE)
@@ -109,17 +95,6 @@ if (MFEM_USE_MPI)
pnonlinearform.hpp)
endif()
if (MFEM_USE_CEED)
list(APPEND SRCS
libceed/ceed.cpp
libceed/diffusion.cpp
libceed/mass.cpp)
list(APPEND HDRS
libceed/ceed.hpp
libceed/diffusion.hpp
libceed/mass.hpp)
endif()
convert_filenames_to_full_paths(SRCS)
convert_filenames_to_full_paths(HDRS)
+14 -205
View File
@@ -204,7 +204,7 @@ void BilinearForm::UseSparsity(SparseMatrix &A)
<< A.Height() << " x " << A.Width());
MFEM_ASSERT(A.Finalized(), "matrix A must be Finalized");
UseSparsity(A.GetI(), A.GetJ(), A.ColumnsAreSorted());
UseSparsity(A.GetI(), A.GetJ(), A.areColumnsSorted());
}
double& BilinearForm::Elem (int i, int j)
@@ -608,41 +608,18 @@ void BilinearForm::ConformingAssemble()
width = mat->Width();
}
void BilinearForm::AssembleDiagonal(Vector &diag) const
{
if (ext)
{
MFEM_ASSERT(diag.Size() == fes->GetTrueVSize(),
"Vector for holding diagonal has wrong size!");
const Operator *P = fes->GetProlongationMatrix();
if (!IsIdentityProlongation(P))
{
Vector local_diag(P->Height());
ext->AssembleDiagonal(local_diag);
P->MultTranspose(local_diag, diag);
}
else
{
ext->AssembleDiagonal(diag);
}
}
else
{
MFEM_ABORT("Not implemented. Maybe assemble your bilinear form into a "
"matrix and use SparseMatrix::GetDiag?");
}
}
void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
Vector &b, OperatorHandle &A, Vector &X,
Vector &B, int copy_interior)
{
const SparseMatrix *P = fes->GetConformingProlongation();
if (ext)
{
ext->FormLinearSystem(ess_tdof_list, x, b, A, X, B, copy_interior);
return;
}
const SparseMatrix *P = fes->GetConformingProlongation();
FormSystemMatrix(ess_tdof_list, A);
// Transform the system and perform the elimination in B, based on the
@@ -983,18 +960,6 @@ void BilinearForm::EliminateVDofsInRHS(
mat->PartMult(vdofs, x, b);
}
void BilinearForm::Mult(const Vector &x, Vector &y) const
{
if (ext)
{
ext->Mult(x, y);
}
else
{
mat->Mult(x, y);
}
}
void BilinearForm::Update(FiniteElementSpace *nfes)
{
bool full_update;
@@ -1070,11 +1035,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
trial_fes = tr_fes;
test_fes = te_fes;
mat = NULL;
mat_e = NULL;
extern_bfs = 0;
assembly = AssemblyLevel::FULL;
ext = NULL;
}
MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
@@ -1085,7 +1046,6 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
trial_fes = tr_fes;
test_fes = te_fes;
mat = NULL;
mat_e = NULL;
extern_bfs = 1;
// Copy the pointers to the integrators
@@ -1096,38 +1056,6 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
bbfi_marker = mbf->bbfi_marker;
btfbfi_marker = mbf->btfbfi_marker;
assembly = AssemblyLevel::FULL;
ext = NULL;
}
void MixedBilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
{
if (ext)
{
MFEM_ABORT("the assembly level has already been set!");
}
assembly = assembly_level;
switch (assembly)
{
case AssemblyLevel::FULL:
// ext = new FAMixedBilinearFormExtension(this);
// Use the original BilinearForm implementation for now
break;
case AssemblyLevel::ELEMENT:
mfem_error("Element assembly not supported yet... stay tuned!");
// ext = new EAMixedBilinearFormExtension(this);
break;
case AssemblyLevel::PARTIAL:
ext = new PAMixedBilinearFormExtension(this);
break;
case AssemblyLevel::NONE:
mfem_error("Matrix-free action not supported yet... stay tuned!");
// ext = new MFMixedBilinearFormExtension(this);
break;
default:
mfem_error("Unknown assembly level");
}
}
double & MixedBilinearForm::Elem (int i, int j)
@@ -1140,63 +1068,31 @@ const double & MixedBilinearForm::Elem (int i, int j) const
return (*mat)(i, j);
}
void MixedBilinearForm::Mult(const Vector & x, Vector & y) const
void MixedBilinearForm::Mult (const Vector & x, Vector & y) const
{
y = 0.0;
AddMult(x, y);
mat -> Mult (x, y);
}
void MixedBilinearForm::AddMult(const Vector & x, Vector & y,
const double a) const
void MixedBilinearForm::AddMult (const Vector & x, Vector & y,
const double a) const
{
if (ext)
{
ext->AddMult(x, y, a);
}
else
{
mat->AddMult(x, y, a);
}
mat -> AddMult (x, y, a);
}
void MixedBilinearForm::MultTranspose(const Vector & x, Vector & y) const
void MixedBilinearForm::AddMultTranspose (const Vector & x, Vector & y,
const double a) const
{
y = 0.0;
AddMultTranspose(x, y);
}
void MixedBilinearForm::AddMultTranspose(const Vector & x, Vector & y,
const double a) const
{
if (ext)
{
ext->AddMultTranspose(x, y, a);
}
else
{
mat->AddMultTranspose(x, y, a);
}
mat -> AddMultTranspose (x, y, a);
}
MatrixInverse * MixedBilinearForm::Inverse() const
{
if (assembly != AssemblyLevel::FULL)
{
MFEM_WARNING("MixedBilinearForm::Inverse not possible with this assembly level!");
return NULL;
}
else
{
return mat -> Inverse ();
}
return mat -> Inverse ();
}
void MixedBilinearForm::Finalize (int skip_zeros)
{
if (assembly == AssemblyLevel::FULL)
{
mat -> Finalize (skip_zeros);
}
mat -> Finalize (skip_zeros);
}
void MixedBilinearForm::GetBlocks(Array2D<SparseMatrix *> &blocks) const
@@ -1249,12 +1145,6 @@ void MixedBilinearForm::AddBdrTraceFaceIntegrator(BilinearFormIntegrator *bfi,
void MixedBilinearForm::Assemble (int skip_zeros)
{
if (ext)
{
ext->Assemble();
return;
}
Array<int> tr_vdofs, te_vdofs;
ElementTransformation *eltrans;
DenseMatrix elemmat;
@@ -1422,12 +1312,6 @@ void MixedBilinearForm::Assemble (int skip_zeros)
void MixedBilinearForm::ConformingAssemble()
{
if (assembly != AssemblyLevel::FULL)
{
MFEM_WARNING("Conforming assemble not supported for this assembly level!");
return;
}
Finalize();
const SparseMatrix *P2 = test_fes->GetConformingProlongation();
@@ -1586,91 +1470,17 @@ void MixedBilinearForm::EliminateTestDofs (const Array<int> &bdr_attr_is_ess)
}
}
void MixedBilinearForm::FormRectangularSystemMatrix(const Array<int>
&trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorHandle &A)
{
if (ext)
{
ext->FormRectangularSystemOperator(trial_tdof_list, test_tdof_list, A);
return;
}
const SparseMatrix *test_P = test_fes->GetConformingProlongation();
const SparseMatrix *trial_P = trial_fes->GetConformingProlongation();
mat->Finalize();
if (test_P) // TODO: Must actually check for trial_P too
{
SparseMatrix *m = RAP(*test_P, *mat, *trial_P);
delete mat;
mat = m;
}
Array<int> ess_trial_tdof_marker, ess_test_tdof_marker;
FiniteElementSpace::ListToMarker(trial_tdof_list, trial_fes->GetTrueVSize(),
ess_trial_tdof_marker);
FiniteElementSpace::ListToMarker(test_tdof_list, test_fes->GetTrueVSize(),
ess_test_tdof_marker);
mat_e = new SparseMatrix(mat->Height(), mat->Width());
mat->EliminateCols(ess_trial_tdof_marker, *mat_e);
for (int i=0; i<test_tdof_list.Size(); ++i)
{
mat->EliminateRow(test_tdof_list[i]);
}
mat_e->Finalize();
A.Reset(mat, false);
}
void MixedBilinearForm::FormRectangularLinearSystem(const Array<int>
&trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A,
Vector &X, Vector &B)
{
if (ext)
{
ext->FormRectangularLinearSystem(trial_tdof_list, test_tdof_list, x, b, A, X,
B);
return;
}
const Operator *Po = this->GetOutputProlongation();
const Operator *Ri = this->GetRestriction();
InitTVectors(Po, Ri, x, b, X, B);
if (!mat_e)
{
FormRectangularSystemMatrix(trial_tdof_list, test_tdof_list,
A); // Set A = mat_e
}
// Eliminate essential BCs with B -= Ab xb
mat_e->AddMult(X, B, -1.0);
B.SetSubVector(test_tdof_list, 0.0);
}
void MixedBilinearForm::Update()
{
delete mat;
mat = NULL;
delete mat_e;
mat_e = NULL;
height = test_fes->GetVSize();
width = trial_fes->GetVSize();
if (ext) { ext->Update(); }
}
MixedBilinearForm::~MixedBilinearForm()
{
if (mat) { delete mat; }
if (mat_e) { delete mat_e; }
if (!extern_bfs)
{
int i;
@@ -1679,7 +1489,6 @@ MixedBilinearForm::~MixedBilinearForm()
for (i = 0; i < tfbfi.Size(); i++) { delete tfbfi[i]; }
for (i = 0; i < btfbfi.Size(); i++) { delete btfbfi[i]; }
}
delete ext;
}
+6 -123
View File
@@ -58,7 +58,7 @@ protected:
/// FE space on which the form lives. Not owned.
FiniteElementSpace *fes;
/// The assembly level of the form (full, partial, etc.)
/// The form assembly level (full, partial, etc.)
AssemblyLevel assembly;
/// Element batch size used in the form action (1, 8, num_elems, etc.)
int batch;
@@ -153,9 +153,6 @@ public:
/** This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level);
/// Get the assembly level
AssemblyLevel GetAssemblyLevel() {return assembly;}
/** Enable the use of static condensation. For details see the description
for class StaticCondensation in fem/staticcond.hpp This method should be
called before assembly. If the number of unknowns after static
@@ -230,7 +227,7 @@ public:
virtual const double &Elem(int i, int j) const;
/// Matrix vector multiplication.
virtual void Mult(const Vector &x, Vector &y) const;
virtual void Mult(const Vector &x, Vector &y) const { mat->Mult(x, y); }
void FullMult(const Vector &x, Vector &y) const
{ mat->Mult(x, y); mat_e->AddMult(x, y); }
@@ -322,26 +319,12 @@ public:
/// Assembles the form i.e. sums over all domain/bdr integrators.
void Assemble(int skip_zeros = 1);
/** @brief Assemble the diagonal of the bilinear form into diag
For adaptively refined meshes, this returns P^T d_e, where d_e is the
locally assembled diagonal on each element and P^T is the transpose of
the conforming prolongation. In general this is not the correct diagonal
for an AMR mesh. */
void AssembleDiagonal(Vector &diag) const;
/// Get the finite element space prolongation matrix
virtual const Operator *GetProlongation() const
{ return fes->GetConformingProlongation(); }
/// Get the finite element space restriction matrix
virtual const Operator *GetRestriction() const
{ return fes->GetConformingRestriction(); }
/// Get the output finite element space prolongation matrix
virtual const Operator *GetOutputProlongation() const
{ return GetProlongation(); }
/// Get the output finite element space restriction matrix
virtual const Operator *GetOutputRestriction() const
{ return GetRestriction(); }
/** @brief Form the linear system A X = B, corresponding to this bilinear
form and the linear form @a b(.). */
@@ -542,9 +525,6 @@ public:
/// Sets diagonal policy used upon construction of the linear system
void SetDiagonalPolicy(DiagonalPolicy policy);
/// Indicate that integrators are not owned by the BilinearForm
void UseExternalIntegrators() { extern_bfs = 1; };
/// Destroys bilinear form.
virtual ~BilinearForm();
};
@@ -569,17 +549,10 @@ class MixedBilinearForm : public Matrix
{
protected:
SparseMatrix *mat; ///< Owned.
SparseMatrix *mat_e; ///< Owned.
FiniteElementSpace *trial_fes, ///< Not owned
*test_fes; ///< Not owned
/// The form assembly level (full, partial, etc.)
AssemblyLevel assembly;
/** Extension for supporting Full Assembly (FA), Element Assembly (EA),
Partial Assembly (PA), or Matrix Free assembly (MF). */
MixedBilinearFormExtension *ext;
/** @brief Indicates the BilinearFormIntegrator%s stored in #dbfi, #bbfi,
#tfbfi and #btfbfi are owned by another MixedBilinearForm. */
int extern_bfs;
@@ -634,13 +607,16 @@ public:
virtual const double &Elem(int i, int j) const;
virtual void Mult(const Vector & x, Vector & y) const;
virtual void AddMult(const Vector & x, Vector & y,
const double a = 1.0) const;
virtual void MultTranspose(const Vector & x, Vector & y) const;
virtual void AddMultTranspose(const Vector & x, Vector & y,
const double a = 1.0) const;
virtual void MultTranspose(const Vector & x, Vector & y) const
{ y = 0.0; AddMultTranspose (x, y); }
virtual MatrixInverse *Inverse() const;
virtual void Finalize(int skip_zeros = 1);
@@ -700,28 +676,8 @@ public:
void operator=(const double a) { *mat = a; }
/// Set the desired assembly level. The default is AssemblyLevel::FULL.
/** This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level);
void Assemble(int skip_zeros = 1);
/// Get the input finite element space prolongation matrix
virtual const Operator *GetProlongation() const
{ return trial_fes->GetProlongationMatrix(); }
/// Get the input finite element space restriction matrix
virtual const Operator *GetRestriction() const
{ return trial_fes->GetRestrictionMatrix(); }
/// Get the test finite element space prolongation matrix
virtual const Operator *GetOutputProlongation() const
{ return test_fes->GetProlongationMatrix(); }
/// Get the test finite element space restriction matrix
virtual const Operator *GetOutputRestriction() const
{ return test_fes->GetRestrictionMatrix(); }
/** For partially conforming trial and/or test FE spaces, complete the
assembly process by performing A := P2^t A P1 where A is the internal
sparse matrix; P1 and P2 are the conforming prolongation matrices of the
@@ -783,81 +739,8 @@ public:
virtual void EliminateTestDofs(const Array<int> &bdr_attr_is_ess);
/** @brief Return in @a A a parallel (on truedofs) version of this operator.
This returns the same operator as FormRectangularLinearSystem(), but does
without the transformations of the right-hand side. */
void FormRectangularSystemMatrix(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorHandle &A);
/** @brief Form the column-constrained linear system matrix A.
See FormRectangularSystemMatrix() for details.
Version of the method FormRectangularSystemMatrix() where the system matrix is
returned in the variable @a A, of type OpType, holding a *reference* to
the system matrix (created with the method OpType::MakeRef()). The
reference will be invalidated when SetOperatorType(), Update(), or the
destructor is called.
Currently, this method can be used only with AssemblyLevel::FULL. */
template <typename OpType>
void FormRectangularSystemMatrix(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list, OpType &A)
{
OperatorHandle Ah;
FormRectangularSystemMatrix(trial_tdof_list, test_tdof_list, Ah);
OpType *A_ptr = Ah.Is<OpType>();
MFEM_VERIFY(A_ptr, "invalid OpType used");
A.MakeRef(*A_ptr);
}
/** @brief Form the linear system A X = B, corresponding to this mixed bilinear
form and the linear form @a b(.).
Return in @a A a *reference* to the system matrix that is column-constrained.
The reference will be invalidated when SetOperatorType(), Update(), or the
destructor is called. */
void FormRectangularLinearSystem(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B);
/** @brief Form the linear system A X = B, corresponding to this bilinear
form and the linear form @a b(.).
Version of the method FormRectangularLinearSystem() where the system matrix is
returned in the variable @a A, of type OpType, holding a *reference* to
the system matrix (created with the method OpType::MakeRef()). The
reference will be invalidated when SetOperatorType(), Update(), or the
destructor is called.
Currently, this method can be used only with AssemblyLevel::FULL. */
template <typename OpType>
void FormRectangularLinearSystem(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OpType &A, Vector &X, Vector &B)
{
OperatorHandle Ah;
FormRectangularLinearSystem(trial_tdof_list, test_tdof_list, x, b, Ah, X, B);
OpType *A_ptr = Ah.Is<OpType>();
MFEM_VERIFY(A_ptr, "invalid OpType used");
A.MakeRef(*A_ptr);
}
void Update();
/// Return the trial FE space associated with the BilinearForm.
FiniteElementSpace *TrialFESpace() { return trial_fes; }
/// Read-only access to the associated trial FiniteElementSpace.
const FiniteElementSpace *TrialFESpace() const { return trial_fes; }
/// Return the test FE space associated with the BilinearForm.
FiniteElementSpace *TestFESpace() { return test_fes; }
/// Read-only access to the associated test FiniteElementSpace.
const FiniteElementSpace *TestFESpace() const { return test_fes; }
virtual ~MixedBilinearForm();
};
+17 -240
View File
@@ -14,7 +14,6 @@
#include "../general/forall.hpp"
#include "bilinearform.hpp"
#include "libceed/ceed.hpp"
namespace mfem
{
@@ -39,8 +38,7 @@ const Operator *BilinearFormExtension::GetRestriction() const
// Data and methods for partially-assembled bilinear forms
PABilinearFormExtension::PABilinearFormExtension(BilinearForm *form)
: BilinearFormExtension(form),
trialFes(a->FESpace()),
testFes(a->FESpace())
trialFes(a->FESpace()), testFes(a->FESpace())
{
elem_restrict_lex = trialFes->GetElementRestriction(
ElementDofOrdering::LEXICOGRAPHIC);
@@ -62,31 +60,6 @@ void PABilinearFormExtension::Assemble()
}
}
void PABilinearFormExtension::AssembleDiagonal(Vector &y) const
{
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int iSz = integrators.Size();
if (elem_restrict_lex)
{
localY = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AssembleDiagonalPA(localY);
}
elem_restrict_lex->MultTranspose(localY, y);
}
else
{
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AssembleDiagonalPA(y);
}
}
}
void PABilinearFormExtension::Update()
{
FiniteElementSpace *fes = a->FESpace();
@@ -105,9 +78,12 @@ void PABilinearFormExtension::Update()
void PABilinearFormExtension::FormSystemMatrix(const Array<int> &ess_tdof_list,
OperatorHandle &A)
{
Operator *oper;
Operator::FormSystemOperator(ess_tdof_list, oper);
A.Reset(oper); // A will own oper
const Operator* trialP = trialFes->GetProlongationMatrix();
const Operator* testP = testFes->GetProlongationMatrix();
Operator *rap = this;
if (trialP) { rap = new RAPOperator(*testP, *this, *trialP); }
const bool own_A = (rap!=this);
A.Reset(new ConstrainedOperator(rap, ess_tdof_list, own_A));
}
void PABilinearFormExtension::FormLinearSystem(const Array<int> &ess_tdof_list,
@@ -126,16 +102,7 @@ void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int iSz = integrators.Size();
if (DeviceCanUseCeed() || !elem_restrict_lex)
{
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(x, y);
}
}
else
if (elem_restrict_lex)
{
elem_restrict_lex->Mult(x, localX);
localY = 0.0;
@@ -145,6 +112,15 @@ void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
}
elem_restrict_lex->MultTranspose(localY, y);
}
else
{
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(x, y);
}
}
}
void PABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
@@ -172,203 +148,4 @@ void PABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
}
}
MixedBilinearFormExtension::MixedBilinearFormExtension(MixedBilinearForm *form)
: Operator(form->Height(), form->Width()), a(form)
{
// empty
}
const Operator *MixedBilinearFormExtension::GetProlongation() const
{
return a->GetProlongation();
}
const Operator *MixedBilinearFormExtension::GetRestriction() const
{
return a->GetRestriction();
}
const Operator *MixedBilinearFormExtension::GetOutputProlongation() const
{
return a->GetOutputProlongation();
}
const Operator *MixedBilinearFormExtension::GetOutputRestriction() const
{
return a->GetOutputRestriction();
}
// Data and methods for partially-assembled bilinear forms
PAMixedBilinearFormExtension::PAMixedBilinearFormExtension(
MixedBilinearForm *form)
: MixedBilinearFormExtension(form),
trialFes(form->TrialFESpace()),
testFes(form->TestFESpace()),
elem_restrict_trial(NULL),
elem_restrict_test(NULL)
{
Update();
}
void PAMixedBilinearFormExtension::Assemble()
{
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int integratorCount = integrators.Size();
for (int i = 0; i < integratorCount; ++i)
{
integrators[i]->AssemblePA(*trialFes, *testFes);
}
}
void PAMixedBilinearFormExtension::Update()
{
trialFes = a->TrialFESpace();
testFes = a->TestFESpace();
height = testFes->GetVSize();
width = trialFes->GetVSize();
elem_restrict_trial = trialFes->GetElementRestriction(
ElementDofOrdering::LEXICOGRAPHIC);
elem_restrict_test = testFes->GetElementRestriction(
ElementDofOrdering::LEXICOGRAPHIC);
if (elem_restrict_trial)
{
localTrial.UseDevice(true);
localTrial.SetSize(elem_restrict_trial->Height(), Device::GetMemoryType());
}
if (elem_restrict_test)
{
localTest.UseDevice(true); // ensure 'localY = 0.0' is done on device
localTest.SetSize(elem_restrict_test->Height(), Device::GetMemoryType());
}
}
void PAMixedBilinearFormExtension::FormRectangularSystemOperator(
const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorHandle &A)
{
Operator * oper;
Operator::FormRectangularSystemOperator(trial_tdof_list, test_tdof_list, oper);
A.Reset(oper); // A will own oper
}
void PAMixedBilinearFormExtension::FormRectangularLinearSystem(
const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A,
Vector &X, Vector &B)
{
Operator *oper;
Operator::FormRectangularLinearSystem(trial_tdof_list, test_tdof_list, x, b,
oper, X, B);
A.Reset(oper); // A will own oper
}
void PAMixedBilinearFormExtension::SetupMultInputs(const Operator
*elem_restrict_x,
const Vector &x,
Vector &localX,
const Operator *elem_restrict_y,
Vector &y,
Vector &localY,
const double c) const
{
// * G operation: localX = c*local(x)
if (elem_restrict_x)
{
elem_restrict_x->Mult(x, localX);
if (c != 1.0)
{
localX *= c;
}
}
else
{
if (c == 1.0)
{
localX.SyncAliasMemory(x);
}
else
{
localX.Set(c, x);
}
}
if (elem_restrict_y)
{
localY = 0.0;
}
else
{
y.UseDevice(true);
localY.SyncAliasMemory(y);
}
}
void PAMixedBilinearFormExtension::Mult(const Vector &x, Vector &y) const
{
y = 0.0;
AddMult(x, y);
}
void PAMixedBilinearFormExtension::AddMult(const Vector &x, Vector &y,
const double c) const
{
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int iSz = integrators.Size();
// * G operation
SetupMultInputs(elem_restrict_trial, x, localTrial,
elem_restrict_test, y, localTest, c);
// * B^TDB operation
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(localTrial, localTest);
}
// * G^T operation
if (elem_restrict_test)
{
tempY.SetSize(y.Size());
elem_restrict_test->MultTranspose(localTest, tempY);
y += tempY;
}
}
void PAMixedBilinearFormExtension::MultTranspose(const Vector &x,
Vector &y) const
{
y = 0.0;
AddMultTranspose(x, y);
}
void PAMixedBilinearFormExtension::AddMultTranspose(const Vector &x, Vector &y,
const double c) const
{
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int iSz = integrators.Size();
// * G operation
SetupMultInputs(elem_restrict_test, x, localTest,
elem_restrict_trial, y, localTrial, c);
// * B^TD^TB operation
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultTransposePA(localTest, localTrial);
}
// * G^T operation
if (elem_restrict_trial)
{
tempY.SetSize(y.Size());
elem_restrict_trial->MultTranspose(localTrial, tempY);
y += tempY;
}
}
} // namespace mfem
-99
View File
@@ -20,7 +20,6 @@ namespace mfem
{
class BilinearForm;
class MixedBilinearForm;
/** @brief Class extending the BilinearForm class to support the different
@@ -43,10 +42,6 @@ public:
virtual const Operator *GetRestriction() const;
virtual void Assemble() = 0;
virtual void AssembleDiagonal(Vector &diag) const
{
MFEM_ABORT("AssembleDiagonal not implemented for this assembly level!");
}
virtual void FormSystemMatrix(const Array<int> &ess_tdof_list,
OperatorHandle &A) = 0;
virtual void FormLinearSystem(const Array<int> &ess_tdof_list,
@@ -108,7 +103,6 @@ public:
PABilinearFormExtension(BilinearForm*);
void Assemble();
void AssembleDiagonal(Vector &diag) const;
void FormSystemMatrix(const Array<int> &ess_tdof_list, OperatorHandle &A);
void FormLinearSystem(const Array<int> &ess_tdof_list,
Vector &x, Vector &b,
@@ -120,7 +114,6 @@ public:
void Update();
};
/// Data and methods for matrix-free bilinear forms
class MFBilinearFormExtension : public BilinearFormExtension
{
@@ -141,98 +134,6 @@ public:
~MFBilinearFormExtension() {}
};
/** @brief Class extending the MixedBilinearForm class to support the different
AssemblyLevel%s. */
class MixedBilinearFormExtension : public Operator
{
protected:
MixedBilinearForm *a; ///< Not owned
public:
MixedBilinearFormExtension(MixedBilinearForm *form);
virtual MemoryClass GetMemoryClass() const
{ return Device::GetMemoryClass(); }
/// Get the finite element space prolongation matrix
virtual const Operator *GetProlongation() const;
/// Get the finite element space restriction matrix
virtual const Operator *GetRestriction() const;
/// Get the output finite element space restriction matrix
virtual const Operator *GetOutputProlongation() const;
/// Get the output finite element space restriction matrix
virtual const Operator *GetOutputRestriction() const;
virtual void Assemble() = 0;
virtual void FormRectangularSystemOperator(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorHandle &A) = 0;
virtual void FormRectangularLinearSystem(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B) = 0;
virtual void AddMult(const Vector &x, Vector &y, const double c=1.0) const = 0;
virtual void AddMultTranspose(const Vector &x, Vector &y,
const double c=1.0) const = 0;
virtual void Update() = 0;
};
/// Data and methods for partially-assembled mixed bilinear forms
class PAMixedBilinearFormExtension : public MixedBilinearFormExtension
{
protected:
const FiniteElementSpace *trialFes, *testFes; // Not owned
mutable Vector localTrial, localTest, tempY;
const Operator *elem_restrict_trial; // Not owned
const Operator *elem_restrict_test; // Not owned
private:
/// Helper function to set up inputs/outputs for Mult or MultTranspose
void SetupMultInputs(const Operator *elem_restrict_x,
const Vector &x, Vector &localX,
const Operator *elem_restrict_y,
Vector &y, Vector &localY, const double c) const;
public:
PAMixedBilinearFormExtension(MixedBilinearForm *form);
/// Partial assembly of all internal integrators
void Assemble();
/**
@brief Setup OperatorHandle A to contain constrained linear operator
OperatorHandle A contains matrix-free constrained operator formed for RAP
system where ess_tdof_list are in trial space and eliminated from
"columns" of A.
*/
void FormRectangularSystemOperator(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorHandle &A);
/**
Setup OperatorHandle A to contain constrained linear operator and
eliminate columns corresponding to essential dofs from system,
updating RHS B vector with the results.
*/
void FormRectangularLinearSystem(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B);
/// y = A*x
void Mult(const Vector &x, Vector &y) const;
/// y += c*A*x
void AddMult(const Vector &x, Vector &y, const double c=1.0) const;
/// y = A^T*x
void MultTranspose(const Vector &x, Vector &y) const;
/// y += c*A^T*x
void AddMultTranspose(const Vector &x, Vector &y, const double c=1.0) const;
/// Update internals for when a new MixedBilinearForm is given to this class
void Update();
};
}
#endif
+90 -181
View File
@@ -22,23 +22,10 @@ namespace mfem
void BilinearFormIntegrator::AssemblePA(const FiniteElementSpace&)
{
mfem_error ("BilinearFormIntegrator::AssemblePA(...)\n"
mfem_error ("BilinearFormIntegrator::Assemble (...)\n"
" is not implemented for this class.");
}
void BilinearFormIntegrator::AssemblePA(const FiniteElementSpace&,
const FiniteElementSpace&)
{
mfem_error ("BilinearFormIntegrator::AssemblePA(...)\n"
" is not implemented for this class.");
}
void BilinearFormIntegrator::AssembleDiagonalPA(Vector &)
{
MFEM_ABORT("BilinearFormIntegrator::AssembleDiagonalPA (...)\n"
" is not implemented for this class.");
}
void BilinearFormIntegrator::AddMultPA(const Vector &, Vector &) const
{
mfem_error ("BilinearFormIntegrator::MultAssembled (...)\n"
@@ -393,72 +380,6 @@ void MixedScalarVectorIntegrator::AssembleElementMatrix2(
}
void GradientIntegrator::AssembleElementMatrix2(
const FiniteElement &trial_fe, const FiniteElement &test_fe,
ElementTransformation &Trans, DenseMatrix &elmat)
{
int dim = test_fe.GetDim();
int trial_dof = trial_fe.GetDof();
int test_dof = test_fe.GetDof();
double c;
Vector d_col;
dshape.SetSize(trial_dof, dim);
gshape.SetSize(trial_dof, dim);
Jadj.SetSize(dim);
shape.SetSize(test_dof);
elmat.SetSize(dim * test_dof, trial_dof);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(trial_fe, test_fe,
Trans);
elmat = 0.0;
elmat_comp.SetSize(test_dof, trial_dof);
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
trial_fe.CalcDShape(ip, dshape);
test_fe.CalcShape(ip, shape);
Trans.SetIntPoint(&ip);
CalcAdjugate(Trans.Jacobian(), Jadj);
Mult(dshape, Jadj, gshape);
c = ip.weight;
if (Q)
{
c *= Q->Eval(Trans, ip);
}
shape *= c;
for (int d = 0; d < dim; ++d)
{
gshape.GetColumnReference(d, d_col);
MultVWt(shape, d_col, elmat_comp);
for (int jj = 0; jj < trial_dof; ++jj)
{
for (int ii = 0; ii < test_dof; ++ii)
{
elmat(d * test_dof + ii, jj) += elmat_comp(ii, jj);
}
}
}
}
}
const IntegrationRule &GradientIntegrator::GetRule(const FiniteElement
&trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans)
{
int order = Trans.OrderGrad(&trial_fe) + test_fe.GetOrder() + Trans.OrderJ();
return IntRules.Get(trial_fe.GetGeomType(), order);
}
void DiffusionIntegrator::AssembleElementMatrix
( const FiniteElement &el, ElementTransformation &Trans,
DenseMatrix &elmat )
@@ -901,16 +822,6 @@ void BoundaryMassIntegrator::AssembleFaceMatrix(
}
const IntegrationRule &ConvectionIntegrator::GetRule(const FiniteElement
&trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans)
{
int order = Trans.OrderGrad(&trial_fe) + Trans.Order() + test_fe.GetOrder();
return IntRules.Get(trial_fe.GetGeomType(), order);
}
void ConvectionIntegrator::AssembleElementMatrix(
const FiniteElement &el, ElementTransformation &Trans, DenseMatrix &elmat)
{
@@ -1841,26 +1752,81 @@ void VectorFEMassIntegrator::AssembleElementMatrix2(
const FiniteElement &trial_fe, const FiniteElement &test_fe,
ElementTransformation &Trans, DenseMatrix &elmat)
{
if (test_fe.GetRangeType() == FiniteElement::SCALAR
&& trial_fe.GetRangeType() == FiniteElement::VECTOR)
if ( test_fe.GetRangeType() == FiniteElement::SCALAR && VQ )
{
// assume test_fe is scalar FE and trial_fe is vector FE
int dim = test_fe.GetDim();
int trial_dof = trial_fe.GetDof();
int test_dof = test_fe.GetDof();
double w;
double Kv;
if (MQ)
mfem_error("VectorFEMassIntegrator::AssembleElementMatrix2(...)\n"
" is not implemented for tensor materials");
#ifdef MFEM_THREAD_SAFE
DenseMatrix trial_vshape(trial_dof, dim);
Vector shape(test_dof);
Vector D(dim);
#else
trial_vshape.SetSize(trial_dof, dim);
shape.SetSize(test_dof);
D.SetSize(dim);
#endif
elmat.SetSize (test_dof, trial_dof);
const IntegrationRule *ir = IntRule;
if (ir == NULL)
{
int order = (Trans.OrderW() + test_fe.GetOrder() + trial_fe.GetOrder());
ir = &IntRules.Get(test_fe.GetGeomType(), order);
}
elmat = 0.0;
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
Trans.SetIntPoint (&ip);
trial_fe.CalcVShape(Trans, trial_vshape);
test_fe.CalcShape(ip, shape);
w = ip.weight * Trans.Weight();
VQ->Eval(D, Trans, ip);
D *= w;
for (int d = 0; d < dim; d++)
{
for (int j = 0; j < test_dof; j++)
{
for (int k = 0; k < trial_dof; k++)
{
elmat(j, k) += D[d] * shape(j) * trial_vshape(k, d);
}
}
}
}
}
else if ( test_fe.GetRangeType() == FiniteElement::SCALAR )
{
// assume test_fe is scalar FE and trial_fe is vector FE
int dim = test_fe.GetDim();
int trial_dof = trial_fe.GetDof();
int test_dof = test_fe.GetDof();
double w;
if (VQ || MQ)
mfem_error("VectorFEMassIntegrator::AssembleElementMatrix2(...)\n"
" is not implemented for vector/tensor permeability");
#ifdef MFEM_THREAD_SAFE
DenseMatrix trial_vshape(trial_dof, dim);
Vector shape(test_dof);
Vector D(VQ ? VQ->GetVDim() : 0);
DenseMatrix K(MQ ? MQ->GetVDim() : 0, MQ ? MQ->GetVDim() : 0);
#else
trial_vshape.SetSize(trial_dof, dim);
shape.SetSize(test_dof);
D.SetSize(VQ ? VQ->GetVDim() : 0);
K.SetSize(MQ ? MQ->GetVDim() : 0, MQ ? MQ->GetVDim() : 0);
#endif
elmat.SetSize (dim*test_dof, trial_dof);
@@ -1883,64 +1849,24 @@ void VectorFEMassIntegrator::AssembleElementMatrix2(
test_fe.CalcShape(ip, shape);
w = ip.weight * Trans.Weight();
if (VQ)
if (Q)
{
VQ->Eval(D, Trans, ip);
D *= w;
for (int d = 0; d < dim; d++)
{
for (int j = 0; j < test_dof; j++)
{
for (int k = 0; k < trial_dof; k++)
{
elmat(d * test_dof + j, k) +=
shape(j) * D(d) * trial_vshape(k, d);
}
}
}
w *= Q -> Eval (Trans, ip);
}
else if (MQ)
for (int d = 0; d < dim; d++)
{
MQ->Eval(K, Trans, ip);
K *= w;
for (int d = 0; d < dim; d++)
for (int j = 0; j < test_dof; j++)
{
for (int j = 0; j < test_dof; j++)
for (int k = 0; k < trial_dof; k++)
{
for (int k = 0; k < trial_dof; k++)
{
Kv = 0.0;
for (int vd = 0; vd < dim; vd++)
{
Kv += K(d, vd) * trial_vshape(k, vd);
}
elmat(d * test_dof + j, k) += shape(j) * Kv;
}
}
}
}
else
{
if (Q)
{
w *= Q->Eval(Trans, ip);
}
for (int d = 0; d < dim; d++)
{
for (int j = 0; j < test_dof; j++)
{
for (int k = 0; k < trial_dof; k++)
{
elmat(d * test_dof + j, k) +=
w * shape(j) * trial_vshape(k, d);
}
elmat(d * test_dof + j, k) += w * shape(j) * trial_vshape(k, d);
}
}
}
}
}
else if (test_fe.GetRangeType() == FiniteElement::VECTOR
&& trial_fe.GetRangeType() == FiniteElement::VECTOR)
else
{
// assume both test_fe and trial_fe are vector FE
int dim = test_fe.GetDim();
@@ -1948,18 +1874,17 @@ void VectorFEMassIntegrator::AssembleElementMatrix2(
int test_dof = test_fe.GetDof();
double w;
if (VQ || MQ)
mfem_error("VectorFEMassIntegrator::AssembleElementMatrix2(...)\n"
" is not implemented for vector/tensor permeability");
#ifdef MFEM_THREAD_SAFE
DenseMatrix trial_vshape(trial_dof, dim);
DenseMatrix test_vshape(test_dof,dim);
Vector D(VQ ? VQ->GetVDim() : 0);
DenseMatrix K(MQ ? MQ->GetVDim() : 0, MQ ? MQ->GetVDim() : 0);
#else
trial_vshape.SetSize(trial_dof, dim);
test_vshape.SetSize(test_dof,dim);
D.SetSize(VQ ? VQ->GetVDim() : 0);
K.SetSize(MQ ? MQ->GetVDim() : 0, MQ ? MQ->GetVDim() : 0);
#endif
DenseMatrix tmp(trial_vshape.Height(), K.Width());
elmat.SetSize (test_dof, trial_dof);
@@ -1981,34 +1906,23 @@ void VectorFEMassIntegrator::AssembleElementMatrix2(
test_fe.CalcVShape(Trans, test_vshape);
w = ip.weight * Trans.Weight();
if (MQ)
if (Q)
{
MQ->Eval(K, Trans, ip);
K *= w;
Mult(test_vshape,K,tmp);
AddMultABt(tmp,trial_vshape,elmat);
w *= Q -> Eval (Trans, ip);
}
else if (VQ)
for (int d = 0; d < dim; d++)
{
VQ->Eval(D, Trans, ip);
D *= w;
AddMultADBt(test_vshape,D,trial_vshape,elmat);
}
else
{
if (Q)
for (int j = 0; j < test_dof; j++)
{
w *= Q -> Eval (Trans, ip);
for (int k = 0; k < trial_dof; k++)
{
elmat(j, k) += w * test_vshape(j, d) * trial_vshape(k, d);
}
}
AddMult_a_ABt(w,test_vshape,trial_vshape,elmat);
}
}
}
else
{
mfem_error("VectorFEMassIntegrator::AssembleElementMatrix2(...)\n"
" is not implemented for given trial and test bases.");
}
}
void VectorDivergenceIntegrator::AssembleElementMatrix2(
@@ -2030,8 +1944,12 @@ void VectorDivergenceIntegrator::AssembleElementMatrix2(
elmat.SetSize (test_dof, dim*trial_dof);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(trial_fe, test_fe,
Trans);
const IntegrationRule *ir = IntRule;
if (ir == NULL)
{
int order = Trans.OrderGrad(&trial_fe) + test_fe.GetOrder();
ir = &IntRules.Get(trial_fe.GetGeomType(), order);
}
elmat = 0.0;
@@ -2061,15 +1979,6 @@ void VectorDivergenceIntegrator::AssembleElementMatrix2(
}
}
const IntegrationRule &VectorDivergenceIntegrator::GetRule(
const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans)
{
int order = Trans.OrderGrad(&trial_fe) + test_fe.GetOrder() + Trans.OrderJ();
return IntRules.Get(trial_fe.GetGeomType(), order);
}
void DivDivIntegrator::AssembleElementMatrix(
const FiniteElement &el,
+11 -207
View File
@@ -15,7 +15,6 @@
#include "../config/config.hpp"
#include "nonlininteg.hpp"
#include "fespace.hpp"
#include "libceed/ceed.hpp"
namespace mfem
{
@@ -40,18 +39,10 @@ public:
// make sense for the action of the nonlinear operator (but they all make
// sense for its Jacobian).
using NonlinearFormIntegrator::AssemblePA;
/// Method defining partial assembly.
/** The result of the partial assembly is stored internally so that it can be
used later in the methods AddMultPA() and AddMultTransposePA(). */
virtual void AssemblePA(const FiniteElementSpace &fes);
/** Used with BilinearFormIntegrators that have different spaces. */
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
/// Assemble diagonal and add it to Vector @a diag.
virtual void AssembleDiagonalPA(Vector &diag);
/// Method for partially assembled action.
/** Perform the action of integrator on the input @a x and add the result to
@@ -1667,55 +1658,6 @@ protected:
}
};
/** Class for integrating the bilinear form a(u,v) := (Q grad u, v) where Q is a
scalar coefficient, and v is a vector with components v_i in the same space
as u. */
class GradientIntegrator : public BilinearFormIntegrator
{
protected:
Coefficient *Q;
private:
Vector shape;
DenseMatrix dshape;
DenseMatrix gshape;
DenseMatrix Jadj;
DenseMatrix elmat_comp;
// PA extension
Vector pa_data;
const DofToQuad *trial_maps, *test_maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq;
int trial_dofs1D, test_dofs1D, quad1D;
public:
GradientIntegrator() :
Q{NULL}, trial_maps{NULL}, test_maps{NULL}, geom{NULL}
{ }
GradientIntegrator(Coefficient *_q) :
Q{_q}, trial_maps{NULL}, test_maps{NULL}, geom{NULL}
{ }
GradientIntegrator(Coefficient &q) :
Q{&q}, trial_maps{NULL}, test_maps{NULL}, geom{NULL}
{ }
virtual void AssembleElementMatrix2(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans,
DenseMatrix &elmat);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
virtual void AddMultTransposePA(const Vector &x, Vector &y) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans);
};
/** Class for integrating the bilinear form a(u,v) := (Q grad u, grad v) where Q
can be a scalar or a matrix coefficient. */
class DiffusionIntegrator: public BilinearFormIntegrator
@@ -1732,60 +1674,22 @@ private:
#endif
// PA extension
const FiniteElementSpace *fespace;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, dofs1D, quad1D;
Vector pa_data;
#ifdef MFEM_USE_CEED
// CEED extension
CeedData* ceedDataPtr;
#endif
public:
/// Construct a diffusion integrator with coefficient Q = 1
DiffusionIntegrator()
{
Q = NULL;
MQ = NULL;
maps = NULL;
geom = NULL;
#ifdef MFEM_USE_CEED
ceedDataPtr = NULL;
#endif
}
DiffusionIntegrator() { Q = NULL; MQ = NULL; maps = NULL; geom = NULL; }
/// Construct a diffusion integrator with a scalar coefficient q
DiffusionIntegrator(Coefficient &q)
: Q(&q)
{
MQ = NULL;
maps = NULL;
geom = NULL;
#ifdef MFEM_USE_CEED
ceedDataPtr = NULL;
#endif
}
: Q(&q) { MQ = NULL; maps = NULL; geom = NULL; }
/// Construct a diffusion integrator with a matrix coefficient q
DiffusionIntegrator(MatrixCoefficient &q)
: MQ(&q)
{
Q = NULL;
maps = NULL;
geom = NULL;
#ifdef MFEM_USE_CEED
ceedDataPtr = NULL;
#endif
}
virtual ~DiffusionIntegrator()
{
#ifdef MFEM_USE_CEED
delete ceedDataPtr;
#endif
}
: MQ(&q) { Q = NULL; maps = NULL; geom = NULL; }
/** Given a particular Finite Element
computes the element stiffness matrix elmat. */
@@ -1813,18 +1717,12 @@ public:
ElementTransformation &Trans,
Vector &flux, Vector *d_energy = NULL);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AssembleDiagonalPA(Vector &diag);
virtual void AssemblePA(const FiniteElementSpace&);
virtual void AddMultPA(const Vector&, Vector&) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe);
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
};
/** Class for local mass matrix assembling a(u,v) := (Q u, v) */
@@ -1836,46 +1734,19 @@ protected:
#endif
Coefficient *Q;
// PA extension
const FiniteElementSpace *fespace;
Vector pa_data;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq, dofs1D, quad1D;
#ifdef MFEM_USE_CEED
// CEED extension
CeedData* ceedDataPtr;
#endif
public:
MassIntegrator(const IntegrationRule *ir = NULL)
: BilinearFormIntegrator(ir)
{
Q = NULL;
maps = NULL;
geom = NULL;
#ifdef MFEM_USE_CEED
ceedDataPtr = NULL;
#endif
}
: BilinearFormIntegrator(ir) { Q = NULL; maps = NULL; geom = NULL; }
/// Construct a mass integrator with coefficient q
MassIntegrator(Coefficient &q, const IntegrationRule *ir = NULL)
: BilinearFormIntegrator(ir), Q(&q)
{
maps = NULL;
geom = NULL;
#ifdef MFEM_USE_CEED
ceedDataPtr = NULL;
#endif
}
: BilinearFormIntegrator(ir), Q(&q) { maps = NULL; geom = NULL; }
virtual ~MassIntegrator()
{
#ifdef MFEM_USE_CEED
delete ceedDataPtr;
#endif
}
/** Given a particular Finite Element
computes the element mass matrix elmat. */
virtual void AssembleElementMatrix(const FiniteElement &el,
@@ -1886,19 +1757,13 @@ public:
ElementTransformation &Trans,
DenseMatrix &elmat);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AssembleDiagonalPA(Vector &diag);
virtual void AssemblePA(const FiniteElementSpace&);
virtual void AddMultPA(const Vector&, Vector&) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans);
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
};
class BoundaryMassIntegrator : public MassIntegrator
@@ -1927,29 +1792,12 @@ private:
Vector shape, vec2, BdFidxT;
#endif
// PA extension
Vector pa_data;
Vector coeff;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq, dofs1D, quad1D;
public:
ConvectionIntegrator(VectorCoefficient &q, double a = 1.0)
: Q(&q) { alpha = a; }
virtual void AssembleElementMatrix(const FiniteElement &,
ElementTransformation &,
DenseMatrix &);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace&);
virtual void AddMultPA(const Vector&, Vector&) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans);
};
/// alpha (q . grad u, v) using the "group" FE discretization
@@ -1987,11 +1835,6 @@ protected:
Coefficient *Q;
VectorCoefficient *VQ;
MatrixCoefficient *MQ;
// PA extension
Vector pa_data;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq, dofs1D, quad1D;
public:
/// Construct an integrator with coefficient 1.0
@@ -2023,9 +1866,6 @@ public:
const FiniteElement &test_fe,
ElementTransformation &Trans,
DenseMatrix &elmat);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
};
@@ -2198,10 +2038,7 @@ public:
const Vector &elfun);
};
/** Integrator for (Q u, v), where Q is an optional coefficient (of type scalar,
vector (diagonal matrix), or matrix), trial function u is in H(Curl) or
H(Div), and test function v is in H(Curl), H(Div), or v=(v1,...,vn), where
vi are in H1. */
/// Integrator for (Q u, v) for VectorFiniteElements
class VectorFEMassIntegrator: public BilinearFormIntegrator
{
private:
@@ -2212,7 +2049,6 @@ private:
Vector shape;
Vector D;
DenseMatrix K;
DenseMatrix partelmat;
DenseMatrix test_vshape;
DenseMatrix trial_vshape;
#endif
@@ -2253,39 +2089,16 @@ private:
DenseMatrix dshape;
DenseMatrix gshape;
DenseMatrix Jadj;
// PA extension
Vector pa_data;
const DofToQuad *trial_maps, *test_maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq;
int trial_dofs1D, test_dofs1D, quad1D;
public:
VectorDivergenceIntegrator() :
Q(NULL), trial_maps(NULL), test_maps(NULL), geom(NULL)
{ }
VectorDivergenceIntegrator(Coefficient *_q) :
Q(_q), trial_maps(NULL), test_maps(NULL), geom(NULL)
{ }
VectorDivergenceIntegrator(Coefficient &q) :
Q(&q), trial_maps(NULL), test_maps(NULL), geom(NULL)
{ }
VectorDivergenceIntegrator() { Q = NULL; }
VectorDivergenceIntegrator(Coefficient *_q) { Q = _q; }
VectorDivergenceIntegrator(Coefficient &q) { Q = &q; }
virtual void AssembleElementMatrix2(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans,
DenseMatrix &elmat);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
virtual void AddMultTransposePA(const Vector &x, Vector &y) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans);
};
/// (Q div u, div v) for RT elements
@@ -2319,12 +2132,6 @@ class VectorDiffusionIntegrator : public BilinearFormIntegrator
protected:
Coefficient *Q;
// PA extension
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, dofs1D, quad1D;
Vector pa_data;
private:
DenseMatrix Jinv;
DenseMatrix dshape;
@@ -2341,9 +2148,6 @@ public:
virtual void AssembleElementVector(const FiniteElement &el,
ElementTransformation &Tr,
const Vector &elfun, Vector &elvect);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
};
/** Integrator for the linear elasticity form:
-500
View File
@@ -1,500 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
using namespace std;
namespace mfem
{
// PA Convection Integrator
// PA Convection Assemble kernel
void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assuming the same element type
Mesh *mesh = fes.GetMesh();
if (mesh->GetNE() == 0) { return; }
const FiniteElement &el = *fes.GetFE(0);
ElementTransformation *T = mesh->GetElementTransformation(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T);
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::COORDINATES |
GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(dim*ne*nq, Device::GetMemoryType());
const int NE = ne;
const int NQ = nq;
Vector e_coeff(dim);
coeff.SetSize(dim*ne*nq);
auto h_C = Reshape(coeff.HostWrite(),dim,nq, ne);
if ( Q == nullptr)
{
for (int e=0; e<NE; ++e)
{
for (int q=0; q<nq; ++q)
{
for (int idim=0; idim < dim; ++idim)
{
h_C(idim,q,e) = alpha;
}
}
}
}
else
{
for (int e=0; e<NE; ++e)
{
ElementTransformation& Te = *fes.GetElementTransformation(e);
for (int q=0; q<nq; ++q)
{
for (int idim=0; idim < dim; ++idim)
{
Q->Eval(e_coeff, Te, ir->IntPoint(q));
h_C(idim,q,e) = alpha*e_coeff(idim);
}
}
}
}
auto C = Reshape(coeff.Read(),dim,nq, ne);
if (dim==2)
{
auto w = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
auto v = Reshape(pa_data.Write(), 2, NQ, NE);
MFEM_FORALL(e, NE,
{
for (int q=0; q<NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double cx = C(0,q,e);
const double cy = C(1,q,e);
const double w_coeff = w[q];
v(0,q,e) = w_coeff*(cx * J22 - cy * J12);
v(1,q,e) = - w_coeff*(cx * J21 - cy * J11);
}
});
}//dim = 2
if (dim==3)
{
auto w = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
auto v = Reshape(pa_data.Write(), 3, NQ, NE);
MFEM_FORALL(e, NE,
{
for (int q=0; q<NQ; ++q)
{
const double J00 = J(q,0,0,e);
const double J01 = J(q,0,1,e);
const double J02 = J(q,0,2,e);
const double J10 = J(q,1,0,e);
const double J11 = J(q,1,1,e);
const double J12 = J(q,1,2,e);
const double J20 = J(q,2,0,e);
const double J21 = J(q,2,1,e);
const double J22 = J(q,2,2,e);
const double A00 = (J11 * J22) - (J12 * J21);
const double A01 = (J02 * J21) - (J01 * J22);
const double A02 = (J01 * J12) - (J02 * J11);
const double A10 = (J12 * J20) - (J10 * J22);
const double A11 = (J00 * J22) - (J02 * J20);
const double A12 = (J02 * J10) - (J00 * J12);
const double A20 = (J10 * J21) - (J11 * J20);
const double A21 = (J01 * J20) - (J00 * J21);
const double A22 = (J00 * J11) - (J01 * J10);
const double w_coeff = w[q];
double cx = C(0,q,e);
double cy = C(1,q,e);
double cz = C(2,q,e);
v(0,q,e) = w_coeff*(cx*A00 + cy*A01 + cz*A02);
v(1,q,e) = w_coeff*(cx*A10 + cy*A11 + cz*A12);
v(2,q,e) = w_coeff*(cx*A20 + cy*A21 + cz*A22);
}
});
}//dim = 3
}
// PA Convection Apply 2D kernel
template<int T_D1D = 0, int T_Q1D = 0> static
void PAConvectionApply2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
{
const int DIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto D = Reshape(_op.Read(), DIM, Q1D, Q1D, NE);
auto xloc = Reshape(_x.Read(), D1D, D1D, NE);
auto yloc = Reshape(_y.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int iDIM = 2;
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double U[iDIM][max_D1D][max_Q1D];
for (int j1=0; j1<Q1D; ++j1)
{
for (int i2=0; i2<D1D; ++i2)
{
double dot0=0.0; double dot1=0.0;
for (int i1=0; i1<D1D; ++i1)
{
dot0 += G(j1,i1)*xloc(i1, i2, e);
dot1 += B(j1,i1)*xloc(i1, i2, e);
}
U[0][i2][j1] = dot0;
U[1][i2][j1] = dot1;
}
}
double W[iDIM][max_Q1D][max_Q1D];
for (int j1=0; j1<Q1D; ++j1)
{
for (int i2=0; i2<Q1D; ++i2)
{
double dot0=0.0; double dot1=0.0;
for (int i1=0; i1<D1D; ++i1)
{
dot0 += B(j1,i1)*U[0][i1][i2];
dot1 += G(j1,i1)*U[1][i1][i2];
}
W[0][i2][j1] = dot0;
W[1][i2][j1] = dot1;
}
}
double Z[max_Q1D][max_Q1D];
for (int k2=0; k2<Q1D; ++k2)
{
for (int k1=0; k1<Q1D; ++k1)
{
double dot(0.0);
for (int c=0; c<2; ++c)
{
dot += D(c, k1, k2, e) * W[c][k1][k2];
}
Z[k1][k2] = dot;
}
}
double Q[max_Q1D][max_D1D];
for (int j1=0; j1<D1D; ++j1)
{
for (int i2=0; i2<Q1D; ++i2)
{
double dot(0.0);
for (int i1=0; i1<Q1D; ++i1)
{
dot += Bt(j1, i1)*Z[i1][i2];
}
Q[i2][j1] = dot;
}
}
for (int j1=0; j1<D1D; ++j1)
{
for (int i2=0; i2<D1D; ++i2)
{
double dot(0.0);
for (int i1=0; i1<Q1D; ++i1)
{
dot += Bt(j1, i1)*Q[i1][i2];
}
yloc(i2,j1,e) += dot;
}
}
});
}
// PA Convection Apply 3D kernel
template<const int T_D1D = 0,
const int T_Q1D = 0> static
void PAConvectionApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
int d1d = 0, int q1d = 0)
{
const int DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto D = Reshape(_op.Read(), DIM, Q1D,Q1D, Q1D, NE);
auto xloc = Reshape(_x.Read(), D1D, D1D, D1D, NE);
auto yloc = Reshape(_y.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
//qpt x dof x dof
double BX[max_Q1D][max_Q1D][max_Q1D];
double GX[max_Q1D][max_Q1D][max_Q1D];
for (int j1=0; j1<Q1D; ++j1)
{
for (int i3=0; i3<D1D; ++i3)
{
for (int i2=0; i2<D1D; ++i2)
{
BX[i2][i3][j1] = 0.0;
GX[i2][i3][j1] = 0.0;
for (int i1=0; i1<D1D; ++i1)
{
BX[i2][i3][j1] += B(j1, i1) * xloc(i1,i2,i3,e);
GX[i2][i3][j1] += G(j1, i1) * xloc(i1,i2,i3,e);
}
}
}
}
double BBX[max_D1D][max_Q1D][max_Q1D];
double GBX[max_D1D][max_Q1D][max_Q1D];
double BGX[max_D1D][max_Q1D][max_Q1D];
for (int j1=0; j1<Q1D; ++j1)
{
for (int i3=0; i3<Q1D; ++i3)
{
for (int i2=0; i2<D1D; ++i2)
{
BBX[i2][i3][j1] = 0.0;
GBX[i2][i3][j1] = 0.0;
BGX[i2][i3][j1] = 0.0;
for (int i1=0; i1<D1D; ++i1)
{
BBX[i2][i3][j1] += B(j1, i1) * BX[i1][i2][i3];
GBX[i2][i3][j1] += G(j1, i1) * BX[i1][i2][i3];
BGX[i2][i3][j1] += B(j1, i1) * GX[i1][i2][i3];
}
}
}
}
double GBBX[max_Q1D][max_Q1D][max_Q1D];
double BGBX[max_Q1D][max_Q1D][max_Q1D];
double BBGX[max_Q1D][max_Q1D][max_Q1D];
for (int j1=0; j1<Q1D; ++j1)
{
for (int i3=0; i3<Q1D; ++i3)
{
for (int i2=0; i2<Q1D; ++i2)
{
GBBX[i2][i3][j1] = 0.0;
BGBX[i2][i3][j1] = 0.0;
BBGX[i2][i3][j1] = 0.0;
for (int i1=0; i1<D1D; ++i1)
{
GBBX[i2][i3][j1] += G(j1, i1) * BBX[i1][i2][i3];
BGBX[i2][i3][j1] += B(j1, i1) * GBX[i1][i2][i3];
BBGX[i2][i3][j1] += B(j1, i1) * BGX[i1][i2][i3];
}
}
}
}
double Z[max_Q1D][max_Q1D][max_Q1D];
for (int k3=0; k3<Q1D; ++k3)
{
for (int k2=0; k2<Q1D; ++k2)
{
for (int k1=0; k1<Q1D; ++k1)
{
double dot(0.0);
{
dot += D(0, k1, k2, k3, e) * BBGX[k1][k2][k3];
dot += D(1, k1, k2, k3, e) * BGBX[k1][k2][k3];
dot += D(2, k1, k2, k3, e) * GBBX[k1][k2][k3];
}
Z[k1][k2][k3] = dot;
}
}
}
//Apply (B1d)^T 3 more times
double BZ[max_Q1D][max_Q1D][max_Q1D];
for (int j1=0; j1<D1D; ++j1)
{
for (int i3=0; i3<Q1D; ++i3)
{
for (int i2=0; i2<Q1D; ++i2)
{
BZ[i2][i3][j1]=0.0;
for (int i1=0; i1<Q1D; ++i1)
{
BZ[i2][i3][j1] += Bt(j1,i1)*Z[i1][i2][i3];
}
}
}
}
double BBZ[max_Q1D][max_Q1D][max_Q1D];
for (int j1=0; j1<D1D; ++j1)
{
for (int i3=0; i3<D1D; ++i3)
{
for (int i2=0; i2<Q1D; ++i2)
{
BBZ[i2][i3][j1]=0.0;
for (int i1=0; i1<Q1D; ++i1)
{
BBZ[i2][i3][j1] += Bt(j1,i1)*BZ[i1][i2][i3];
}
}
}
}
for (int j1=0; j1<D1D; ++j1)
{
for (int i3=0; i3<D1D; ++i3)
{
for (int i2=0; i2<D1D; ++i2)
{
double dot(0.0);
for (int i1=0; i1<Q1D; ++i1)
{
dot += Bt(j1,i1)*BBZ[i1][i2][i3];
}
yloc(i2,i3,j1,e) += dot;
}
}
}
});
}
static void PAConvectionApply(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<double> &B,
const Array<double> &G,
const Array<double> &Bt,
const Array<double> &Gt,
const Vector &op,
const Vector &x,
Vector &y)
{
if (dim==2)
{
switch ((D1D << 4 ) | Q1D)
{
//case 0x22: PAConvectionApply2D<2,2>(NE, B, G, Bt, Gt, op, x, y); break;
//case 0x33: PAConvectionApply2D<3,3>(NE, B, G, Bt, Gt, op, x, y); break;
//case 0x44: PAConvectionApply2D<4,4>(NE, B, G, Bt, Gt, op, x, y); break;
//case 0x55: PAConvectionApply2D<5,5>(NE, B, G, Bt, Gt, op, x, y); break;
default: PAConvectionApply2D(NE, B, G, Bt, Gt, op, x, y,D1D,Q1D); break;
}
return;
}
if (dim == 3)
{
switch ((D1D << 4 ) | Q1D)
{
//case 0x23: PAConvectionApply3D<2,3>(NE, B, G, Bt, G, op, x, y); break;
//case 0x34: PAConvectionApply3D<3,4>(NE, B, G, Bt, G, op, x, y); break;
//case 0x45: PAConvectionApply3D<4,5>(NE, B, G, Bt, G, op, x, y); break;
//case 0x56: PAConvectionApply3D<5,6>(NE, B, G, Bt, G, op, x, y); break;
default: PAConvectionApply3D(NE, B, G, Bt, G, op, x, y,D1D,Q1D); break;
}
return;
}
MFEM_ABORT("Unknown kernel.");
}
void ConvectionIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
PAConvectionApply(dim, dofs1D, quad1D, ne,maps->B, maps->G, maps->Bt, maps->Gt,
pa_data, x, y);
}
} // namespace mfem
+136 -604
View File
@@ -12,7 +12,6 @@
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
#include "libceed/diffusion.hpp"
using namespace std;
@@ -91,6 +90,7 @@ static void PADiffusionSetup2D(const int Q1D,
const int NQ = Q1D*Q1D;
const bool const_c = c.Size() == 1;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
auto C = const_c ? Reshape(c.Read(), 1, 1) : Reshape(c.Read(), NQ, NE);
auto D = Reshape(d.Write(), NQ, 3, NE);
@@ -199,25 +199,12 @@ static void PADiffusionSetup(const int dim,
}
}
void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
const bool force)
void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assuming the same element type
fespace = &fes;
// Assumes tensor-product elements
Mesh *mesh = fes.GetMesh();
if (mesh->GetNE() == 0) { return; }
const FiniteElement &el = *fes.GetFE(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed() && !force)
{
if (ceedDataPtr) { delete ceedDataPtr; }
CeedData* ptr = new CeedData();
ceedDataPtr = ptr;
InitCeedCoeff(Q, ptr);
return CeedPADiffusionAssemble(fes, *ir, *ptr);
}
#endif
const int dims = el.GetDim();
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
const int nq = ir->GetNPoints();
@@ -242,7 +229,7 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
else
{
coeff.SetSize(nq * ne);
auto C = Reshape(coeff.HostWrite(), nq, ne);
auto C = Reshape(coeff.Write(), nq, ne);
for (int e = 0; e < ne; ++e)
{
ElementTransformation& T = *fes.GetElementTransformation(e);
@@ -256,435 +243,6 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
pa_data);
}
void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
SetupPA(fes);
}
template<int T_D1D = 0, int T_Q1D = 0>
static void PADiffusionDiagonal2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Vector &d,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
// note the different shape for D, this is a (symmetric) matrix so we only
// store necessary entries
auto D = Reshape(d.Read(), Q1D*Q1D, 3, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
// gradphi \cdot Q \gradphi has four terms
double QD0[MQ1][MD1];
double QD1[MQ1][MD1];
double QD2[MQ1][MD1];
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dy = 0; dy < D1D; ++dy)
{
QD0[qx][dy] = 0.0;
QD1[qx][dy] = 0.0;
QD2[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const int q = qx + qy * Q1D;
const double D0 = D(q,0,e);
const double D1 = D(q,1,e);
const double D2 = D(q,2,e);
QD0[qx][dy] += B(qy, dy) * B(qy, dy) * D0;
QD1[qx][dy] += B(qy, dy) * G(qy, dy) * D1;
QD2[qx][dy] += G(qy, dy) * G(qy, dy) * D2;
}
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
for (int qx = 0; qx < Q1D; ++qx)
{
Y(dx,dy,e) += G(qx, dx) * G(qx, dx) * QD0[qx][dy];
Y(dx,dy,e) += G(qx, dx) * B(qx, dx) * QD1[qx][dy];
Y(dx,dy,e) += B(qx, dx) * G(qx, dx) * QD1[qx][dy];
Y(dx,dy,e) += B(qx, dx) * B(qx, dx) * QD2[qx][dy];
}
}
}
});
}
// Shared memory PA Diffusion Diagonal 2D kernel
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
static void SmemPADiffusionDiagonal2D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Vector &d_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D*Q1D, 3, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_SHARED double BG[2][MQ1*MD1];
double (*B)[MD1] = (double (*)[MD1]) (BG+0);
double (*G)[MD1] = (double (*)[MD1]) (BG+1);
MFEM_SHARED double QD[4][NBZ][MD1][MQ1];
double (*QD0)[MD1] = (double (*)[MD1])(QD[0] + tidz);
double (*QD1)[MD1] = (double (*)[MD1])(QD[1] + tidz);
double (*QD2)[MD1] = (double (*)[MD1])(QD[3] + tidz);
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
G[q][d] = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QD0[qx][dy] = 0.0;
QD1[qx][dy] = 0.0;
QD2[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const int q = qx + qy * Q1D;
const double D0 = D(q,0,e);
const double D1 = D(q,1,e);
const double D2 = D(q,2,e);
const double By = B[qy][dy];
const double Gy = G[qy][dy];
const double BB = By * By;
const double BG = By * Gy;
const double GG = Gy * Gy;
QD0[qx][dy] += BB * D0;
QD1[qx][dy] += BG * D1;
QD2[qx][dy] += GG * D2;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B[qx][dx];
const double Gx = G[qx][dx];
const double BB = Bx * Bx;
const double BG = Bx * Gx;
const double GG = Gx * Gx;
Y(dx,dy,e) += GG * QD0[qx][dy];
Y(dx,dy,e) += BG * QD1[qx][dy];
Y(dx,dy,e) += BG * QD1[qx][dy];
Y(dx,dy,e) += BB * QD2[qx][dy];
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
static void PADiffusionDiagonal3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Vector &d,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
constexpr int DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, 6, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
double QQD[MQ1][MQ1][MD1];
double QDD[MQ1][MD1][MD1];
for (int i = 0; i < DIM; ++i)
{
for (int j = 0; j < DIM; ++j)
{
// first tensor contraction, along z direction
for (int qx = 0; qx < Q1D; ++qx)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int dz = 0; dz < D1D; ++dz)
{
QQD[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const int k = j >= i ?
3 - (3-i)*(2-i)/2 + j:
3 - (3-j)*(2-j)/2 + i;
const double O = Q(q,k,e);
const double Bz = B(qz,dz);
const double Gz = G(qz,dz);
const double L = i==2 ? Gz : Bz;
const double R = j==2 ? Gz : Bz;
QQD[qx][qy][dz] += L * O * R;
}
}
}
}
// second tensor contraction, along y direction
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
QDD[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const double By = B(qy,dy);
const double Gy = G(qy,dy);
const double L = i==1 ? Gy : By;
const double R = j==1 ? Gy : By;
QDD[qx][dy][dz] += L * QQD[qx][qy][dz] * R;
}
}
}
}
// third tensor contraction, along x direction
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B(qx,dx);
const double Gx = G(qx,dx);
const double L = i==0 ? Gx : Bx;
const double R = j==0 ? Gx : Bx;
Y(dx, dy, dz, e) += L * QDD[qx][dy][dz] * R;
}
}
}
}
}
}
});
}
// Shared memory PA Diffusion Diagonal 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void SmemPADiffusionDiagonal3D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Vector &d_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
constexpr int DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D*Q1D*Q1D, 6, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_SHARED double BG[2][MQ1*MD1];
double (*B)[MD1] = (double (*)[MD1]) (BG+0);
double (*G)[MD1] = (double (*)[MD1]) (BG+1);
MFEM_SHARED double QQD[MQ1][MQ1][MD1];
MFEM_SHARED double QDD[MQ1][MD1][MD1];
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
G[q][d] = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
for (int i = 0; i < DIM; ++i)
{
for (int j = 0; j < DIM; ++j)
{
// first tensor contraction, along z direction
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
QQD[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const int k = j >= i ?
3 - (3-i)*(2-i)/2 + j:
3 - (3-j)*(2-j)/2 + i;
const double O = D(q,k,e);
const double Bz = B[qz][dz];
const double Gz = G[qz][dz];
const double L = i==2 ? Gz : Bz;
const double R = j==2 ? Gz : Bz;
QQD[qx][qy][dz] += L * O * R;
}
}
}
}
MFEM_SYNC_THREAD;
// second tensor contraction, along y direction
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QDD[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const double By = B[qy][dy];
const double Gy = G[qy][dy];
const double L = i==1 ? Gy : By;
const double R = j==1 ? Gy : By;
QDD[qx][dy][dz] += L * QQD[qx][qy][dz] * R;
}
}
}
}
MFEM_SYNC_THREAD;
// third tensor contraction, along x direction
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B[qx][dx];
const double Gx = G[qx][dx];
const double L = i==0 ? Gx : Bx;
const double R = j==0 ? Gx : Bx;
Y(dx, dy, dz, e) += L * QDD[qx][dy][dz] * R;
}
}
}
}
}
}
});
}
static void PADiffusionAssembleDiagonal(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<double> &B,
const Array<double> &G,
const Vector &D,
Vector &Y)
{
if (dim == 2)
{
switch ((D1D << 4 ) | Q1D)
{
case 0x22: return SmemPADiffusionDiagonal2D<2,2,8>(NE,B,G,D,Y);
case 0x33: return SmemPADiffusionDiagonal2D<3,3,8>(NE,B,G,D,Y);
case 0x44: return SmemPADiffusionDiagonal2D<4,4,4>(NE,B,G,D,Y);
case 0x55: return SmemPADiffusionDiagonal2D<5,5,4>(NE,B,G,D,Y);
case 0x66: return SmemPADiffusionDiagonal2D<6,6,2>(NE,B,G,D,Y);
case 0x77: return SmemPADiffusionDiagonal2D<7,7,2>(NE,B,G,D,Y);
case 0x88: return SmemPADiffusionDiagonal2D<8,8,1>(NE,B,G,D,Y);
case 0x99: return SmemPADiffusionDiagonal2D<9,9,1>(NE,B,G,D,Y);
default: return PADiffusionDiagonal2D(NE,B,G,D,Y,D1D,Q1D);
}
}
else if (dim == 3)
{
switch ((D1D << 4 ) | Q1D)
{
case 0x23: return SmemPADiffusionDiagonal3D<2,3>(NE,B,G,D,Y);
case 0x34: return SmemPADiffusionDiagonal3D<3,4>(NE,B,G,D,Y);
case 0x45: return SmemPADiffusionDiagonal3D<4,5>(NE,B,G,D,Y);
case 0x56: return SmemPADiffusionDiagonal3D<5,6>(NE,B,G,D,Y);
case 0x67: return SmemPADiffusionDiagonal3D<6,7>(NE,B,G,D,Y);
case 0x78: return SmemPADiffusionDiagonal3D<7,8>(NE,B,G,D,Y);
case 0x89: return SmemPADiffusionDiagonal3D<8,9>(NE,B,G,D,Y);
case 0x9A: return SmemPADiffusionDiagonal3D<9,10>(NE,B,G,D,Y);
default: return PADiffusionDiagonal3D(NE,B,G,D,Y,D1D,Q1D);
}
}
MFEM_ABORT("Unknown kernel.");
}
void DiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
{
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
maps->B, maps->G, pa_data, diag);
}
#ifdef MFEM_USE_OCCA
// OCCA PA Diffusion Apply 2D kernel
static void OccaPADiffusionApply2D(const int D1D,
@@ -694,9 +252,9 @@ static void OccaPADiffusionApply2D(const int D1D,
const Array<double> &G,
const Array<double> &Bt,
const Array<double> &Gt,
const Vector &D,
const Vector &X,
Vector &Y)
const Vector &op,
const Vector &x,
Vector &y)
{
occa::properties props;
props["defines/D1D"] = D1D;
@@ -705,9 +263,9 @@ static void OccaPADiffusionApply2D(const int D1D,
const occa::memory o_G = OccaMemoryRead(G.GetMemory(), G.Size());
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
const occa::memory o_Gt = OccaMemoryRead(Gt.GetMemory(), Gt.Size());
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
const occa::memory o_op = OccaMemoryRead(op.GetMemory(), op.Size());
const occa::memory o_x = OccaMemoryRead(x.GetMemory(), x.Size());
occa::memory o_y = OccaMemoryReadWrite(y.GetMemory(), y.Size());
const occa_id_t id = std::make_pair(D1D,Q1D);
if (!Device::Allows(Backend::OCCA_CUDA))
{
@@ -719,7 +277,7 @@ static void OccaPADiffusionApply2D(const int D1D,
"DiffusionApply2D_CPU", props);
OccaDiffApply2D_cpu.emplace(id, DiffusionApply2D_CPU);
}
OccaDiffApply2D_cpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
OccaDiffApply2D_cpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_op, o_x, o_y);
}
else
{
@@ -731,7 +289,7 @@ static void OccaPADiffusionApply2D(const int D1D,
"DiffusionApply2D_GPU", props);
OccaDiffApply2D_gpu.emplace(id, DiffusionApply2D_GPU);
}
OccaDiffApply2D_gpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
OccaDiffApply2D_gpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_op, o_x, o_y);
}
}
@@ -743,9 +301,9 @@ static void OccaPADiffusionApply3D(const int D1D,
const Array<double> &G,
const Array<double> &Bt,
const Array<double> &Gt,
const Vector &D,
const Vector &X,
Vector &Y)
const Vector &op,
const Vector &x,
Vector &y)
{
occa::properties props;
props["defines/D1D"] = D1D;
@@ -754,9 +312,9 @@ static void OccaPADiffusionApply3D(const int D1D,
const occa::memory o_G = OccaMemoryRead(G.GetMemory(), G.Size());
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
const occa::memory o_Gt = OccaMemoryRead(Gt.GetMemory(), Gt.Size());
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
const occa::memory o_op = OccaMemoryRead(op.GetMemory(), op.Size());
const occa::memory o_x = OccaMemoryRead(x.GetMemory(), x.Size());
occa::memory o_y = OccaMemoryReadWrite(y.GetMemory(), y.Size());
const occa_id_t id = std::make_pair(D1D,Q1D);
if (!Device::Allows(Backend::OCCA_CUDA))
{
@@ -768,7 +326,7 @@ static void OccaPADiffusionApply3D(const int D1D,
"DiffusionApply3D_CPU", props);
OccaDiffApply3D_cpu.emplace(id, DiffusionApply3D_CPU);
}
OccaDiffApply3D_cpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
OccaDiffApply3D_cpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_op, o_x, o_y);
}
else
{
@@ -780,35 +338,35 @@ static void OccaPADiffusionApply3D(const int D1D,
"DiffusionApply3D_GPU", props);
OccaDiffApply3D_gpu.emplace(id, DiffusionApply3D_GPU);
}
OccaDiffApply3D_gpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
OccaDiffApply3D_gpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_op, o_x, o_y);
}
}
#endif // MFEM_USE_OCCA
// PA Diffusion Apply 2D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void PADiffusionApply2D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Array<double> &bt_,
const Array<double> &gt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
template<int T_D1D = 0, int T_Q1D = 0> static
void PADiffusionApply2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b_.Read(), Q1D, D1D);
auto G = Reshape(g_.Read(), Q1D, D1D);
auto Bt = Reshape(bt_.Read(), D1D, Q1D);
auto Gt = Reshape(gt_.Read(), D1D, Q1D);
auto D = Reshape(d_.Read(), Q1D*Q1D, 3, NE);
auto X = Reshape(x_.Read(), D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Gt = Reshape(gt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D, 3, NE);
auto x = Reshape(_x.Read(), D1D, D1D, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -836,7 +394,7 @@ static void PADiffusionApply2D(const int NE,
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = X(dx,dy,e);
const double s = x(dx,dy,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * B(qx,dx);
@@ -861,9 +419,9 @@ static void PADiffusionApply2D(const int NE,
{
const int q = qx + qy * Q1D;
const double O11 = D(q,0,e);
const double O12 = D(q,1,e);
const double O22 = D(q,2,e);
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O22 = op(q,2,e);
const double gradX = grad[qy][qx][0];
const double gradY = grad[qy][qx][1];
@@ -898,7 +456,7 @@ static void PADiffusionApply2D(const int NE,
const double wDy = Gt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
Y(dx,dy,e) += ((gradX[dx][0] * wy) + (gradX[dx][1] * wDy));
y(dx,dy,e) += ((gradX[dx][0] * wy) + (gradX[dx][1] * wDy));
}
}
}
@@ -906,15 +464,17 @@ static void PADiffusionApply2D(const int NE,
}
// Shared memory PA Diffusion Apply 2D kernel
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
template<const int T_D1D = 0,
const int T_Q1D = 0,
const int T_NBZ = 0>
static void SmemPADiffusionApply2D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Array<double> &bt_,
const Array<double> &gt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const Array<double> &_b,
const Array<double> &_g,
const Array<double> &_bt,
const Array<double> &_gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
{
@@ -925,11 +485,11 @@ static void SmemPADiffusionApply2D(const int NE,
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D*Q1D, 3, NE);
auto x = Reshape(x_.Read(), D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
auto b = Reshape(_b.Read(), Q1D, D1D);
auto g = Reshape(_g.Read(), Q1D, D1D);
auto op = Reshape(_op.Read(), Q1D*Q1D, 3, NE);
auto x = Reshape(_x.Read(), D1D, D1D, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int tidz = MFEM_THREAD_ID(z);
@@ -960,12 +520,12 @@ static void SmemPADiffusionApply2D(const int NE,
}
if (tidz == 0)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][dy] = b(q,dy);
G[q][dy] = g(q,dy);
B[q][d] = b(q,d);
G[q][d] = g(q,d);
}
}
}
@@ -1008,9 +568,9 @@ static void SmemPADiffusionApply2D(const int NE,
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const int q = (qx + ((qy) * Q1D));
const double O11 = D(q,0,e);
const double O12 = D(q,1,e);
const double O22 = D(q,2,e);
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O22 = op(q,2,e);
const double gX = QQ0[qy][qx];
const double gY = QQ1[qy][qx];
QQ0[qy][qx] = (O11 * gX) + (O12 * gY);
@@ -1020,12 +580,12 @@ static void SmemPADiffusionApply2D(const int NE,
MFEM_SYNC_THREAD;
if (tidz == 0)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
Bt[dy][q] = b(q,dy);
Gt[dy][q] = g(q,dy);
Bt[d][q] = b(q,d);
Gt[d][q] = g(q,d);
}
}
}
@@ -1057,23 +617,25 @@ static void SmemPADiffusionApply2D(const int NE,
u += DQ0[qy][dx] * Bt[dy][qy];
v += DQ1[qy][dx] * Gt[dy][qy];
}
Y(dx,dy,e) += (u + v);
y(dx,dy,e) += (u + v);
}
}
});
}
// PA Diffusion Apply 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void PADiffusionApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &d_,
const Vector &x_,
Vector &y_,
int d1d = 0, int q1d = 0)
template<const int T_D1D = 0,
const int T_Q1D = 0> static
void PADiffusionApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
int d1d = 0, int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
@@ -1083,9 +645,9 @@ static void PADiffusionApply3D(const int NE,
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Gt = Reshape(gt.Read(), D1D, Q1D);
auto D = Reshape(d_.Read(), Q1D*Q1D*Q1D, 6, NE);
auto X = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
auto op = Reshape(_op.Read(), Q1D*Q1D*Q1D, 6, NE);
auto x = Reshape(_x.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -1127,7 +689,7 @@ static void PADiffusionApply3D(const int NE,
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = X(dx,dy,dz,e);
const double s = x(dx,dy,dz,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * B(qx,dx);
@@ -1171,12 +733,12 @@ static void PADiffusionApply3D(const int NE,
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const double O11 = D(q,0,e);
const double O12 = D(q,1,e);
const double O13 = D(q,2,e);
const double O22 = D(q,3,e);
const double O23 = D(q,4,e);
const double O33 = D(q,5,e);
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O13 = op(q,2,e);
const double O22 = op(q,3,e);
const double O23 = op(q,4,e);
const double O33 = op(q,5,e);
const double gradX = grad[qz][qy][qx][0];
const double gradY = grad[qz][qy][qx][1];
const double gradZ = grad[qz][qy][qx][2];
@@ -1241,7 +803,7 @@ static void PADiffusionApply3D(const int NE,
{
for (int dx = 0; dx < D1D; ++dx)
{
Y(dx,dy,dz,e) +=
y(dx,dy,dz,e) +=
((gradXY[dy][dx][0] * wz) +
(gradXY[dy][dx][1] * wz) +
(gradXY[dy][dx][2] * wDz));
@@ -1253,15 +815,16 @@ static void PADiffusionApply3D(const int NE,
}
// Shared memory PA Diffusion Apply 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
template<const int T_D1D = 0,
const int T_Q1D = 0>
static void SmemPADiffusionApply3D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Array<double> &bt_,
const Array<double> &gt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const Array<double> &_b,
const Array<double> &_g,
const Array<double> &_bt,
const Array<double> &_gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
{
@@ -1271,11 +834,11 @@ static void SmemPADiffusionApply3D(const int NE,
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto d = Reshape(d_.Read(), Q1D*Q1D*Q1D, 6, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
auto b = Reshape(_b.Read(), Q1D, D1D);
auto g = Reshape(_g.Read(), Q1D, D1D);
auto op = Reshape(_op.Read(), Q1D*Q1D*Q1D, 6, NE);
auto x = Reshape(_x.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
@@ -1399,12 +962,12 @@ static void SmemPADiffusionApply3D(const int NE,
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const int q = qx + ((qy*Q1D) + (qz*Q1D*Q1D));
const double O11 = d(q,0,e);
const double O12 = d(q,1,e);
const double O13 = d(q,2,e);
const double O22 = d(q,3,e);
const double O23 = d(q,4,e);
const double O33 = d(q,5,e);
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O13 = op(q,2,e);
const double O22 = op(q,3,e);
const double O23 = op(q,4,e);
const double O33 = op(q,5,e);
const double gX = QQQ0[qz][qy][qx];
const double gY = QQQ1[qz][qy][qx];
const double gZ = QQQ2[qz][qy][qx];
@@ -1501,21 +1064,21 @@ static void PADiffusionApply(const int dim,
const Array<double> &G,
const Array<double> &Bt,
const Array<double> &Gt,
const Vector &D,
const Vector &X,
Vector &Y)
const Vector &op,
const Vector &x,
Vector &y)
{
#ifdef MFEM_USE_OCCA
if (DeviceCanUseOcca())
{
if (dim == 2)
{
OccaPADiffusionApply2D(D1D,Q1D,NE,B,G,Bt,Gt,D,X,Y);
OccaPADiffusionApply2D(D1D, Q1D, NE, B, G, Bt, Gt, op, x, y);
return;
}
if (dim == 3)
{
OccaPADiffusionApply3D(D1D,Q1D,NE,B,G,Bt,Gt,D,X,Y);
OccaPADiffusionApply3D(D1D, Q1D, NE, B, G, Bt, Gt, op, x, y);
return;
}
MFEM_ABORT("OCCA PADiffusionApply unknown kernel!");
@@ -1525,29 +1088,29 @@ static void PADiffusionApply(const int dim,
{
switch ((D1D << 4 ) | Q1D)
{
case 0x22: return SmemPADiffusionApply2D<2,2,16>(NE,B,G,Bt,Gt,D,X,Y);
case 0x33: return SmemPADiffusionApply2D<3,3,16>(NE,B,G,Bt,Gt,D,X,Y);
case 0x44: return SmemPADiffusionApply2D<4,4,8>(NE,B,G,Bt,Gt,D,X,Y);
case 0x55: return SmemPADiffusionApply2D<5,5,8>(NE,B,G,Bt,Gt,D,X,Y);
case 0x66: return SmemPADiffusionApply2D<6,6,4>(NE,B,G,Bt,Gt,D,X,Y);
case 0x77: return SmemPADiffusionApply2D<7,7,4>(NE,B,G,Bt,Gt,D,X,Y);
case 0x88: return SmemPADiffusionApply2D<8,8,2>(NE,B,G,Bt,Gt,D,X,Y);
case 0x99: return SmemPADiffusionApply2D<9,9,2>(NE,B,G,Bt,Gt,D,X,Y);
default: return PADiffusionApply2D(NE,B,G,Bt,Gt,D,X,Y,D1D,Q1D);
case 0x22: return SmemPADiffusionApply2D<2,2,16>(NE,B,G,Bt,Gt,op,x,y);
case 0x33: return SmemPADiffusionApply2D<3,3,16>(NE,B,G,Bt,Gt,op,x,y);
case 0x44: return SmemPADiffusionApply2D<4,4,8>(NE,B,G,Bt,Gt,op,x,y);
case 0x55: return SmemPADiffusionApply2D<5,5,8>(NE,B,G,Bt,Gt,op,x,y);
case 0x66: return SmemPADiffusionApply2D<6,6,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x77: return SmemPADiffusionApply2D<7,7,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x88: return SmemPADiffusionApply2D<8,8,2>(NE,B,G,Bt,Gt,op,x,y);
case 0x99: return SmemPADiffusionApply2D<9,9,2>(NE,B,G,Bt,Gt,op,x,y);
default: return PADiffusionApply2D(NE,B,G,Bt,Gt,op,x,y,D1D,Q1D);
}
}
else if (dim == 3)
{
switch ((D1D << 4 ) | Q1D)
{
case 0x23: return SmemPADiffusionApply3D<2,3>(NE,B,G,Bt,Gt,D,X,Y);
case 0x34: return SmemPADiffusionApply3D<3,4>(NE,B,G,Bt,Gt,D,X,Y);
case 0x45: return SmemPADiffusionApply3D<4,5>(NE,B,G,Bt,Gt,D,X,Y);
case 0x56: return SmemPADiffusionApply3D<5,6>(NE,B,G,Bt,Gt,D,X,Y);
case 0x67: return SmemPADiffusionApply3D<6,7>(NE,B,G,Bt,Gt,D,X,Y);
case 0x78: return SmemPADiffusionApply3D<7,8>(NE,B,G,Bt,Gt,D,X,Y);
case 0x89: return SmemPADiffusionApply3D<8,9>(NE,B,G,Bt,Gt,D,X,Y);
default: return PADiffusionApply3D(NE,B,G,Bt,Gt,D,X,Y,D1D,Q1D);
case 0x23: return SmemPADiffusionApply3D<2,3>(NE,B,G,Bt,Gt,op,x,y);
case 0x34: return SmemPADiffusionApply3D<3,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x45: return SmemPADiffusionApply3D<4,5>(NE,B,G,Bt,Gt,op,x,y);
case 0x56: return SmemPADiffusionApply3D<5,6>(NE,B,G,Bt,Gt,op,x,y);
case 0x67: return SmemPADiffusionApply3D<6,7>(NE,B,G,Bt,Gt,op,x,y);
case 0x78: return SmemPADiffusionApply3D<7,8>(NE,B,G,Bt,Gt,op,x,y);
case 0x89: return SmemPADiffusionApply3D<8,9>(NE,B,G,Bt,Gt,op,x,y);
default: return PADiffusionApply3D(NE,B,G,Bt,Gt,op,x,y,D1D,Q1D);
}
}
MFEM_ABORT("Unknown kernel.");
@@ -1556,40 +1119,9 @@ static void PADiffusionApply(const int dim,
// PA Diffusion Apply kernel
void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
const CeedScalar *x_ptr;
CeedScalar *y_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
x_ptr = x.Read();
y_ptr = y.ReadWrite();
}
else
{
x_ptr = x.HostRead();
y_ptr = y.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
const_cast<CeedScalar*>(x_ptr));
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorSyncArray(ceedDataPtr->v, mem);
}
else
#endif
{
PADiffusionApply(dim, dofs1D, quad1D, ne,
maps->B, maps->G, maps->Bt, maps->Gt,
pa_data, x, y);
}
PADiffusionApply(dim, dofs1D, quad1D, ne,
maps->B, maps->G, maps->Bt, maps->Gt,
pa_data, x, y);
}
} // namespace mfem
File diff suppressed because it is too large Load Diff
-823
View File
@@ -1,823 +0,0 @@
// Copyright (c) 2019, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
using namespace std;
namespace mfem
{
// PA Gradient Integrator
/* Description of the *SetupND functions
Inputs are as follows
\b Q1D number of quadrature points in one dimension.
\b w quadrature weights.
\b j element Jacobians.
\b COEFF coefficient at quadrature points.
The function is used precompute data needed at quadrature points during
the action. */
/* Description of the *ApplyND functions
The template parameters are
\b T_D1D number of degrees of freedom in one dimension,
\b T_Q1D number of quadrature points in one dimension,
and are necessary to allow for compiler optimizations inside the kernel.
Inputs are as follows
\b NE number of elements.
\b B matrix of basis functions.
\b G matrix of derivatives of the basis functions.
\b Bt transpose of matrix of basis functions.
\b Gt transpose matrix of derivatives of the basis functions.
\b op data used during action of the element matrix in the tensor
product application.
\b x input vector of degrees of freedom on the element.
\b y output vector of degrees of freedom on the element.
The function computes the kernel for one dimension that is suitable for
tensor product action to form ND operators.
Most of the ND inputs are reshaped as NQ*(ND*ND)*NE data structure, i.e
to allow indexing such as op(qpt,i,j,el).
The output data structure is dependent on the kernel and layout of the
dimension ND and element number, but in general resembles the action of the
element matrix in the tensor product application. */
/* Description of the Smem*ApplyND functions
The shared memory (Smem) versions of the kernels differ from the regular
versions in the following properties.
\b MFEM_FORALL is using only one level of parallelism.
\b MFEM_FORALL_ND uses an additional level of parallelism
\b MFEM_FOREACH_THREAD
These macros allow automatic mapping of manually defined blocks to
underlying hardware threads. These threads can share memory by using
the \b MFEM_SHARED keyword for local arrays. */
// PA Gradient Assemble 2D kernel
static void PAGradientSetup2D(const int Q1D,
const int NE,
const Array<double> &w,
const Vector &j,
const double COEFF,
Vector &op)
{
const int NQ = Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
auto y = Reshape(op.Write(), NQ, 2, 2, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J12 = J(q,0,1,e);
const double J21 = J(q,1,0,e);
const double J22 = J(q,1,1,e);
// Store wq * Q * adj(J)
y(q,0,0,e) = W[q] * COEFF * J22; // 1,1
y(q,0,1,e) = W[q] * COEFF * -J12; // 1,2
y(q,1,0,e) = W[q] * COEFF * -J21; // 2,1
y(q,1,1,e) = W[q] * COEFF * J11; // 2,2
}
});
}
// PA Gradient Assemble 3D kernel
static void PAGradientSetup3D(const int Q1D,
const int NE,
const Array<double> &w,
const Vector &j,
const double COEFF,
Vector &op)
{
const int NQ = Q1D*Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
auto y = Reshape(op.Write(), NQ, 3, 3, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J31 = J(q,2,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double J32 = J(q,2,1,e);
const double J13 = J(q,0,2,e);
const double J23 = J(q,1,2,e);
const double J33 = J(q,2,2,e);
const double cw = W[q] * COEFF;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// Store wq * Q * adj(J)
y(q,0,0,e) = cw * A11; // 1,1
y(q,0,1,e) = cw * A12; // 1,2
y(q,0,2,e) = cw * A13; // 1,3
y(q,1,0,e) = cw * A21; // 2,1
y(q,1,1,e) = cw * A22; // 2,2
y(q,1,2,e) = cw * A23; // 2,3
y(q,2,0,e) = cw * A31; // 3,1
y(q,2,1,e) = cw * A32; // 3,2
y(q,2,2,e) = cw * A33; // 3,3
}
});
}
static void PAGradientSetup(const int dim,
const int TR_D1D,
const int TE_D1D,
const int Q1D,
const int NE,
const Array<double> &W,
const Vector &J,
const double COEFF,
Vector &op)
{
if (dim == 1) { MFEM_ABORT("dim==1 not supported in PAGradientSetup"); }
if (dim == 2)
{
PAGradientSetup2D(Q1D, NE, W, J, COEFF, op);
}
if (dim == 3)
{
PAGradientSetup3D(Q1D, NE, W, J, COEFF, op);
}
}
void GradientIntegrator::AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes)
{
// Assumes tensor-product elements ordered by nodes
MFEM_ASSERT(trial_fes.GetOrdering() == Ordering::byNODES,
"PA Only supports Ordering::byNODES!");
Mesh *mesh = trial_fes.GetMesh();
const FiniteElement &trial_fe = *trial_fes.GetFE(0);
const FiniteElement &test_fe = *test_fes.GetFE(0);
ElementTransformation *trans = mesh->GetElementTransformation(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(trial_fe, test_fe,
*trans);
const int dims = trial_fe.GetDim();
const int dimsToStore = dims * dims;
const int nq = ir->GetNPoints();
dim = mesh->Dimension();
ne = trial_fes.GetNE();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
trial_maps = &trial_fe.GetDofToQuad(*ir, DofToQuad::TENSOR);
trial_dofs1D = trial_maps->ndof;
quad1D = trial_maps->nqpt;
test_maps = &test_fe.GetDofToQuad(*ir, DofToQuad::TENSOR);
test_dofs1D = test_maps->ndof;
MFEM_ASSERT(quad1D == test_maps->nqpt,
"PA requires test and trial space to have same number of quadrature points!");
pa_data.SetSize(nq * dimsToStore * ne, Device::GetMemoryType());
double coeff = 1.0;
if (Q)
{
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient*>(Q);
MFEM_VERIFY(cQ != NULL, "only ConstantCoefficient is supported!");
coeff = cQ->constant;
}
PAGradientSetup(dim, trial_dofs1D, test_dofs1D, quad1D,
ne, ir->GetWeights(), geom->J, coeff, pa_data);
}
// PA Gradient Apply 2D kernel
template<int T_TR_D1D = 0, int T_TE_D1D = 0, int T_Q1D = 0>
static void PAGradientApply2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int tr_d1d = 0,
const int te_d1d = 0,
const int q1d = 0)
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(TR_D1D <= MAX_D1D, "");
MFEM_VERIFY(TE_D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, TR_D1D);
auto G = Reshape(g.Read(), Q1D, TR_D1D);
auto Bt = Reshape(bt.Read(), TE_D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D, 2,2, NE);
auto x = Reshape(_x.Read(), TR_D1D, TR_D1D, NE);
auto y = Reshape(_y.ReadWrite(), TE_D1D, TE_D1D, 2, NE);
MFEM_FORALL(e, NE,
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = 2;
// the following variables are evaluated at compile time
constexpr int max_TE_D1D = T_TE_D1D ? T_TE_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double grad[max_Q1D][max_Q1D][VDIM];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qy][qx][0] = 0.0;
grad[qy][qx][1] = 0.0;
}
}
for (int dy = 0; dy < TR_D1D; ++dy)
{
double gradX[max_Q1D][VDIM];
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] = 0.0;
gradX[qx][1] = 0.0;
}
for (int dx = 0; dx < TR_D1D; ++dx)
{
const double s = x(dx,dy,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * G(qx,dx);
gradX[qx][1] += s * B(qx,dx);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = B(qy,dy);
const double wDy = G(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qy][qx][0] += gradX[qx][0] * wy;
grad[qy][qx][1] += gradX[qx][1] * wDy;
}
}
}
// We've now calculated grad(p) = [Dxy, xDy] in plane
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + qy * Q1D;
const double gradX = grad[qy][qx][0];
const double gradY = grad[qy][qx][1];
grad[qy][qx][0] = gradX*op(q,0,0,e) + gradY*op(q,1,0,e);
grad[qy][qx][1] = gradX*op(q,0,1,e) + gradY*op(q,1,1,e);
}
}
// We've now calculated grad = grad p * op
for (int qy = 0; qy < Q1D; ++qy)
{
double opX[max_TE_D1D][VDIM];
for (int dx = 0; dx < TE_D1D; ++dx)
{
opX[dx][0] = 0.0;
opX[dx][1] = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
opX[dx][0] += Bt(dx,qx)*grad[qy][qx][0];
opX[dx][1] += Bt(dx,qx)*grad[qy][qx][1];
}
}
for (int dy = 0; dy < TE_D1D; ++dy)
{
for (int dx = 0; dx < TE_D1D; ++dx)
{
y(dx,dy,0,e) += Bt(dy,qy)*opX[dx][0];
y(dx,dy,1,e) += Bt(dy,qy)*opX[dx][1];
}
}
}
// We've now calculated y = u * grad
});
}
// PA Gradient Apply 2D kernel transpose
template<int T_TR_D1D = 0, int T_TE_D1D = 0, int T_Q1D = 0>
static void PAGradientApplyTranspose2D(const int NE,
const Array<double> &bt,
const Array<double> &gt,
const Array<double> &b,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int tr_d1d = 0,
const int te_d1d = 0,
const int q1d = 0)
{
// TODO
MFEM_ASSERT(false, "GradientPAApplyTranspose 3D not implemented.");
}
// PA Gradient Apply 3D kernel
template<const int T_TR_D1D = 0, const int T_TE_D1D = 0, const int T_Q1D = 0>
static void PAGradientApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Vector &_op,
const Vector &_x,
Vector &_y,
int tr_d1d = 0,
int te_d1d = 0,
int q1d = 0)
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(TR_D1D <= MAX_D1D, "");
MFEM_VERIFY(TE_D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, TR_D1D);
auto G = Reshape(g.Read(), Q1D, TR_D1D);
auto Bt = Reshape(bt.Read(), TE_D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D*Q1D, 3,3, NE);
auto x = Reshape(_x.Read(), TR_D1D, TR_D1D, TR_D1D, NE);
auto y = Reshape(_y.ReadWrite(), TE_D1D, TE_D1D, TE_D1D, 3, NE);
MFEM_FORALL(e, NE,
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = 3;
// the following variables are evaluated at compile time
constexpr int max_TE_D1D = T_TE_D1D ? T_TE_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double grad[max_Q1D][max_Q1D][max_Q1D][VDIM];
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qz][qy][qx][0] = 0.0;
grad[qz][qy][qx][1] = 0.0;
grad[qz][qy][qx][2] = 0.0;
}
}
}
for (int dz = 0; dz < TR_D1D; ++dz)
{
double gradXY[max_Q1D][max_Q1D][3];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
gradXY[qy][qx][0] = 0.0;
gradXY[qy][qx][1] = 0.0;
gradXY[qy][qx][2] = 0.0;
}
}
for (int dy = 0; dy < TR_D1D; ++dy)
{
double gradX[max_Q1D][2];
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] = 0.0;
gradX[qx][1] = 0.0;
}
for (int dx = 0; dx < TR_D1D; ++dx)
{
const double s = x(dx,dy,dz,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * B(qx,dx);
gradX[qx][1] += s * G(qx,dx);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = B(qy,dy);
const double wDy = G(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
const double wx = gradX[qx][0];
const double wDx = gradX[qx][1];
gradXY[qy][qx][0] += wDx * wy;
gradXY[qy][qx][1] += wx * wDy;
gradXY[qy][qx][2] += wx * wy;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const double wz = B(qz,dz);
const double wDz = G(qz,dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qz][qy][qx][0] += gradXY[qy][qx][0] * wz;
grad[qz][qy][qx][1] += gradXY[qy][qx][1] * wz;
grad[qz][qy][qx][2] += gradXY[qy][qx][2] * wDz;
}
}
}
}
// We've now calculated grad(p) = [Dxyz, xDyz, xyDz] in plane
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const double gradX = grad[qz][qy][qx][0];
const double gradY = grad[qz][qy][qx][1];
const double gradZ = grad[qz][qy][qx][2];
grad[qz][qy][qx][0] = gradX*op(q,0,0,e) + gradY*op(q,1,0,e) + gradZ*op(q,2,0,e);
grad[qz][qy][qx][1] = gradX*op(q,0,1,e) + gradY*op(q,1,1,e) + gradZ*op(q,2,1,e);
grad[qz][qy][qx][2] = gradX*op(q,0,2,e) + gradY*op(q,1,2,e) + gradZ*op(q,2,2,e);
}
}
}
// We've now calculated grad = grad p * op
for (int qz = 0; qz < Q1D; ++qz)
{
double opXY[max_TE_D1D][max_TE_D1D][VDIM];
for (int dy = 0; dy < TE_D1D; ++dy)
{
for (int dx = 0; dx < TE_D1D; ++dx)
{
opXY[dy][dx][0] = 0.0;
opXY[dy][dx][1] = 0.0;
opXY[dy][dx][2] = 0.0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double opX[max_TE_D1D][VDIM];
for (int dx = 0; dx < TE_D1D; ++dx)
{
opX[dx][0] = 0.0;
opX[dx][1] = 0.0;
opX[dx][2] = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
opX[dx][0] += Bt(dx,qx)*grad[qz][qy][qx][0];
opX[dx][1] += Bt(dx,qx)*grad[qz][qy][qx][1];
opX[dx][2] += Bt(dx,qx)*grad[qz][qy][qx][2];
}
}
for (int dy = 0; dy < TE_D1D; ++dy)
{
for (int dx = 0; dx < TE_D1D; ++dx)
{
opXY[dy][dx][0] += Bt(dy,qy)*opX[dx][0];
opXY[dy][dx][1] += Bt(dy,qy)*opX[dx][1];
opXY[dy][dx][2] += Bt(dy,qy)*opX[dx][2];
}
}
}
for (int dz = 0; dz < TE_D1D; ++dz)
{
for (int dy = 0; dy < TE_D1D; ++dy)
{
for (int dx = 0; dx < TE_D1D; ++dx)
{
y(dx,dy,dz,0,e) += Bt(dz,qz)*opXY[dy][dx][0];
y(dx,dy,dz,1,e) += Bt(dz,qz)*opXY[dy][dx][1];
y(dx,dy,dz,2,e) += Bt(dz,qz)*opXY[dy][dx][2];
}
}
}
}
// We've now calculated y = u * grad
});
}
// PA Gradient Apply 3D kernel
template<const int T_TR_D1D = 0, const int T_TE_D1D = 0, const int T_Q1D = 0>
static void PAGradientApplyTranspose3D(const int NE,
const Array<double> &bt,
const Array<double> &gt,
const Array<double> &b,
const Vector &_op,
const Vector &_x,
Vector &_y,
int tr_d1d = 0,
int te_d1d = 0,
int q1d = 0)
{
MFEM_ASSERT(false, "Gradient PA Apply Transpose 3D not implemented.");
}
// Shared memory PA Gradient Apply 3D kernel
template<const int T_TR_D1D = 0, const int T_TE_D1D = 0, const int T_Q1D = 0>
static void SmemPAGradientApply3D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Array<double> &bt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int tr_d1d = 0,
const int te_d1d = 0,
const int q1d = 0)
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(TR_D1D <= MAX_D1D, "");
MFEM_VERIFY(TE_D1D <= MAX_D1D, "");
MFEM_VERIFY(TR_D1D <= Q1D, "");
MFEM_VERIFY(TE_D1D <= Q1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto b = Reshape(b_.Read(), Q1D, TR_D1D);
auto g = Reshape(g_.Read(), Q1D, TR_D1D);
auto bt = Reshape(bt_.Read(), TE_D1D, Q1D);
auto D = Reshape(d_.Read(), Q1D*Q1D*Q1D, 3, 3, NE);
auto x = Reshape(x_.Read(), TR_D1D, TR_D1D, TR_D1D, NE);
auto y = Reshape(y_.ReadWrite(), TE_D1D, TE_D1D, TE_D1D, 3, NE);
MFEM_FORALL_3D(e, NE, (Q1D>8)?8:Q1D, (Q1D>8)?8:Q1D, (Q1D>8)?8:Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1DR = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int D1DE = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1R = T_TR_D1D ? T_TR_D1D : MAX_D1D;
constexpr int MD1E = T_TE_D1D ? T_TE_D1D : MAX_D1D;
constexpr int MD1 = MD1E > MD1R ? MD1E : MD1R;
constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
MFEM_SHARED double sBG[2][MQ1*MD1];
double (*B)[MD1] = (double (*)[MD1]) (sBG+0);
double (*G)[MD1] = (double (*)[MD1]) (sBG+1);
double (*Bt)[MQ1] = (double (*)[MQ1]) (sBG+0);
MFEM_SHARED double sm0[3][MDQ*MDQ*MDQ];
MFEM_SHARED double sm1[3][MDQ*MDQ*MDQ];
double (*X)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
double (*DDQ0)[MD1][MQ1] = (double (*)[MD1][MQ1]) (sm0+0);
double (*DDQ1)[MD1][MQ1] = (double (*)[MD1][MQ1]) (sm0+1);
double (*DQQ0)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm1+0);
double (*DQQ1)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm1+1);
double (*DQQ2)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm1+2);
double (*QQQ0)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm0+0);
double (*QQQ1)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm0+1);
double (*QQQ2)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm0+2);
double (*QQD0)[MQ1][MD1] = (double (*)[MQ1][MD1]) (sm1+0);
double (*QQD1)[MQ1][MD1] = (double (*)[MQ1][MD1]) (sm1+1);
double (*QQD2)[MQ1][MD1] = (double (*)[MQ1][MD1]) (sm1+2);
double (*QDD0)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+0);
double (*QDD1)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+1);
double (*QDD2)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
MFEM_FOREACH_THREAD(dz,z,D1DR)
{
MFEM_FOREACH_THREAD(dy,y,D1DR)
{
MFEM_FOREACH_THREAD(dx,x,D1DR)
{
X[dz][dy][dx] = x(dx,dy,dz,e);
}
}
}
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1DR)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
G[q][d] = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1DR)
{
MFEM_FOREACH_THREAD(dy,y,D1DR)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
for (int dx = 0; dx < D1DR; ++dx)
{
const double coord = X[dz][dy][dx];
u += coord * B[qx][dx];
v += coord * G[qx][dx];
}
DDQ0[dz][dy][qx] = u;
DDQ1[dz][dy][qx] = v;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1DR)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dy = 0; dy < D1DR; ++dy)
{
u += DDQ1[dz][dy][qx] * B[qy][dy];
v += DDQ0[dz][dy][qx] * G[qy][dy];
w += DDQ0[dz][dy][qx] * B[qy][dy];
}
DQQ0[dz][qy][qx] = u;
DQQ1[dz][qy][qx] = v;
DQQ2[dz][qy][qx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dz = 0; dz < D1DR; ++dz)
{
u += DQQ0[dz][qy][qx] * B[qz][dz];
v += DQQ1[dz][qy][qx] * B[qz][dz];
w += DQQ2[dz][qy][qx] * G[qz][dz];
}
QQQ0[qz][qy][qx] = u;
QQQ1[qz][qy][qx] = v;
QQQ2[qz][qy][qx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const double gX = QQQ0[qz][qy][qx];
const double gY = QQQ1[qz][qy][qx];
const double gZ = QQQ2[qz][qy][qx];
QQQ0[qz][qy][qx] = (D(q,0,0,e)*gX) + (D(q,1,0,e)*gY) + (D(q,2,0,e)*gZ);
QQQ1[qz][qy][qx] = (D(q,0,1,e)*gX) + (D(q,1,1,e)*gY) + (D(q,2,1,e)*gZ);
QQQ2[qz][qy][qx] = (D(q,0,2,e)*gX) + (D(q,1,2,e)*gY) + (D(q,2,2,e)*gZ);
}
}
}
MFEM_SYNC_THREAD;
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1DE)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
Bt[d][q] = bt(d,q);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(dx,x,D1DE)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
u += QQQ0[qz][qy][qx] * Bt[dx][qx];
v += QQQ1[qz][qy][qx] * Bt[dx][qx];
w += QQQ2[qz][qy][qx] * Bt[dx][qx];
}
QQD0[qz][qy][dx] = u;
QQD1[qz][qy][dx] = v;
QQD2[qz][qy][dx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1DE)
{
MFEM_FOREACH_THREAD(dx,x,D1DE)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += QQD0[qz][qy][dx] * Bt[dy][qy];
v += QQD1[qz][qy][dx] * Bt[dy][qy];
w += QQD2[qz][qy][dx] * Bt[dy][qy];
}
QDD0[qz][dy][dx] = u;
QDD1[qz][dy][dx] = v;
QDD2[qz][dy][dx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1DE)
{
MFEM_FOREACH_THREAD(dy,y,D1DE)
{
MFEM_FOREACH_THREAD(dx,x,D1DE)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
u += QDD0[qz][dy][dx] * Bt[dz][qz];
v += QDD1[qz][dy][dx] * Bt[dz][qz];
w += QDD2[qz][dy][dx] * Bt[dz][qz];
}
y(dx,dy,dz,0,e) += u;
y(dx,dy,dz,1,e) += v;
y(dx,dy,dz,2,e) += w;
}
}
}
});
}
static void PAGradientApply(const int dim,
const int TR_D1D,
const int TE_D1D,
const int Q1D,
const int NE,
const Array<double> &B,
const Array<double> &G,
const Array<double> &Bt,
const Vector &op,
const Vector &x,
Vector &y,
bool transpose=false)
{
if (dim == 2)
{
return PAGradientApply2D(NE,B,G,Bt,op,x,y,TR_D1D,TE_D1D,Q1D);
}
if (dim == 3)
{
return PAGradientApply3D(NE,B,G,Bt,op,x,y,TR_D1D,TE_D1D,Q1D);
}
MFEM_ABORT("Unknown kernel.");
}
// PA Gradient Apply kernel
void GradientIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
PAGradientApply(dim, trial_dofs1D, test_dofs1D, quad1D, ne,
trial_maps->B, trial_maps->G, test_maps->Bt, pa_data, x, y,
false);
}
// PA Gradient Apply kernel
void GradientIntegrator::AddMultTransposePA(const Vector &x, Vector &y) const
{
MFEM_ABORT("PA Gradient AddMultTransposePA not implemented.");
}
} // namespace mfem
+86 -441
View File
@@ -12,7 +12,6 @@
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
#include "libceed/mass.hpp"
using namespace std;
@@ -22,26 +21,14 @@ namespace mfem
// PA Mass Integrator
// PA Mass Assemble kernel
void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assuming the same element type
fespace = &fes;
Mesh *mesh = fes.GetMesh();
if (mesh->GetNE() == 0) { return; }
const FiniteElement &el = *fes.GetFE(0);
ElementTransformation *T = mesh->GetElementTransformation(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T);
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed() && !force)
{
if (ceedDataPtr) { delete ceedDataPtr; }
CeedData* ptr = new CeedData();
ceedDataPtr = ptr;
InitCeedCoeff(Q, ptr);
return CeedPAMassAssemble(fes, *ir, *ptr);
}
#endif
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
@@ -65,7 +52,7 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
else
{
coeff.SetSize(nq * ne);
auto C = Reshape(coeff.HostWrite(), nq, ne);
auto C = Reshape(coeff.Write(), nq, ne);
for (int e = 0; e < ne; ++e)
{
ElementTransformation& T = *fes.GetElementTransformation(e);
@@ -127,324 +114,6 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
}
}
void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
SetupPA(fes);
}
template<int T_D1D = 0, int T_Q1D = 0>
static void PAMassAssembleDiagonal2D(const int NE,
const Array<double> &b,
const Vector &d,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto D = Reshape(d.Read(), Q1D, Q1D, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
double QD[MQ1][MD1];
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dy = 0; dy < D1D; ++dy)
{
QD[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
QD[qx][dy] += B(qy, dy) * B(qy, dy) * D(qx, qy, e);
}
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
for (int qx = 0; qx < Q1D; ++qx)
{
Y(dx,dy,e) += B(qx, dx) * B(qx, dx) * QD[qx][dy];
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
static void SmemPAMassAssembleDiagonal2D(const int NE,
const Array<double> &b_,
const Vector &d_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D, Q1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_SHARED double B[MQ1][MD1];
MFEM_SHARED double QDZ[NBZ][MQ1][MD1];
double (*QD)[MD1] = (double (*)[MD1])(QDZ + tidz);
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QD[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
QD[qx][dy] += B[qy][dy] * B[qy][dy] * D(qx, qy, e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
for (int qx = 0; qx < Q1D; ++qx)
{
// might need absolute values on next line
Y(dx,dy,e) += B[qx][dx] * B[qx][dx] * QD[qx][dy];
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
static void PAMassAssembleDiagonal3D(const int NE,
const Array<double> &b,
const Vector &d,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto D = Reshape(d.Read(), Q1D, Q1D, Q1D, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
double QQD[MQ1][MQ1][MD1];
double QDD[MQ1][MD1][MD1];
for (int qx = 0; qx < Q1D; ++qx)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int dz = 0; dz < D1D; ++dz)
{
QQD[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
QQD[qx][qy][dz] += B(qz, dz) * B(qz, dz) * D(qx, qy, qz, e);
}
}
}
}
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
QDD[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
QDD[qx][dy][dz] += B(qy, dy) * B(qy, dy) * QQD[qx][qy][dz];
}
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
double t = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
t += B(qx, dx) * B(qx, dx) * QDD[qx][dy][dz];
}
Y(dx, dy, dz, e) += t;
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
static void SmemPAMassAssembleDiagonal3D(const int NE,
const Array<double> &b_,
const Vector &d_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D, Q1D, Q1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
MFEM_SHARED double B[MQ1][MD1];
MFEM_SHARED double QQD[MQ1][MQ1][MD1];
MFEM_SHARED double QDD[MQ1][MD1][MD1];
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
QQD[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
QQD[qx][qy][dz] += B[qz][dz] * B[qz][dz] * D(qx, qy, qz, e);
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QDD[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
QDD[qx][dy][dz] += B[qy][dy] * B[qy][dy] * QQD[qx][qy][dz];
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double t = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
t += B[qx][dx] * B[qx][dx] * QDD[qx][dy][dz];
}
Y(dx, dy, dz, e) += t;
}
}
}
});
}
static void PAMassAssembleDiagonal(const int dim, const int D1D,
const int Q1D, const int NE,
const Array<double> &B,
const Vector &D,
Vector &Y)
{
if (dim == 2)
{
switch ((D1D << 4 ) | Q1D)
{
case 0x22: return SmemPAMassAssembleDiagonal2D<2,2,16>(NE,B,D,Y);
case 0x33: return SmemPAMassAssembleDiagonal2D<3,3,16>(NE,B,D,Y);
case 0x44: return SmemPAMassAssembleDiagonal2D<4,4,8>(NE,B,D,Y);
case 0x55: return SmemPAMassAssembleDiagonal2D<5,5,8>(NE,B,D,Y);
case 0x66: return SmemPAMassAssembleDiagonal2D<6,6,4>(NE,B,D,Y);
case 0x77: return SmemPAMassAssembleDiagonal2D<7,7,4>(NE,B,D,Y);
case 0x88: return SmemPAMassAssembleDiagonal2D<8,8,2>(NE,B,D,Y);
case 0x99: return SmemPAMassAssembleDiagonal2D<9,9,2>(NE,B,D,Y);
default: return PAMassAssembleDiagonal2D(NE,B,D,Y,D1D,Q1D);
}
}
else if (dim == 3)
{
switch ((D1D << 4 ) | Q1D)
{
case 0x23: return SmemPAMassAssembleDiagonal3D<2,3>(NE,B,D,Y);
case 0x34: return SmemPAMassAssembleDiagonal3D<3,4>(NE,B,D,Y);
case 0x45: return SmemPAMassAssembleDiagonal3D<4,5>(NE,B,D,Y);
case 0x56: return SmemPAMassAssembleDiagonal3D<5,6>(NE,B,D,Y);
case 0x67: return SmemPAMassAssembleDiagonal3D<6,7>(NE,B,D,Y);
case 0x78: return SmemPAMassAssembleDiagonal3D<7,8>(NE,B,D,Y);
case 0x89: return SmemPAMassAssembleDiagonal3D<8,9>(NE,B,D,Y);
default: return PAMassAssembleDiagonal3D(NE,B,D,Y,D1D,Q1D);
}
}
MFEM_ABORT("Unknown kernel.");
}
void MassIntegrator::AssembleDiagonalPA(Vector &diag)
{
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
}
#ifdef MFEM_USE_OCCA
// OCCA PA Mass Apply 2D kernel
static void OccaPAMassApply2D(const int D1D,
@@ -452,18 +121,18 @@ static void OccaPAMassApply2D(const int D1D,
const int NE,
const Array<double> &B,
const Array<double> &Bt,
const Vector &D,
const Vector &X,
Vector &Y)
const Vector &op,
const Vector &x,
Vector &y)
{
occa::properties props;
props["defines/D1D"] = D1D;
props["defines/Q1D"] = Q1D;
const occa::memory o_B = OccaMemoryRead(B.GetMemory(), B.Size());
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
const occa::memory o_op = OccaMemoryRead(op.GetMemory(), op.Size());
const occa::memory o_x = OccaMemoryRead(x.GetMemory(), x.Size());
occa::memory o_y = OccaMemoryReadWrite(y.GetMemory(), y.Size());
const occa_id_t id = std::make_pair(D1D,Q1D);
if (!Device::Allows(Backend::OCCA_CUDA))
{
@@ -475,7 +144,7 @@ static void OccaPAMassApply2D(const int D1D,
"MassApply2D_CPU", props);
OccaMassApply2D_cpu.emplace(id, MassApply2D_CPU);
}
OccaMassApply2D_cpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
OccaMassApply2D_cpu.at(id)(NE, o_B, o_Bt, o_op, o_x, o_y);
}
else
{
@@ -487,7 +156,7 @@ static void OccaPAMassApply2D(const int D1D,
"MassApply2D_GPU", props);
OccaMassApply2D_gpu.emplace(id, MassApply2D_GPU);
}
OccaMassApply2D_gpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
OccaMassApply2D_gpu.at(id)(NE, o_B, o_Bt, o_op, o_x, o_y);
}
}
@@ -497,18 +166,18 @@ static void OccaPAMassApply3D(const int D1D,
const int NE,
const Array<double> &B,
const Array<double> &Bt,
const Vector &D,
const Vector &X,
Vector &Y)
const Vector &op,
const Vector &x,
Vector &y)
{
occa::properties props;
props["defines/D1D"] = D1D;
props["defines/Q1D"] = Q1D;
const occa::memory o_B = OccaMemoryRead(B.GetMemory(), B.Size());
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
const occa::memory o_op = OccaMemoryRead(op.GetMemory(), op.Size());
const occa::memory o_x = OccaMemoryRead(x.GetMemory(), x.Size());
occa::memory o_y = OccaMemoryReadWrite(y.GetMemory(), y.Size());
const occa_id_t id = std::make_pair(D1D,Q1D);
if (!Device::Allows(Backend::OCCA_CUDA))
{
@@ -520,7 +189,7 @@ static void OccaPAMassApply3D(const int D1D,
"MassApply3D_CPU", props);
OccaMassApply3D_cpu.emplace(id, MassApply3D_CPU);
}
OccaMassApply3D_cpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
OccaMassApply3D_cpu.at(id)(NE, o_B, o_Bt, o_op, o_x, o_y);
}
else
{
@@ -532,16 +201,17 @@ static void OccaPAMassApply3D(const int D1D,
"MassApply3D_GPU", props);
OccaMassApply3D_gpu.emplace(id, MassApply3D_GPU);
}
OccaMassApply3D_gpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
OccaMassApply3D_gpu.at(id)(NE, o_B, o_Bt, o_op, o_x, o_y);
}
}
#endif // MFEM_USE_OCCA
template<int T_D1D = 0, int T_Q1D = 0>
template<const int T_D1D = 0,
const int T_Q1D = 0>
static void PAMassApply2D(const int NE,
const Array<double> &b_,
const Array<double> &bt_,
const Vector &d_,
const Array<double> &B_,
const Array<double> &Bt_,
const Vector &op_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
@@ -551,11 +221,11 @@ static void PAMassApply2D(const int NE,
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b_.Read(), Q1D, D1D);
auto Bt = Reshape(bt_.Read(), D1D, Q1D);
auto D = Reshape(d_.Read(), Q1D, Q1D, NE);
auto X = Reshape(x_.Read(), D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
auto B = Reshape(B_.Read(), Q1D, D1D);
auto Bt = Reshape(Bt_.Read(), D1D, Q1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
@@ -580,7 +250,7 @@ static void PAMassApply2D(const int NE,
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = X(dx,dy,e);
const double s = x(dx,dy,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx)* s;
@@ -599,7 +269,7 @@ static void PAMassApply2D(const int NE,
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] *= D(qx,qy,e);
sol_xy[qy][qx] *= op(qx,qy,e);
}
}
for (int qy = 0; qy < Q1D; ++qy)
@@ -622,18 +292,20 @@ static void PAMassApply2D(const int NE,
const double q2d = Bt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
Y(dx,dy,e) += q2d * sol_x[dx];
y(dx,dy,e) += q2d * sol_x[dx];
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
template<const int T_D1D = 0,
const int T_Q1D = 0,
const int T_NBZ = 0>
static void SmemPAMassApply2D(const int NE,
const Array<double> &b_,
const Array<double> &bt_,
const Vector &d_,
const Vector &op_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
@@ -647,9 +319,9 @@ static void SmemPAMassApply2D(const int NE,
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D, Q1D, NE);
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int tidz = MFEM_THREAD_ID(z);
@@ -677,11 +349,11 @@ static void SmemPAMassApply2D(const int NE,
}
if (tidz == 0)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][dy] = b(q,dy);
B[q][d] = b(q,d);
}
}
}
@@ -708,17 +380,17 @@ static void SmemPAMassApply2D(const int NE,
{
qq += DQ[dy][qx] * B[qy][dy];
}
QQ[qy][qx] = qq * D(qx, qy, e);
QQ[qy][qx] = qq * op(qx, qy, e);
}
}
MFEM_SYNC_THREAD;
if (tidz == 0)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
Bt[dy][q] = b(q,dy);
Bt[d][q] = b(q,d);
}
}
}
@@ -745,17 +417,18 @@ static void SmemPAMassApply2D(const int NE,
{
dd += (QD[qy][dx] * Bt[dy][qy]);
}
Y(dx, dy, e) += dd;
y(dx, dy, e) += dd;
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
template<const int T_D1D = 0,
const int T_Q1D = 0>
static void PAMassApply3D(const int NE,
const Array<double> &b_,
const Array<double> &bt_,
const Vector &d_,
const Array<double> &B_,
const Array<double> &Bt_,
const Vector &op_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
@@ -765,11 +438,11 @@ static void PAMassApply3D(const int NE,
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b_.Read(), Q1D, D1D);
auto Bt = Reshape(bt_.Read(), D1D, Q1D);
auto D = Reshape(d_.Read(), Q1D, Q1D, Q1D, NE);
auto X = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
auto B = Reshape(B_.Read(), Q1D, D1D);
auto Bt = Reshape(Bt_.Read(), D1D, Q1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -806,7 +479,7 @@ static void PAMassApply3D(const int NE,
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = X(dx,dy,dz,e);
const double s = x(dx,dy,dz,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx) * s;
@@ -839,7 +512,7 @@ static void PAMassApply3D(const int NE,
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] *= D(qx,qy,qz,e);
sol_xyz[qz][qy][qx] *= op(qx,qy,qz,e);
}
}
}
@@ -884,7 +557,7 @@ static void PAMassApply3D(const int NE,
{
for (int dx = 0; dx < D1D; ++dx)
{
Y(dx,dy,dz,e) += wz * sol_xy[dy][dx];
y(dx,dy,dz,e) += wz * sol_xy[dy][dx];
}
}
}
@@ -892,11 +565,12 @@ static void PAMassApply3D(const int NE,
});
}
template<int T_D1D = 0, int T_Q1D = 0>
template<const int T_D1D = 0,
const int T_Q1D = 0>
static void SmemPAMassApply3D(const int NE,
const Array<double> &b_,
const Array<double> &bt_,
const Vector &d_,
const Vector &op_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
@@ -909,7 +583,7 @@ static void SmemPAMassApply3D(const int NE,
MFEM_VERIFY(D1D <= M1D, "");
MFEM_VERIFY(Q1D <= M1Q, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, NE);
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
@@ -995,7 +669,7 @@ static void SmemPAMassApply3D(const int NE,
{
u += DQQ[dz][qy][qx] * B[qz][dz];
}
QQQ[qz][qy][qx] = u * d(qx,qy,qz,e);
QQQ[qz][qy][qx] = u * op(qx,qy,qz,e);
}
}
}
@@ -1067,20 +741,22 @@ static void PAMassApply(const int dim,
const int NE,
const Array<double> &B,
const Array<double> &Bt,
const Vector &D,
const Vector &X,
Vector &Y)
const Vector &op,
const Vector &x,
Vector &y)
{
#ifdef MFEM_USE_OCCA
if (DeviceCanUseOcca())
{
if (dim == 2)
{
return OccaPAMassApply2D(D1D,Q1D,NE,B,Bt,D,X,Y);
OccaPAMassApply2D(D1D, Q1D, NE, B, Bt, op, x, y);
return;
}
if (dim == 3)
{
return OccaPAMassApply3D(D1D,Q1D,NE,B,Bt,D,X,Y);
OccaPAMassApply3D(D1D, Q1D, NE, B, Bt, op, x, y);
return;
}
MFEM_ABORT("OCCA PA Mass Apply unknown kernel!");
}
@@ -1089,29 +765,29 @@ static void PAMassApply(const int dim,
{
switch ((D1D << 4) | Q1D)
{
case 0x22: return SmemPAMassApply2D<2,2,16>(NE,B,Bt,D,X,Y);
case 0x33: return SmemPAMassApply2D<3,3,16>(NE,B,Bt,D,X,Y);
case 0x44: return SmemPAMassApply2D<4,4,8>(NE,B,Bt,D,X,Y);
case 0x55: return SmemPAMassApply2D<5,5,8>(NE,B,Bt,D,X,Y);
case 0x66: return SmemPAMassApply2D<6,6,4>(NE,B,Bt,D,X,Y);
case 0x77: return SmemPAMassApply2D<7,7,4>(NE,B,Bt,D,X,Y);
case 0x88: return SmemPAMassApply2D<8,8,2>(NE,B,Bt,D,X,Y);
case 0x99: return SmemPAMassApply2D<9,9,2>(NE,B,Bt,D,X,Y);
default: return PAMassApply2D(NE,B,Bt,D,X,Y,D1D,Q1D);
case 0x22: return SmemPAMassApply2D<2,2,16>(NE, B, Bt, op, x, y);
case 0x33: return SmemPAMassApply2D<3,3,16>(NE, B, Bt, op, x, y);
case 0x44: return SmemPAMassApply2D<4,4,8>(NE, B, Bt, op, x, y);
case 0x55: return SmemPAMassApply2D<5,5,8>(NE, B, Bt, op, x, y);
case 0x66: return SmemPAMassApply2D<6,6,4>(NE, B, Bt, op, x, y);
case 0x77: return SmemPAMassApply2D<7,7,4>(NE, B, Bt, op, x, y);
case 0x88: return SmemPAMassApply2D<8,8,2>(NE, B, Bt, op, x, y);
case 0x99: return SmemPAMassApply2D<9,9,2>(NE, B, Bt, op, x, y);
default: return PAMassApply2D(NE, B, Bt, op, x, y, D1D, Q1D);
}
}
else if (dim == 3)
{
switch ((D1D << 4) | Q1D)
{
case 0x23: return SmemPAMassApply3D<2,3>(NE,B,Bt,D,X,Y);
case 0x34: return SmemPAMassApply3D<3,4>(NE,B,Bt,D,X,Y);
case 0x45: return SmemPAMassApply3D<4,5>(NE,B,Bt,D,X,Y);
case 0x56: return SmemPAMassApply3D<5,6>(NE,B,Bt,D,X,Y);
case 0x67: return SmemPAMassApply3D<6,7>(NE,B,Bt,D,X,Y);
case 0x78: return SmemPAMassApply3D<7,8>(NE,B,Bt,D,X,Y);
case 0x89: return SmemPAMassApply3D<8,9>(NE,B,Bt,D,X,Y);
default: return PAMassApply3D(NE,B,Bt,D,X,Y,D1D,Q1D);
case 0x23: return SmemPAMassApply3D<2,3>(NE, B, Bt, op, x, y);
case 0x34: return SmemPAMassApply3D<3,4>(NE, B, Bt, op, x, y);
case 0x45: return SmemPAMassApply3D<4,5>(NE, B, Bt, op, x, y);
case 0x56: return SmemPAMassApply3D<5,6>(NE, B, Bt, op, x, y);
case 0x67: return SmemPAMassApply3D<6,7>(NE, B, Bt, op, x, y);
case 0x78: return SmemPAMassApply3D<7,8>(NE, B, Bt, op, x, y);
case 0x89: return SmemPAMassApply3D<8,9>(NE, B, Bt, op, x, y);
default: return PAMassApply3D(NE, B, Bt, op, x, y, D1D, Q1D);
}
}
MFEM_ABORT("Unknown kernel.");
@@ -1119,38 +795,7 @@ static void PAMassApply(const int dim,
void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
const CeedScalar *x_ptr;
CeedScalar *y_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
x_ptr = x.Read();
y_ptr = y.ReadWrite();
}
else
{
x_ptr = x.HostRead();
y_ptr = y.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
const_cast<CeedScalar*>(x_ptr));
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorSyncArray(ceedDataPtr->v, mem);
}
else
#endif
{
PAMassApply(dim, dofs1D, quad1D, ne, maps->B, maps->Bt, pa_data, x, y);
}
PAMassApply(dim, dofs1D, quad1D, ne, maps->B, maps->Bt, pa_data, x, y);
}
} // namespace mfem
-503
View File
@@ -1,503 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
using namespace std;
namespace mfem
{
// PA Vector Diffusion Integrator
// PA Diffusion Assemble 2D kernel
static void PAVectorDiffusionSetup2D(const int Q1D,
const int NE,
const Array<double> &w,
const Vector &j,
const double COEFF,
Vector &op)
{
const int NQ = Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
auto y = Reshape(op.Write(), NQ, 3, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double c_detJ = W[q] * COEFF / ((J11*J22)-(J21*J12));
y(q,0,e) = c_detJ * (J12*J12 + J22*J22); // 1,1
y(q,1,e) = -c_detJ * (J12*J11 + J22*J21); // 1,2
y(q,2,e) = c_detJ * (J11*J11 + J21*J21); // 2,2
}
});
}
// PA Diffusion Assemble 3D kernel
static void PAVectorDiffusionSetup3D(const int Q1D,
const int NE,
const Array<double> &w,
const Vector &j,
const double COEFF,
Vector &op)
{
const int NQ = Q1D*Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
auto y = Reshape(op.Write(), NQ, 6, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J31 = J(q,2,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double J32 = J(q,2,1,e);
const double J13 = J(q,0,2,e);
const double J23 = J(q,1,2,e);
const double J33 = J(q,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double c_detJ = W[q] * COEFF / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
y(q,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
y(q,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
y(q,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
y(q,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
y(q,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
y(q,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
}
});
}
static void PAVectorDiffusionSetup(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<double> &W,
const Vector &J,
const double COEFF,
Vector &op)
{
if (!(dim == 2 || dim == 3))
{
MFEM_ABORT("Dimension not supported.");
}
if (dim == 2)
{
PAVectorDiffusionSetup2D(Q1D, NE, W, J, COEFF, op);
}
if (dim == 3)
{
PAVectorDiffusionSetup3D(Q1D, NE, W, J, COEFF, op);
}
}
void VectorDiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assumes tensor-product elements
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetFE(0);
const IntegrationRule *ir
= IntRule ? IntRule : &DiffusionIntegrator::GetRule(el, el);
const int dims = el.GetDim();
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
const int nq = ir->GetNPoints();
dim = mesh->Dimension();
ne = fes.GetNE();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(symmDims * nq * ne, Device::GetMemoryType());
double coeff = 1.0;
if (Q)
{
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient*>(Q);
MFEM_VERIFY(cQ != NULL, "only ConstantCoefficient is supported!");
coeff = cQ->constant;
}
PAVectorDiffusionSetup(dim, dofs1D, quad1D, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
// PA Diffusion Apply 2D kernel
template<int T_D1D = 0, int T_Q1D = 0> static
void PAVectorDiffusionApply2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 2;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Gt = Reshape(gt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D, 3, NE);
auto x = Reshape(_x.Read(), D1D, D1D, VDIM, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, VDIM, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
for (int c = 0; c < VDIM; ++ c)
{
double grad[max_Q1D][max_Q1D][2];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qy][qx][0] = 0.0;
grad[qy][qx][1] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double gradX[max_Q1D][2];
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] = 0.0;
gradX[qx][1] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * B(qx,dx);
gradX[qx][1] += s * G(qx,dx);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = B(qy,dy);
const double wDy = G(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qy][qx][0] += gradX[qx][1] * wy;
grad[qy][qx][1] += gradX[qx][0] * wDy;
}
}
}
// Calculate Dxy, xDy in plane
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + qy * Q1D;
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O22 = op(q,2,e);
const double gradX = grad[qy][qx][0];
const double gradY = grad[qy][qx][1];
grad[qy][qx][0] = (O11 * gradX) + (O12 * gradY);
grad[qy][qx][1] = (O12 * gradX) + (O22 * gradY);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double gradX[max_D1D][2];
for (int dx = 0; dx < D1D; ++dx)
{
gradX[dx][0] = 0;
gradX[dx][1] = 0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const double gX = grad[qy][qx][0];
const double gY = grad[qy][qx][1];
for (int dx = 0; dx < D1D; ++dx)
{
const double wx = Bt(dx,qx);
const double wDx = Gt(dx,qx);
gradX[dx][0] += gX * wDx;
gradX[dx][1] += gY * wx;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const double wy = Bt(dy,qy);
const double wDy = Gt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,c,e) += ((gradX[dx][0] * wy) + (gradX[dx][1] * wDy));
}
}
}
}
});
}
// PA Diffusion Apply 3D kernel
template<const int T_D1D = 0,
const int T_Q1D = 0> static
void PAVectorDiffusionApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
int d1d = 0, int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 3;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Gt = Reshape(gt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D*Q1D, 6, NE);
auto x = Reshape(_x.Read(), D1D, D1D, D1D, VDIM, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
for (int c = 0; c < VDIM; ++ c)
{
double grad[max_Q1D][max_Q1D][max_Q1D][3];
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qz][qy][qx][0] = 0.0;
grad[qz][qy][qx][1] = 0.0;
grad[qz][qy][qx][2] = 0.0;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
double gradXY[max_Q1D][max_Q1D][3];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
gradXY[qy][qx][0] = 0.0;
gradXY[qy][qx][1] = 0.0;
gradXY[qy][qx][2] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double gradX[max_Q1D][2];
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] = 0.0;
gradX[qx][1] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,dz,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * B(qx,dx);
gradX[qx][1] += s * G(qx,dx);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = B(qy,dy);
const double wDy = G(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
const double wx = gradX[qx][0];
const double wDx = gradX[qx][1];
gradXY[qy][qx][0] += wDx * wy;
gradXY[qy][qx][1] += wx * wDy;
gradXY[qy][qx][2] += wx * wy;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const double wz = B(qz,dz);
const double wDz = G(qz,dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qz][qy][qx][0] += gradXY[qy][qx][0] * wz;
grad[qz][qy][qx][1] += gradXY[qy][qx][1] * wz;
grad[qz][qy][qx][2] += gradXY[qy][qx][2] * wDz;
}
}
}
}
// Calculate Dxyz, xDyz, xyDz in plane
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O13 = op(q,2,e);
const double O22 = op(q,3,e);
const double O23 = op(q,4,e);
const double O33 = op(q,5,e);
const double gradX = grad[qz][qy][qx][0];
const double gradY = grad[qz][qy][qx][1];
const double gradZ = grad[qz][qy][qx][2];
grad[qz][qy][qx][0] = (O11*gradX)+(O12*gradY)+(O13*gradZ);
grad[qz][qy][qx][1] = (O12*gradX)+(O22*gradY)+(O23*gradZ);
grad[qz][qy][qx][2] = (O13*gradX)+(O23*gradY)+(O33*gradZ);
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
double gradXY[max_D1D][max_D1D][3];
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
gradXY[dy][dx][0] = 0;
gradXY[dy][dx][1] = 0;
gradXY[dy][dx][2] = 0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double gradX[max_D1D][3];
for (int dx = 0; dx < D1D; ++dx)
{
gradX[dx][0] = 0;
gradX[dx][1] = 0;
gradX[dx][2] = 0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const double gX = grad[qz][qy][qx][0];
const double gY = grad[qz][qy][qx][1];
const double gZ = grad[qz][qy][qx][2];
for (int dx = 0; dx < D1D; ++dx)
{
const double wx = Bt(dx,qx);
const double wDx = Gt(dx,qx);
gradX[dx][0] += gX * wDx;
gradX[dx][1] += gY * wx;
gradX[dx][2] += gZ * wx;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const double wy = Bt(dy,qy);
const double wDy = Gt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
gradXY[dy][dx][0] += gradX[dx][0] * wy;
gradXY[dy][dx][1] += gradX[dx][1] * wDy;
gradXY[dy][dx][2] += gradX[dx][2] * wy;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
const double wz = Bt(dz,qz);
const double wDz = Gt(dz,qz);
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,dz,c,e) +=
((gradXY[dy][dx][0] * wz) +
(gradXY[dy][dx][1] * wz) +
(gradXY[dy][dx][2] * wDz));
}
}
}
}
}
});
}
static void PAVectorDiffusionApply(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<double> &B,
const Array<double> &G,
const Array<double> &Bt,
const Array<double> &Gt,
const Vector &op,
const Vector &x,
Vector &y)
{
if (dim == 2)
{
return PAVectorDiffusionApply2D(NE,B,G,Bt,Gt,op,x,y,D1D,Q1D);
}
if (dim == 3)
{
return PAVectorDiffusionApply3D(NE,B,G,Bt,Gt,op,x,y,D1D,Q1D);
}
MFEM_ABORT("Unknown kernel.");
}
// PA Diffusion Apply kernel
void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
PAVectorDiffusionApply(dim, dofs1D, quad1D, ne,
maps->B, maps->G, maps->Bt, maps->Gt,
pa_data, x, y);
}
} // namespace mfem
-367
View File
@@ -1,367 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
using namespace std;
namespace mfem
{
// PA Mass Integrator
// PA Mass Assemble kernel
void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assuming the same element type
Mesh *mesh = fes.GetMesh();
if (mesh->GetNE() == 0) { return; }
const FiniteElement &el = *fes.GetFE(0);
ElementTransformation *T = mesh->GetElementTransformation(0);
const IntegrationRule *ir
= IntRule ? IntRule : &MassIntegrator::GetRule(el, el, *T);
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::COORDINATES |
GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(ne*nq, Device::GetMemoryType());
double coeff = 1.0;
if (Q)
{
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient*>(Q);
MFEM_VERIFY(cQ != NULL, "Only ConstantCoefficient is supported.");
coeff = cQ->constant;
}
if (!(dim == 2 || dim == 3))
{
MFEM_ABORT("Dimension not supported.");
}
if (dim == 2)
{
const double constant = coeff;
const int NE = ne;
const int NQ = nq;
auto w = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
auto v = Reshape(pa_data.Write(), NQ, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J12 = J(q,1,0,e);
const double J21 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double detJ = (J11*J22)-(J21*J12);
v(q,e) = w[q] * constant * detJ;
}
});
}
if (dim == 3)
{
const double constant = coeff;
const int NE = ne;
const int NQ = nq;
auto W = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
auto v = Reshape(pa_data.Write(), NQ,NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e), J12 = J(q,0,1,e), J13 = J(q,0,2,e);
const double J21 = J(q,1,0,e), J22 = J(q,1,1,e), J23 = J(q,1,2,e);
const double J31 = J(q,2,0,e), J32 = J(q,2,1,e), J33 = J(q,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
v(q,e) = W[q] * constant * detJ;
}
});
}
}
template<const int T_D1D = 0,
const int T_Q1D = 0>
static void PAVectorMassApply2D(const int NE,
const Array<double> &_B,
const Array<double> &_Bt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 2;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(_B.Read(), Q1D, D1D);
auto Bt = Reshape(_Bt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D, Q1D, NE);
auto x = Reshape(_x.Read(), D1D, D1D, VDIM, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, VDIM, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double sol_xy[max_Q1D][max_Q1D];
for (int c = 0; c < VDIM; ++c)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double sol_x[max_Q1D];
for (int qy = 0; qy < Q1D; ++qy)
{
sol_x[qy] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx)* s;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double d2q = B(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] += d2q * sol_x[qx];
}
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] *= op(qx,qy,e);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double sol_x[max_D1D];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] = 0.0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const double s = sol_xy[qy][qx];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] += Bt(dx,qx) * s;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const double q2d = Bt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,c,e) += q2d * sol_x[dx];
}
}
}
}
});
}
template<const int T_D1D = 0,
const int T_Q1D = 0>
static void PAVectorMassApply3D(const int NE,
const Array<double> &_B,
const Array<double> &_Bt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 3;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(_B.Read(), Q1D, D1D);
auto Bt = Reshape(_Bt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D, Q1D, Q1D, NE);
auto x = Reshape(_x.Read(), D1D, D1D, D1D, VDIM, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double sol_xyz[max_Q1D][max_Q1D][max_Q1D];
for (int c = 0; c < VDIM; ++ c)
{
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] = 0.0;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
double sol_xy[max_Q1D][max_Q1D];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double sol_x[max_Q1D];
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] = 0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,dz,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx) * s;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = B(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] += wy * sol_x[qx];
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const double wz = B(qz,dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] += wz * sol_xy[qy][qx];
}
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] *= op(qx,qy,qz,e);
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
double sol_xy[max_D1D][max_D1D];
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
sol_xy[dy][dx] = 0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double sol_x[max_D1D];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] = 0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const double s = sol_xyz[qz][qy][qx];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] += Bt(dx,qx) * s;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const double wy = Bt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
sol_xy[dy][dx] += wy * sol_x[dx];
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
const double wz = Bt(dz,qz);
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,dz,c,e) += wz * sol_xy[dy][dx];
}
}
}
}
}
});
}
static void PAVectorMassApply(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<double> &B,
const Array<double> &Bt,
const Vector &op,
const Vector &x,
Vector &y)
{
if (dim == 2)
{
return PAVectorMassApply2D(NE, B, Bt, op, x, y, D1D, Q1D);
}
if (dim == 3)
{
return PAVectorMassApply3D(NE, B, Bt, op, x, y, D1D, Q1D);
}
MFEM_ABORT("Unknown kernel.");
}
void VectorMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
PAVectorMassApply(dim, dofs1D, quad1D, ne, maps->B, maps->Bt, pa_data, x, y);
}
} // namespace mfem
-1125
View File
File diff suppressed because it is too large Load Diff
-604
View File
@@ -1,604 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_COMPLEX_FEM
#define MFEM_COMPLEX_FEM
#include "../linalg/complex_operator.hpp"
#include "gridfunc.hpp"
#include "linearform.hpp"
#include "bilinearform.hpp"
#ifdef MFEM_USE_MPI
#include "pgridfunc.hpp"
#include "plinearform.hpp"
#include "pbilinearform.hpp"
#endif
#include <complex>
namespace mfem
{
/// Class for complex-valued grid function - real + imaginary part Vector with
/// associated FE space.
class ComplexGridFunction : public Vector
{
private:
GridFunction * gfr;
GridFunction * gfi;
protected:
void Destroy() { delete gfr; delete gfi; }
public:
/* @brief Construct a ComplexGridFunction associated with the
FiniteElementSpace @a *f. */
ComplexGridFunction(FiniteElementSpace *f);
void Update();
/// Assign constant values to the ComplexGridFunction data.
ComplexGridFunction &operator=(const std::complex<double> & value)
{ *gfr = value.real(); *gfi = value.imag(); return *this; }
virtual void ProjectCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff);
virtual void ProjectCoefficient(VectorCoefficient &real_vcoeff,
VectorCoefficient &imag_vcoeff);
virtual void ProjectBdrCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff,
Array<int> &attr);
virtual void ProjectBdrCoefficientNormal(VectorCoefficient &real_coeff,
VectorCoefficient &imag_coeff,
Array<int> &attr);
virtual void ProjectBdrCoefficientTangent(VectorCoefficient &real_coeff,
VectorCoefficient &imag_coeff,
Array<int> &attr);
FiniteElementSpace *FESpace() { return gfr->FESpace(); }
const FiniteElementSpace *FESpace() const { return gfr->FESpace(); }
GridFunction & real() { return *gfr; }
GridFunction & imag() { return *gfi; }
const GridFunction & real() const { return *gfr; }
const GridFunction & imag() const { return *gfi; }
/// Destroys the grid function.
virtual ~ComplexGridFunction() { Destroy(); }
};
/** Class for a complex-valued linear form
The @a convention argument in the class's constructor is documented in the
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
When supplying integrators to the ComplexLinearForm either the real or
imaginary integrator can be NULL. This indicates that the corresponding
portion of the complex-valued field is equal to zero.
*/
class ComplexLinearForm : public Vector
{
private:
ComplexOperator::Convention conv;
protected:
LinearForm * lfr;
LinearForm * lfi;
public:
ComplexLinearForm(FiniteElementSpace *fes,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
/** @brief Create a ComplexLinearForm on the FiniteElementSpace @a f, using
the same integrators as the LinearForms @a lfr (real) and @a lfi (imag) .
The pointer @a fes is not owned by the newly constructed object.
The integrators are copied as pointers and they are not owned by the
newly constructed ComplexLinearForm. */
ComplexLinearForm(FiniteElementSpace *fes, LinearForm *lf_r, LinearForm *lf_i,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
virtual ~ComplexLinearForm();
ComplexOperator::Convention GetConvention() const { return conv; }
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
/// Adds new Domain Integrator.
void AddDomainIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/// Adds new Boundary Integrator.
void AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/** @brief Add new Boundary Integrator, restricted to the given boundary
attributes.
Assumes ownership of @a lfi_real and @a lfi_imag.
The array @a bdr_attr_marker is stored internally as a pointer to the
given Array<int> object. */
void AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker);
/// Adds new Boundary Face Integrator. Assumes ownership of @a lfi.
void AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/** @brief Add new Boundary Face Integrator, restricted to the given boundary
attributes.
Assumes ownership of @a lfi_real and @a lfi_imag.
The array @a bdr_attr_marker is stored internally as a pointer to the
given Array<int> object. */
void AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker);
FiniteElementSpace *FESpace() const { return lfr->FESpace(); }
LinearForm & real() { return *lfr; }
LinearForm & imag() { return *lfi; }
const LinearForm & real() const { return *lfr; }
const LinearForm & imag() const { return *lfi; }
void Update();
void Update(FiniteElementSpace *f);
/// Assembles the linear form i.e. sums over all domain/bdr integrators.
void Assemble();
std::complex<double> operator()(const ComplexGridFunction &gf) const;
};
/** Class for sesquilinear form
A sesquilinear form is a generalization of a bilinear form to complex-valued
fields. Sesquilinear forms are linear in the second argument but the first
argument involves a complex conjugate in the sense that:
a(alpha u, beta v) = conj(alpha) beta a(u, v)
The @a convention argument in the class's constructor is documented in the
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
When supplying integrators to the SesquilinearForm either the real or
imaginary integrator can be NULL. This indicates that the corresponding
portion of the complex-valued material coefficient is equal to zero.
*/
class SesquilinearForm
{
private:
ComplexOperator::Convention conv;
/** This data member allows one to specify what should be done to the
diagonal matrix entries and corresponding RHS values upon elimination of
the constrained DoFs. */
mfem::Matrix::DiagonalPolicy diag_policy = mfem::Matrix::DIAG_ONE;
BilinearForm *blfr;
BilinearForm *blfi;
/* These methods check if the real/imag parts of the sesqulinear form are not
empty */
bool RealInteg();
bool ImagInteg();
public:
SesquilinearForm(FiniteElementSpace *fes,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
/** @brief Create a SesquilinearForm on the FiniteElementSpace @a f, using
the same integrators as the BilinearForms @a bfr and @a bfi .
The pointer @a fes is not owned by the newly constructed object.
The integrators are copied as pointers and they are not owned by the
newly constructed SesquilinearForm. */
SesquilinearForm(FiniteElementSpace *fes, BilinearForm *bfr, BilinearForm *bfi,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
ComplexOperator::Convention GetConvention() const { return conv; }
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
BilinearForm & real() { return *blfr; }
BilinearForm & imag() { return *blfi; }
const BilinearForm & real() const { return *blfr; }
const BilinearForm & imag() const { return *blfi; }
/// Adds new Domain Integrator.
void AddDomainIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/// Adds new Boundary Integrator.
void AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/// Adds new Boundary Integrator, restricted to specific boundary attributes.
void AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> &bdr_marker);
/// Adds new interior Face Integrator. Assumes ownership of @a bfi.
void AddInteriorFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/// Adds new boundary Face Integrator. Assumes ownership of @a bfi.
void AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/** @brief Adds new boundary Face Integrator, restricted to specific boundary
attributes.
Assumes ownership of @a bfi.
The array @a bdr_marker is stored internally as a pointer to the given
Array<int> object. */
void AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> &bdr_marker);
/// Assemble the local matrix
void Assemble(int skip_zeros = 1);
/// Finalizes the matrix initialization.
void Finalize(int skip_zeros = 1);
/// Returns the matrix assembled on the true dofs, i.e. P^t A P.
/** The returned matrix has to be deleted by the caller. */
ComplexSparseMatrix *AssembleComplexSparseMatrix();
/// Return the parallel FE space associated with the ParBilinearForm.
FiniteElementSpace *FESpace() const { return blfr->FESpace(); }
void FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B,
int copy_interior = 0);
void FormSystemMatrix(const Array<int> &ess_tdof_list,
OperatorHandle &A);
/** Call this method after solving a linear system constructed using the
FormLinearSystem method to recover the solution as a ParGridFunction-size
vector in x. Use the same arguments as in the FormLinearSystem call. */
virtual void RecoverFEMSolution(const Vector &X, const Vector &b, Vector &x);
virtual void Update(FiniteElementSpace *nfes = NULL);
/// Sets diagonal policy used upon construction of the linear system
void SetDiagonalPolicy(mfem::Matrix::DiagonalPolicy dpolicy);
/// Returns the diagonal policy of the sesquilinear form
Matrix::DiagonalPolicy GetDiagonalPolicy() const {return diag_policy;}
virtual ~SesquilinearForm();
};
#ifdef MFEM_USE_MPI
/// Class for parallel complex-valued grid function - real + imaginary part
/// Vector with associated parallel FE space.
class ParComplexGridFunction : public Vector
{
private:
ParGridFunction * pgfr;
ParGridFunction * pgfi;
protected:
void Destroy() { delete pgfr; delete pgfi; }
public:
/* @brief Construct a ParComplexGridFunction associated with the
ParFiniteElementSpace @a *f. */
ParComplexGridFunction(ParFiniteElementSpace *pf);
void Update();
/// Assign constant values to the ParComplexGridFunction data.
ParComplexGridFunction &operator=(const std::complex<double> & value)
{ *pgfr = value.real(); *pgfi = value.imag(); return *this; }
virtual void ProjectCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff);
virtual void ProjectCoefficient(VectorCoefficient &real_vcoeff,
VectorCoefficient &imag_vcoeff);
virtual void ProjectBdrCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff,
Array<int> &attr);
virtual void ProjectBdrCoefficientNormal(VectorCoefficient &real_coeff,
VectorCoefficient &imag_coeff,
Array<int> &attr);
virtual void ProjectBdrCoefficientTangent(VectorCoefficient &real_coeff,
VectorCoefficient &imag_coeff,
Array<int> &attr);
void Distribute(const Vector *tv);
void Distribute(const Vector &tv) { Distribute(&tv); }
/// Returns the vector restricted to the true dofs.
void ParallelProject(Vector &tv) const;
FiniteElementSpace *FESpace() { return pgfr->FESpace(); }
const FiniteElementSpace *FESpace() const { return pgfr->FESpace(); }
ParFiniteElementSpace *ParFESpace() { return pgfr->ParFESpace(); }
const ParFiniteElementSpace *ParFESpace() const { return pgfr->ParFESpace(); }
ParGridFunction & real() { return *pgfr; }
ParGridFunction & imag() { return *pgfi; }
const ParGridFunction & real() const { return *pgfr; }
const ParGridFunction & imag() const { return *pgfi; }
virtual double ComputeL2Error(Coefficient &exsolr, Coefficient &exsoli,
const IntegrationRule *irs[] = NULL) const
{
double err_r = pgfr->ComputeL2Error(exsolr, irs);
double err_i = pgfi->ComputeL2Error(exsoli, irs);
return sqrt(err_r * err_r + err_i * err_i);
}
virtual double ComputeL2Error(VectorCoefficient &exsolr,
VectorCoefficient &exsoli,
const IntegrationRule *irs[] = NULL,
Array<int> *elems = NULL) const
{
double err_r = pgfr->ComputeL2Error(exsolr, irs, elems);
double err_i = pgfi->ComputeL2Error(exsoli, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
}
/// Destroys grid function.
virtual ~ParComplexGridFunction() { Destroy(); }
};
/** Class for a complex-valued, parallel linear form
The @a convention argument in the class's constructor is documented in the
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
When supplying integrators to the ParComplexLinearForm either the real or
imaginary integrator can be NULL. This indicates that the corresponding
portion of the complex-valued field is equal to zero.
*/
class ParComplexLinearForm : public Vector
{
private:
ComplexOperator::Convention conv;
protected:
ParLinearForm * plfr;
ParLinearForm * plfi;
HYPRE_Int * tdof_offsets;
public:
ParComplexLinearForm(ParFiniteElementSpace *pf,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
/** @brief Create a ParComplexLinearForm on the ParFiniteElementSpace @a pf,
using the same integrators as the LinearForms @a plfr (real) and @a plfi
(imag) .
The pointer @a fes is not owned by the newly constructed object.
The integrators are copied as pointers and they are not owned by the newly
constructed ParComplexLinearForm. */
ParComplexLinearForm(ParFiniteElementSpace *pf, ParLinearForm *plf_r,
ParLinearForm *plf_i,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
virtual ~ParComplexLinearForm();
ComplexOperator::Convention GetConvention() const { return conv; }
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
/// Adds new Domain Integrator.
void AddDomainIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/// Adds new Boundary Integrator.
void AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/** @brief Add new Boundary Integrator, restricted to the given boundary
attributes.
Assumes ownership of @a lfi_real and @a lfi_imag.
The array @a bdr_attr_marker is stored internally as a pointer to the
given Array<int> object. */
void AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker);
/// Adds new Boundary Face Integrator. Assumes ownership of @a lfi.
void AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/** @brief Add new Boundary Face Integrator, restricted to the given boundary
attributes.
Assumes ownership of @a lfi_real and @a lfi_imag.
The array @a bdr_attr_marker is stored internally as a pointer to the
given Array<int> object. */
void AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker);
ParFiniteElementSpace *ParFESpace() const { return plfr->ParFESpace(); }
ParLinearForm & real() { return *plfr; }
ParLinearForm & imag() { return *plfi; }
const ParLinearForm & real() const { return *plfr; }
const ParLinearForm & imag() const { return *plfi; }
void Update(ParFiniteElementSpace *pf = NULL);
/// Assembles the linear form i.e. sums over all domain/bdr integrators.
void Assemble();
/// Assemble the vector on the true dofs, i.e. P^t v.
void ParallelAssemble(Vector &tv);
/// Returns the vector assembled on the true dofs, i.e. P^t v.
HypreParVector *ParallelAssemble();
std::complex<double> operator()(const ParComplexGridFunction &gf) const;
};
/** Class for a parallel sesquilinear form
A sesquilinear form is a generalization of a bilinear form to complex-valued
fields. Sesquilinear forms are linear in the second argument but but the
first argument involves a complex conjugate in the sense that:
a(alpha u, beta v) = conj(alpha) beta a(u, v)
The @a convention argument in the class's constructor is documented in the
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
When supplying integrators to the ParSesquilinearForm either the real or
imaginary integrator can be NULL. This indicates that the corresponding
portion of the complex-valued material coefficient is equal to zero.
*/
class ParSesquilinearForm
{
private:
ComplexOperator::Convention conv;
ParBilinearForm *pblfr;
ParBilinearForm *pblfi;
/* These methods check if the real/imag parts of the sesqulinear form are not
empty */
bool RealInteg();
bool ImagInteg();
public:
ParSesquilinearForm(ParFiniteElementSpace *pf,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
/** @brief Create a ParSesquilinearForm on the ParFiniteElementSpace @a pf,
using the same integrators as the ParBilinearForms @a pbfr and @a pbfi .
The pointer @a pf is not owned by the newly constructed object.
The integrators are copied as pointers and they are not owned by the
newly constructed ParSesquilinearForm. */
ParSesquilinearForm(ParFiniteElementSpace *pf, ParBilinearForm *pbfr,
ParBilinearForm *pbfi,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
ComplexOperator::Convention GetConvention() const { return conv; }
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
ParBilinearForm & real() { return *pblfr; }
ParBilinearForm & imag() { return *pblfi; }
const ParBilinearForm & real() const { return *pblfr; }
const ParBilinearForm & imag() const { return *pblfi; }
/// Adds new Domain Integrator.
void AddDomainIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/// Adds new Boundary Integrator.
void AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/** @brief Adds new boundary Integrator, restricted to specific boundary
attributes.
Assumes ownership of @a bfi.
The array @a bdr_marker is stored internally as a pointer to the given
Array<int> object. */
void AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> &bdr_marker);
/// Adds new interior Face Integrator. Assumes ownership of @a bfi.
void AddInteriorFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/// Adds new boundary Face Integrator. Assumes ownership of @a bfi.
void AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/** @brief Adds new boundary Face Integrator, restricted to specific boundary
attributes.
Assumes ownership of @a bfi.
The array @a bdr_marker is stored internally as a pointer to the given
Array<int> object. */
void AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> &bdr_marker);
/// Assemble the local matrix
void Assemble(int skip_zeros = 1);
/// Finalizes the matrix initialization.
void Finalize(int skip_zeros = 1);
/// Returns the matrix assembled on the true dofs, i.e. P^t A P.
/** The returned matrix has to be deleted by the caller. */
ComplexHypreParMatrix *ParallelAssemble();
/// Return the parallel FE space associated with the ParBilinearForm.
ParFiniteElementSpace *ParFESpace() const { return pblfr->ParFESpace(); }
void FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B,
int copy_interior = 0);
void FormSystemMatrix(const Array<int> &ess_tdof_list,
OperatorHandle &A);
/** Call this method after solving a linear system constructed using the
FormLinearSystem method to recover the solution as a ParGridFunction-size
vector in x. Use the same arguments as in the FormLinearSystem call. */
virtual void RecoverFEMSolution(const Vector &X, const Vector &b, Vector &x);
virtual void Update(FiniteElementSpace *nfes = NULL);
virtual ~ParSesquilinearForm();
};
#endif // MFEM_USE_MPI
}
#endif // MFEM_COMPLEX_FEM
+1 -369
View File
@@ -14,6 +14,7 @@
#include "../general/text.hpp"
#include "picojson.h"
#include <fstream>
#include <cerrno> // errno
#include <sstream>
@@ -725,373 +726,4 @@ void VisItDataCollection::ParseVisItRootString(const std::string& json)
}
}
ParaViewDataCollection::~ParaViewDataCollection()
{
if (myrank==0)
{
// Close the data collection
pvd_stream << "</Collection>" << std::endl;
pvd_stream << "</VTKFile>" << std::endl;
pvd_stream.close();
}
}
ParaViewDataCollection::ParaViewDataCollection(const std::string&
collection_name,
mfem::Mesh *mesh_)
:DataCollection(collection_name, mesh_)
{
myrank = 0;
nprocs = 1;
levels_of_detail = 1;
#ifdef MFEM_USE_MPI
lcomm = MPI_COMM_SELF;
#endif
std::string dpath=GenerateCollectionPath();
std::string pvdname=dpath+"/"+GeneratePVDFileName();
create_directory(dpath); // this one is a serial
pvd_stream.open(pvdname.c_str(),std::ios::out);
// initialize the file
pvd_stream << "<?xml version=\"1.0\"?>" << std::endl;
pvd_stream << "<VTKFile type=\"Collection\" version=\"0.1\"" << std::endl;
pvd_stream << " byte_order=\"LittleEndian\"" << std::endl;
pvd_stream << " compressor=\"vtkZLibDataCompressor\">" << std::endl;
pvd_stream << "<Collection>" << std::endl;
}
void ParaViewDataCollection::SetMesh(mfem::Mesh * new_mesh)
{
DataCollection::SetMesh(new_mesh);
}
void ParaViewDataCollection::RegisterField(const std::string& field_name,
mfem::GridFunction *gf)
{
DataCollection::RegisterField(field_name,gf);
}
void ParaViewDataCollection::SetLevelsOfDetail(int levels_of_detail_)
{
levels_of_detail = levels_of_detail_;
}
void ParaViewDataCollection::Load(int )
{
MFEM_WARNING("ParaViewDataCollection::Load() is not implemented!");
}
std::string ParaViewDataCollection::GenerateCollectionPath()
{
std::string out = "";
out=DataCollection::GetPrefixPath() + DataCollection::GetCollectionName();
return out;
}
std::string ParaViewDataCollection::GeneratePVTUPath()
{
std::string out = "Cycle" + to_padded_string(cycle,pad_digits_cycle);
return out;
}
std::string ParaViewDataCollection::GenerateVTUPath()
{
std::string out = GeneratePVTUPath();
return out;
}
std::string ParaViewDataCollection::GeneratePVDFileName()
{
std::string out = GetCollectionName()+".pvd";
return out;
}
std::string ParaViewDataCollection::GeneratePVTUFileName()
{
std::string out = "data.pvtu";
return out;
}
std::string ParaViewDataCollection::GenerateVTUFileName()
{
std::string out = "proc" + to_padded_string(myrank,pad_digits_rank)+".vtu";
return out;
}
std::string ParaViewDataCollection::GenerateVTUFileName(int crank)
{
std::string out = "proc" + to_padded_string(crank,pad_digits_rank)+".vtu";
return out;
}
void ParaViewDataCollection::Save()
{
// add a new collection to the PDV file
// check if the directories are created
{
std::string path = GenerateCollectionPath()+"/"+GenerateVTUPath();
#ifndef MFEM_USE_MPI
int err = create_directory(path);
#else
int err;
if (nprocs==1)
{
err = create_directory(path);
}
else
{
err = create_directory(path,myrank,lcomm);
}
#endif
if (err)
{
error = WRITE_ERROR;
MFEM_WARNING("Error creating directory: " << path);
return; // do not even try to write the mesh
}
}
// the directory is created
// define the vtu file
{
std::string fname = GenerateCollectionPath()+"/"+GenerateVTUPath()+"/"
+GenerateVTUFileName();
std::fstream out; out.open(fname.c_str(), std::ios::out);
SaveDataVTU(out,levels_of_detail);
out.close();
}
// define the pvtu file only on process 0
if (myrank==0)
{
std::string fname = GenerateCollectionPath()+"/"+GeneratePVTUPath()+"/"
+GeneratePVTUFileName();
std::fstream out; out.open(fname.c_str(), std::ios::out);
out << "<?xml version=\"1.0\"?>" << std::endl;
out << "<VTKFile type=\"PUnstructuredGrid\"";
out << " version =\"0.1\" byte_order=\"LittleEndian\"> " << std::endl;
out << "<PUnstructuredGrid GhostLevel=\"0\">" << std::endl ;
out << "<PPoints>" << std::endl;
out << "\t<PDataArray type=\"Float64\" ";
out << " Name=\"Points\" NumberOfComponents=\"3\"/>" << std::endl;
out << "</PPoints>" << std::endl;
out << "<PCells>" << std::endl ;
out << "\t<PDataArray type=\"Int32\" ";
out << " Name=\"connectivity\" NumberOfComponents=\"1\"/>" << std::endl ;
out << "\t<PDataArray type=\"Int32\" ";
out << " Name=\"offsets\" NumberOfComponents=\"1\"/>" << std::endl ;
out << "\t<PDataArray type=\"UInt8\" ";
out << " Name=\"types\" NumberOfComponents=\"1\"/>" << std::endl ;
out << "</PCells>" << std::endl ;
out << "<PPointData>" << std::endl ;
for (FieldMapIterator it=field_map.begin(); it!=field_map.end(); ++it)
{
out << "<PDataArray type=\"Float64\" Name=\"" << it->first;
int vec_dim=it->second->VectorDim();
out<<"\" NumberOfComponents=\""<< vec_dim <<"\" format=\"ascii\" />" <<
std::endl;
}
out << "</PPointData>" << std::endl ;
// CELL DATA
out << "<PCellData>" << std::endl ;
out << "\t<PDataArray type=\"Int32\" Name=\"" << "material"
<<"\" NumberOfComponents=\"1\"/> " << std::endl ;
out << "</PCellData>" << std::endl ;
for (int ii=0; ii<nprocs; ii++)
{
// this one is generated without the path
std::string nfname=GenerateVTUFileName(ii);
out << "<Piece Source=\"" << nfname << "\"/>" << std::endl;
}
out << "</PUnstructuredGrid>" << std::endl;
out << "</VTKFile>" << std::endl;
out.close();
fname = GeneratePVTUPath()+"/"+GeneratePVTUFileName();
// add the pvtu file to the pvd_stream
pvd_stream << "<DataSet timestep=\"" << GetTime(); // GetCycle();
pvd_stream << "\" group=\"\" part=\"" << 0 << "\" file=\"";
pvd_stream << fname << "\"/>" << std::endl;
}
}
void ParaViewDataCollection::SaveDataVTU(std::ostream &out, int ref)
{
out << "<VTKFile type=\"UnstructuredGrid\" ";
out << " version=\"0.1\" byte_order=\"LittleEndian\">" << std::endl;
out << "<UnstructuredGrid>" << std::endl;
mesh->PrintVTU(out,ref);
// dump out the grid functions as point data
out << "<PointData >" << std::endl;
// save the grid functions
// iterate over all grid functions
for (FieldMapIterator it=field_map.begin(); it!=field_map.end(); ++it)
{
SaveGFieldVTU(out,ref,it);
}
// iterate over all quadrature functions
// if the Quadrature functions are dumped as cell data
// the cycle should be moved before the grid functions
// and the PrintVTU CellData section should be open in the mesh dump
for (QFieldMapIterator it=q_field_map.begin(); it!=q_field_map.end(); ++it)
{
// save the quadrature functions
// this one is not implemented yet
SaveQFieldVTU(out,ref,it);
}
out << "</PointData>" << std::endl;
// close the mesh
out << "</Piece>" << std::endl; // close the piece open in the PrintVTU method
out << "</UnstructuredGrid>" << std::endl;
out << "</VTKFile>" << std::endl;
}
void ParaViewDataCollection::SaveQFieldVTU(std::ostream &out, int ref,
const QFieldMapIterator& it )
{
MFEM_WARNING("SaveQFieldVTU is wotk in progress - field name:"<<it->second);
}
void ParaViewDataCollection::SaveGFieldVTU(std::ostream &out, int ref_,
const FieldMapIterator& it)
{
RefinedGeometry *RefG;
Vector val;
DenseMatrix vval, pmat;
int vec_dim = it->second->VectorDim();
if (vec_dim == 1)
{
// scalar data
out << "<DataArray type=\"Float64\" Name=\"" << it->first;
out << "\" NumberOfComponents=\"1\" format=\"ascii\" >" << std::endl;
for (int i = 0; i < mesh->GetNE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetElementBaseGeometry(i), ref_, 1);
it->second->GetValues(i, RefG->RefPts, val, pmat);
for (int j = 0; j < val.Size(); j++)
{
out << val(j) << '\n';
}
}
}
else
{
// vector data
out << "<DataArray type=\"Float64\" Name=\"" << it->first;
out << "\" NumberOfComponents=\"" << vec_dim << "\" format=\"ascii\" >" <<
std::endl;
for (int i = 0; i < mesh->GetNE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetElementBaseGeometry(i), ref_, 1);
it->second->GetVectorValues(i, RefG->RefPts, vval, pmat);
for (int jj = 0; jj < vval.Width(); jj++)
{
for (int ii = 0; ii < vval.Height(); ii++)
{
out << vval(ii, jj) << ' ';
}
out << std::endl;
}
}
}
out << "</DataArray>" << std::endl;
out.flush();
}
int ParaViewDataCollection::create_directory(const std::string &dir_name)
{
// create directories recursively
const char path_delim = '/';
std::string::size_type pos = 0;
int err;
do
{
pos = dir_name.find(path_delim, pos+1);
std::string subdir = dir_name.substr(0, pos);
err = mkdir(subdir.c_str(), 0777);
err = (err && (errno != EEXIST)) ? 1 : 0;
}
while ( pos != std::string::npos );
return err;
}
#ifdef MFEM_USE_MPI
ParaViewDataCollection::ParaViewDataCollection(const std::string&
collection_name,
mfem::ParMesh *mesh_)
:DataCollection(collection_name,mesh_)
{
lcomm = mesh_->GetComm();
MPI_Comm_rank(lcomm, &myrank);
MPI_Comm_size(lcomm, &nprocs);
levels_of_detail = 1;
std::string dpath = GenerateCollectionPath();
std::string pvdname = dpath+"/"+GeneratePVDFileName();
int err = create_directory(dpath,myrank,lcomm);
if (err) { MFEM_ABORT("Cannot create the directory:"<<dpath);}
if (myrank==0)
{
pvd_stream.open(pvdname.c_str(),std::ios::out);
pvd_stream << "<?xml version=\"1.0\"?>" << std::endl;
pvd_stream << "<VTKFile type=\"Collection\" version=\"0.1\"" << std::endl;
pvd_stream << " byte_order=\"LittleEndian\"" << std::endl;
pvd_stream << " compressor=\"vtkZLibDataCompressor\">" << std::endl;
pvd_stream << "<Collection>" << std::endl;
}
}
int ParaViewDataCollection::create_directory(const std::string &dir_name,
int myid,
MPI_Comm lcomm_)
{
// create directories recursively
const char path_delim = '/';
std::string::size_type pos = 0;
int err;
// create the directories only on process 0
if (myid==0)
{
do
{
pos = dir_name.find(path_delim, pos+1);
std::string subdir = dir_name.substr(0, pos);
err = mkdir(subdir.c_str(), 0777);
err = (err && (errno != EEXIST)) ? 1 : 0;
}
while ( pos != std::string::npos );
}
// broadcast the error
MPI_Bcast(&err, 1, MPI_INT, 0, lcomm_);
return err;
}
void ParaViewDataCollection::SetMesh(MPI_Comm comm, mfem::Mesh *new_mesh)
{
DataCollection::SetMesh(new_mesh);
lcomm = comm;
MPI_Comm_rank(comm, &myrank);
MPI_Comm_size(comm, &nprocs);
}
#endif
} // end namespace MFEM
+1 -74
View File
@@ -19,7 +19,6 @@
#endif
#include <string>
#include <map>
#include <fstream>
namespace mfem
{
@@ -468,78 +467,6 @@ public:
virtual ~VisItDataCollection() {}
};
/// Helper class for ParaView visualization data
class ParaViewDataCollection : public DataCollection
{
private:
#ifdef MFEM_USE_MPI
MPI_Comm lcomm;
#endif
int myrank;
int nprocs;
int levels_of_detail;
std::fstream pvd_stream;
protected:
void SaveDataVTU(std::ostream &out, int ref);
void SaveGFieldVTU(std::ostream& out, int ref_, const FieldMapIterator& it);
void SaveQFieldVTU(std::ostream &out, int ref, const QFieldMapIterator& it);
std::string GenerateCollectionPath();
std::string GenerateVTUFileName();
std::string GenerateVTUFileName(int rank);
std::string GenerateVTUPath();
std::string GeneratePVDFileName();
std::string GeneratePVTUFileName();
std::string GeneratePVTUPath();
public:
/// Constructor. The collection name is used when saving the data.
/** If @a mesh_ is NULL, then the mesh can be set later by calling SetMesh().
The constructor works only in serial. */
ParaViewDataCollection(const std::string& collection_name,
mfem::Mesh *mesh_ = NULL);
#ifdef MFEM_USE_MPI
/// Construct a parallel ParaViewDataCollection.
/** Before saving the data collection, some parameters in the collection can
be adjusted, e.g. SetPadDigits(), SetPrefixPath(), etc. */
ParaViewDataCollection(const std::string& collection_name,
mfem::ParMesh *mesh_ = NULL);
#endif
virtual ~ParaViewDataCollection() override;
virtual void SetMesh(mfem::Mesh * new_mesh) override;
#ifdef MFEM_USE_MPI
/// Set/change the mesh associated with the collection.
virtual void SetMesh(MPI_Comm comm, mfem::Mesh *new_mesh) override;
#endif
/// Add a grid function to the collection
virtual void RegisterField(const std::string& field_name,
mfem::GridFunction *gf) override;
/// Set refinement levels - every element is uniformly split based on
/// levels_of_detail_
void SetLevelsOfDetail(int levels_of_detail_);
/// Save the collection - the directory name is constructed based on the
/// cycle value
virtual void Save() override;
/// Load the collection - not implemented in the ParaView writer
virtual void Load(int cycle_ = 0) override;
static int create_directory(const std::string &dir_name);
#ifdef MFEM_USE_MPI
static int create_directory(const std::string &dir_name, int myid,
MPI_Comm mycom);
#endif
};
}
#endif
-20
View File
@@ -413,26 +413,6 @@ const DenseMatrix &IsoparametricTransformation::EvalJacobian()
return dFdx;
}
const DenseMatrix &IsoparametricTransformation::EvalHessian()
{
MFEM_ASSERT(space_dim == PointMat.Height(),
"the IsoparametricTransformation has not been finalized;"
" call FinilizeTransformation() after setup");
MFEM_ASSERT((EvalState & HESSIAN_MASK) == 0, "");
int Dim = FElem->GetDim();
d2shape.SetSize(FElem->GetDof(), (Dim*(Dim+1))/2);
d2Fdx2.SetSize(PointMat.Height(), d2shape.Width());
if (d2shape.Width() > 0)
{
FElem->CalcHessian(*IntPoint, d2shape);
Mult(PointMat, d2shape, d2Fdx2);
}
EvalState |= HESSIAN_MASK;
return d2Fdx2;
}
int IsoparametricTransformation::OrderJ()
{
switch (FElem->Space())
+3 -11
View File
@@ -25,7 +25,6 @@ class ElementTransformation
protected:
const IntegrationPoint *IntPoint;
DenseMatrix dFdx, adjJ, invJ;
DenseMatrix d2Fdx2;
double Wght;
int EvalState;
enum StateMasks
@@ -33,8 +32,7 @@ protected:
JACOBIAN_MASK = 1,
WEIGHT_MASK = 2,
ADJUGATE_MASK = 4,
INVERSE_MASK = 8,
HESSIAN_MASK = 16
INVERSE_MASK = 8
};
Geometry::Type geom;
int space_dim;
@@ -42,7 +40,6 @@ protected:
// Evaluate the Jacobian of the transformation at the IntPoint and store it
// in dFdx.
virtual const DenseMatrix &EvalJacobian() = 0;
virtual const DenseMatrix &EvalHessian() = 0;
double EvalWeight();
const DenseMatrix &EvalAdjugateJ();
@@ -71,9 +68,6 @@ public:
const DenseMatrix &Jacobian()
{ return (EvalState & JACOBIAN_MASK) ? dFdx : EvalJacobian(); }
const DenseMatrix &Hessian()
{ return (EvalState & HESSIAN_MASK) ? d2Fdx2 : EvalHessian(); }
double Weight() { return (EvalState & WEIGHT_MASK) ? Wght : EvalWeight(); }
const DenseMatrix &AdjugateJacobian()
@@ -291,7 +285,7 @@ public:
class IsoparametricTransformation : public ElementTransformation
{
private:
DenseMatrix dshape,d2shape;
DenseMatrix dshape;
Vector shape;
const FiniteElement *FElem;
@@ -300,9 +294,7 @@ private:
// Evaluate the Jacobian of the transformation at the IntPoint and store it
// in dFdx.
virtual const DenseMatrix &EvalJacobian();
// Evaluate the Hessian of the transformation at the IntPoint and store it
// in d2Fdx2.
virtual const DenseMatrix &EvalHessian();
public:
void SetFE(const FiniteElement *FE) { FElem = FE; geom = FE->GetGeomType(); }
const FiniteElement* GetFE() const { return FElem; }
+3 -574
View File
@@ -203,170 +203,6 @@ void FiniteElement::CalcPhysDShape(ElementTransformation &Trans,
Mult(vshape, Trans.InverseJacobian(), dshape);
}
void FiniteElement::CalcPhysLaplacian(ElementTransformation &Trans,
Vector &Laplacian) const
{
MFEM_ASSERT(MapType == VALUE, "");
// Simpler routine if mapping is affine
if (Trans.Hessian().FNorm2() < 1e-20)
{
CalcPhysLinLaplacian(Trans, Laplacian);
return;
}
// Compute full Hessian first if non-affine
int size = (Dim*(Dim+1))/2;
DenseMatrix hess(Dof, size);
CalcPhysHessian(Trans,hess);
if (Dim == 3)
{
for (int nd = 0; nd < Dof; nd++)
{
Laplacian[nd] = hess(nd,0) + hess(nd,4) + hess(nd,5);
}
}
else if (Dim == 2)
{
for (int nd = 0; nd < Dof; nd++)
{
Laplacian[nd] = hess(nd,0) + hess(nd,2);
}
}
else
{
for (int nd = 0; nd < Dof; nd++)
{
Laplacian[nd] = hess(nd,0);
}
}
}
// Assume a linear mapping
void FiniteElement::CalcPhysLinLaplacian(ElementTransformation &Trans,
Vector &Laplacian) const
{
MFEM_ASSERT(MapType == VALUE, "");
int size = (Dim*(Dim+1))/2;
DenseMatrix hess(Dof, size);
DenseMatrix Gij(Dim,Dim);
Vector scale(size);
CalcHessian (Trans.GetIntPoint(), hess);
MultAAt(Trans.InverseJacobian(), Gij);
if (Dim == 3)
{
scale[0] = Gij(0,0);
scale[1] = 2*Gij(0,1);
scale[2] = 2*Gij(0,2);
scale[3] = 2*Gij(1,2);
scale[4] = Gij(2,2);
scale[5] = Gij(1,1);
}
else if (Dim == 2)
{
scale[0] = Gij(0,0);
scale[1] = 2*Gij(0,1);
scale[2] = Gij(1,1);
}
else
{
scale[0] = Gij(0,0);
}
for (int nd = 0; nd < Dof; nd++)
{
Laplacian[nd] = 0.0;
for (int ii = 0; ii < size; ii++)
{
Laplacian[nd] += hess(nd,ii)*scale[ii];
}
}
}
void FiniteElement::CalcPhysHessian(ElementTransformation &Trans,
DenseMatrix& Hessian) const
{
MFEM_ASSERT(MapType == VALUE, "");
// Roll 2-Tensors in vectors and 4-Tensor in Matrix, exploiting symmetry
Array<int> map(Dim*Dim);
if (Dim == 3)
{
map[0] = 0;
map[1] = 1;
map[2] = 2;
map[3] = 1;
map[4] = 5;
map[5] = 3;
map[6] = 2;
map[7] = 3;
map[8] = 4;
}
else if (Dim == 2)
{
map[0] = 0;
map[1] = 1;
map[2] = 1;
map[3] = 2;
}
else
{
map[0] = 0;
}
// Hessian in ref coords
int size = (Dim*(Dim+1))/2;
DenseMatrix hess(Dof, size);
CalcHessian(Trans.GetIntPoint(), hess);
// Gradient in physical coords
if (Trans.Hessian().FNorm2() > 1e-10)
{
DenseMatrix grad(Dof, Dim);
CalcPhysDShape(Trans, grad);
DenseMatrix gmap(Dof, size);
Mult(grad,Trans.Hessian(),gmap);
hess -= gmap;
}
// LHM
DenseMatrix lhm(size,size);
DenseMatrix invJ = Trans.Jacobian();
lhm = 0.0;
for (int i = 0; i < Dim; i++)
{
for (int j = 0; j < Dim; j++)
{
for (int k = 0; k < Dim; k++)
{
for (int l = 0; l < Dim; l++)
{
lhm(map[i*Dim+j],map[k*Dim+l]) += invJ(i,k)*invJ(j,l);
}
}
}
}
// Correct multiplicity
Vector mult(size);
mult = 0.0;
for (int i = 0; i < Dim*Dim; i++) { mult[map[i]]++; }
lhm.InvRightScaling(mult);
// Hessian in physical coords
lhm.Invert();
Mult( hess, lhm, Hessian);
}
const DofToQuad &FiniteElement::GetDofToQuad(const IntegrationRule &,
DofToQuad::Mode) const
{
@@ -1914,233 +1750,6 @@ void BiQuad2DFiniteElement::ProjectDelta(int vertex, Vector &dofs) const
#endif
}
H1Ser_QuadrilateralElement::H1Ser_QuadrilateralElement(const int p)
: ScalarFiniteElement(2, Geometry::SQUARE, (p*p + 3*p +6) / 2, p,
FunctionSpace::Qk)
{
// Store the dof_map of the associated TensorBasisElement, which will be used
// to create the serendipity dof map. Its size is larger than the size of
// the serendipity element.
TensorBasisElement tbeTemp =
TensorBasisElement(2, p, BasisType::GaussLobatto,
TensorBasisElement::DofMapType::Sr_DOF_MAP);
const Array<int> tp_dof_map = tbeTemp.GetDofMap();
const double *cp = poly1d.ClosedPoints(p, BasisType::GaussLobatto);
// Fixing the Nodes is exactly the same as the H1_QuadrilateralElement
// constructor except we only use those values of the associated tensor
// product dof_map that are <= the number of serendipity Dofs e.g. only DoFs
// 0-7 out of the 9 tensor product dofs (at quadratic order)
int o = 0;
for (int j = 0; j <= p; j++)
{
for (int i = 0; i <= p; i++)
{
if (tp_dof_map[o] < Nodes.Size())
{
Nodes.IntPoint(tp_dof_map[o]).x = cp[i];
Nodes.IntPoint(tp_dof_map[o]).y = cp[j];
}
o++;
}
}
}
void H1Ser_QuadrilateralElement::CalcShape(const IntegrationPoint &ip,
Vector &shape) const
{
int p = (this)->GetOrder();
double x = ip.x, y = ip.y;
Poly_1D::Basis edgeNodalBasis(poly1d.GetBasis(p, BasisType::GaussLobatto));
Vector nodalX(p+1);
Vector nodalY(p+1);
edgeNodalBasis.Eval(x, nodalX);
edgeNodalBasis.Eval(y, nodalY);
// First, fix edge-based shape functions. Use a nodal interpolant for edge
// points, weighted by the linear function that vanishes on opposite edge.
for (int i = 0; i < p-1; i++)
{
shape(4 + 0*(p-1) + i) = (nodalX(i+1))*(1.-y); // south edge 0->1
shape(4 + 1*(p-1) + i) = (nodalY(i+1))*x; // east edge 1->2
shape(4 + 3*(p-1) - i - 1) = (nodalX(i+1)) * y; // north edge 3->2
shape(4 + 4*(p-1) - i - 1) = (nodalY(i+1)) * (1. - x); // west edge 0->3
}
BiLinear2DFiniteElement bilinear = BiLinear2DFiniteElement();
Vector bilinearsAtIP(4);
bilinear.CalcShape(ip, bilinearsAtIP);
const double *edgePts(poly1d.ClosedPoints(p, BasisType::GaussLobatto));
// Next, set the shape function associated with vertex V, evaluated at (x,y)
// to be: bilinear function associated to V, evaluated at (x,y) - sum (shape
// function at edge point P, weighted by bilinear function for V evaluated at
// P) where the sum is taken only for points P on edges incident to V.
double vtx0fix =0;
double vtx1fix =0;
double vtx2fix =0;
double vtx3fix =0;
for (int i = 0; i<p-1; i++)
{
vtx0fix += (1-edgePts[i+1])*(shape(4 + i) +
shape(4 + 4*(p-1) - i - 1)); // bot+left edge
vtx1fix += (1-edgePts[i+1])*(shape(4 + 1*(p-1) + i) +
shape(4 + (p-2)-i)); // right+bot edge
vtx2fix += (1-edgePts[i+1])*(shape(4 + 2*(p-1) + i) +
shape(1 + 2*p-i)); // top+right edge
vtx3fix += (1-edgePts[i+1])*(shape(4 + 3*(p-1) + i) +
shape(3*p - i)); // left+top edge
}
shape(0) = bilinearsAtIP(0) - vtx0fix;
shape(1) = bilinearsAtIP(1) - vtx1fix;
shape(2) = bilinearsAtIP(2) - vtx2fix;
shape(3) = bilinearsAtIP(3) - vtx3fix;
// Interior basis functions appear starting at order p=4. These are non-nodal
// bubble functions.
if (p > 3)
{
double *legX = new double[p-1];
double *legY = new double[p-1];
Poly_1D *storeLegendre = new Poly_1D();
storeLegendre->CalcLegendre(p-2, x, legX);
storeLegendre->CalcLegendre(p-2, y, legY);
int interior_total = 0;
for (int j = 4; j < p + 1; j++)
{
for (int k = 0; k < j-3; k++)
{
shape(4 + 4*(p-1) + interior_total)
= legX[k] * legY[j-4-k] * x * (1. - x) * y * (1. - y);
interior_total++;
}
}
delete[] legX;
delete[] legY;
delete storeLegendre;
}
}
void H1Ser_QuadrilateralElement::CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const
{
int p = (this)->GetOrder();
double x = ip.x, y = ip.y;
Poly_1D::Basis edgeNodalBasis(poly1d.GetBasis(p, BasisType::GaussLobatto));
Vector nodalX(p+1);
Vector DnodalX(p+1);
Vector nodalY(p+1);
Vector DnodalY(p+1);
edgeNodalBasis.Eval(x, nodalX, DnodalX);
edgeNodalBasis.Eval(y, nodalY, DnodalY);
for (int i = 0; i < p-1; i++)
{
dshape(4 + 0*(p-1) + i,0) = DnodalX(i+1) * (1.-y);
dshape(4 + 0*(p-1) + i,1) = -nodalX(i+1);
dshape(4 + 1*(p-1) + i,0) = nodalY(i+1);
dshape(4 + 1*(p-1) + i,1) = DnodalY(i+1)*x;
dshape(4 + 3*(p-1) - i - 1,0) = DnodalX(i+1)*y;
dshape(4 + 3*(p-1) - i - 1,1) = nodalX(i+1);
dshape(4 + 4*(p-1) - i - 1,0) = -nodalY(i+1);
dshape(4 + 4*(p-1) - i - 1,1) = DnodalY(i+1) * (1.-x);
}
BiLinear2DFiniteElement bilinear = BiLinear2DFiniteElement();
DenseMatrix DbilinearsAtIP(4);
bilinear.CalcDShape(ip, DbilinearsAtIP);
const double *edgePts(poly1d.ClosedPoints(p, BasisType::GaussLobatto));
dshape(0,0) = DbilinearsAtIP(0,0);
dshape(0,1) = DbilinearsAtIP(0,1);
dshape(1,0) = DbilinearsAtIP(1,0);
dshape(1,1) = DbilinearsAtIP(1,1);
dshape(2,0) = DbilinearsAtIP(2,0);
dshape(2,1) = DbilinearsAtIP(2,1);
dshape(3,0) = DbilinearsAtIP(3,0);
dshape(3,1) = DbilinearsAtIP(3,1);
for (int i = 0; i<p-1; i++)
{
dshape(0,0) -= (1-edgePts[i+1])*(dshape(4 + 0*(p-1) + i, 0) +
dshape(4 + 4*(p-1) - i - 1,0));
dshape(0,1) -= (1-edgePts[i+1])*(dshape(4 + 0*(p-1) + i, 1) +
dshape(4 + 4*(p-1) - i - 1,1));
dshape(1,0) -= (1-edgePts[i+1])*(dshape(4 + 1*(p-1) + i, 0) +
dshape(4 + (p-2)-i, 0));
dshape(1,1) -= (1-edgePts[i+1])*(dshape(4 + 1*(p-1) + i, 1) +
dshape(4 + (p-2)-i, 1));
dshape(2,0) -= (1-edgePts[i+1])*(dshape(4 + 2*(p-1) + i, 0) +
dshape(1 + 2*p-i, 0));
dshape(2,1) -= (1-edgePts[i+1])*(dshape(4 + 2*(p-1) + i, 1) +
dshape(1 + 2*p-i, 1));
dshape(3,0) -= (1-edgePts[i+1])*(dshape(4 + 3*(p-1) + i, 0) +
dshape(3*p - i, 0));
dshape(3,1) -= (1-edgePts[i+1])*(dshape(4 + 3*(p-1) + i, 1) +
dshape(3*p - i, 1));
}
if (p > 3)
{
double *legX = new double[p-1];
double *legY = new double[p-1];
double *DlegX = new double[p-1];
double *DlegY = new double[p-1];
Poly_1D *storeLegendre = new Poly_1D();
storeLegendre->CalcLegendre(p-2, x, legX, DlegX);
storeLegendre->CalcLegendre(p-2, y, legY, DlegY);
int interior_total = 0;
for (int j = 4; j < p + 1; j++)
{
for (int k = 0; k < j-3; k++)
{
dshape(4 + 4*(p-1) + interior_total, 0) =
legY[j-4-k]*y*(1-y) * (DlegX[k]*x*(1-x) + legX[k]*(1-2*x));
dshape(4 + 4*(p-1) + interior_total, 1) =
legX[k]*x*(1-x) * (DlegY[j-4-k]*y*(1-y) + legY[j-4-k]*(1-2*y));
interior_total++;
}
}
delete[] legX;
delete[] legY;
delete[] DlegX;
delete[] DlegY;
delete storeLegendre;
}
}
void H1Ser_QuadrilateralElement::GetLocalInterpolation(ElementTransformation
&Trans,
DenseMatrix &I) const
{
// For p<=4, the basis is nodal; for p>4, the quad-interior functions are
// non-nodal.
if (Order <= 4)
{
NodalLocalInterpolation(Trans, I, *this);
}
else
{
ScalarLocalInterpolation(Trans, I, *this);
}
}
BiQuadPos2DFiniteElement::BiQuadPos2DFiniteElement()
: PositiveFiniteElement(2, Geometry::SQUARE, 9, 2, FunctionSpace::Qk)
{
@@ -7356,7 +6965,7 @@ TensorBasisElement::TensorBasisElement(const int dims, const int p,
: b_type(btype),
basis1d(poly1d.GetBasis(p, b_type))
{
if (dmtype == H1_DOF_MAP || dmtype == Sr_DOF_MAP)
if (dmtype == H1_DOF_MAP)
{
switch (dims)
{
@@ -12183,30 +11792,6 @@ void NURBS1DFiniteElement::CalcDShape(const IntegrationPoint &ip,
add(sum, grad, -dsum*sum*sum, shape_x, grad);
}
void NURBS1DFiniteElement::CalcHessian (const IntegrationPoint &ip,
DenseMatrix &hessian) const
{
Vector grad(Dof);
Vector hess(hessian.Data(), Dof);
kv[0]->CalcShape (shape_x, ijk[0], ip.x);
kv[0]->CalcDShape(grad, ijk[0], ip.x);
kv[0]->CalcD2Shape(hess, ijk[0], ip.x);
double sum = 0.0, dsum = 0.0, d2sum = 0.0;
for (int i = 0; i <= Order; i++)
{
sum += (shape_x(i) *= weights(i));
dsum += ( grad(i) *= weights(i));
d2sum += ( hess(i) *= weights(i));
}
sum = 1.0/sum;
add(sum, hess, -2*dsum*sum*sum, grad, hess);
add(1.0, hess, (-d2sum + 2*dsum*dsum*sum)*sum*sum, shape_x, hess);
}
void NURBS2DFiniteElement::SetOrder() const
{
Orders[0] = kv[0]->GetOrder();
@@ -12215,13 +11800,10 @@ void NURBS2DFiniteElement::SetOrder() const
shape_y.SetSize(Orders[1]+1);
dshape_x.SetSize(Orders[0]+1);
dshape_y.SetSize(Orders[1]+1);
d2shape_x.SetSize(Orders[0]+1);
d2shape_y.SetSize(Orders[1]+1);
Order = max(Orders[0], Orders[1]);
Dof = (Orders[0] + 1)*(Orders[1] + 1);
u.SetSize(Dof);
du.SetSize(Dof);
weights.SetSize(Dof);
}
@@ -12279,65 +11861,7 @@ void NURBS2DFiniteElement::CalcDShape(const IntegrationPoint &ip,
}
}
void NURBS2DFiniteElement::CalcHessian (const IntegrationPoint &ip,
DenseMatrix &hessian) const
{
double sum, dsum[2], d2sum[3];
kv[0]->CalcShape ( shape_x, ijk[0], ip.x);
kv[1]->CalcShape ( shape_y, ijk[1], ip.y);
kv[0]->CalcDShape(dshape_x, ijk[0], ip.x);
kv[1]->CalcDShape(dshape_y, ijk[1], ip.y);
kv[0]->CalcD2Shape(d2shape_x, ijk[0], ip.x);
kv[1]->CalcD2Shape(d2shape_y, ijk[1], ip.y);
sum = dsum[0] = dsum[1] = 0.0;
d2sum[0] = d2sum[1] = d2sum[2] = 0.0;
for (int o = 0, j = 0; j <= Orders[1]; j++)
{
const double sy = shape_y(j), dsy = dshape_y(j), d2sy = d2shape_y(j);
for (int i = 0; i <= Orders[0]; i++, o++)
{
const double sx = shape_x(i), dsx = dshape_x(i), d2sx = d2shape_x(i);
sum += ( u(o) = sx*sy*weights(o) );
dsum[0] += ( du(o,0) = dsx*sy*weights(o) );
dsum[1] += ( du(o,1) = sx*dsy*weights(o) );
d2sum[0] += ( hessian(o,0) = d2sx*sy*weights(o) );
d2sum[1] += ( hessian(o,1) = dsx*dsy*weights(o) );
d2sum[2] += ( hessian(o,2) = sx*d2sy*weights(o) );
}
}
sum = 1.0/sum;
dsum[0] *= sum;
dsum[1] *= sum;
d2sum[0] *= sum;
d2sum[1] *= sum;
d2sum[2] *= sum;
for (int o = 0; o < Dof; o++)
{
hessian(o,0) = hessian(o,0)*sum
- 2*du(o,0)*sum*dsum[0]
+ u[o]*sum*(2*dsum[0]*dsum[0] - d2sum[0]);
hessian(o,1) = hessian(o,1)*sum
- du(o,0)*sum*dsum[1]
- du(o,1)*sum*dsum[0]
+ u[o]*sum*(2*dsum[0]*dsum[1] - d2sum[1]);
hessian(o,2) = hessian(o,2)*sum
- 2*du(o,1)*sum*dsum[1]
+ u[o]*sum*(2*dsum[1]*dsum[1] - d2sum[2]);
}
}
//---------------------------------------------------------------------
void NURBS3DFiniteElement::SetOrder() const
{
Orders[0] = kv[0]->GetOrder();
@@ -12351,14 +11875,9 @@ void NURBS3DFiniteElement::SetOrder() const
dshape_y.SetSize(Orders[1]+1);
dshape_z.SetSize(Orders[2]+1);
d2shape_x.SetSize(Orders[0]+1);
d2shape_y.SetSize(Orders[1]+1);
d2shape_z.SetSize(Orders[2]+1);
Order = max(max(Orders[0], Orders[1]), Orders[2]);
Dof = (Orders[0] + 1)*(Orders[1] + 1)*(Orders[2] + 1);
u.SetSize(Dof);
du.SetSize(Dof);
weights.SetSize(Dof);
}
@@ -12432,100 +11951,10 @@ void NURBS3DFiniteElement::CalcDShape(const IntegrationPoint &ip,
}
}
void NURBS3DFiniteElement::CalcHessian (const IntegrationPoint &ip,
DenseMatrix &hessian) const
{
double sum, dsum[3], d2sum[6];
kv[0]->CalcShape ( shape_x, ijk[0], ip.x);
kv[1]->CalcShape ( shape_y, ijk[1], ip.y);
kv[2]->CalcShape ( shape_z, ijk[2], ip.z);
kv[0]->CalcDShape(dshape_x, ijk[0], ip.x);
kv[1]->CalcDShape(dshape_y, ijk[1], ip.y);
kv[2]->CalcDShape(dshape_z, ijk[2], ip.z);
kv[0]->CalcD2Shape(d2shape_x, ijk[0], ip.x);
kv[1]->CalcD2Shape(d2shape_y, ijk[1], ip.y);
kv[2]->CalcD2Shape(d2shape_z, ijk[2], ip.z);
sum = dsum[0] = dsum[1] = dsum[2] = 0.0;
d2sum[0] = d2sum[1] = d2sum[2] = d2sum[3] = d2sum[4] = d2sum[5] = 0.0;
for (int o = 0, k = 0; k <= Orders[2]; k++)
{
const double sz = shape_z(k), dsz = dshape_z(k), d2sz = d2shape_z(k);
for (int j = 0; j <= Orders[1]; j++)
{
const double sy = shape_y(j), dsy = dshape_y(j), d2sy = d2shape_y(j);
for (int i = 0; i <= Orders[0]; i++, o++)
{
const double sx = shape_x(i), dsx = dshape_x(i), d2sx = d2shape_x(i);
sum += ( u(o) = sx*sy*sz*weights(o) );
dsum[0] += ( du(o,0) = dsx*sy*sz*weights(o) );
dsum[1] += ( du(o,1) = sx*dsy*sz*weights(o) );
dsum[2] += ( du(o,2) = sx*sy*dsz*weights(o) );
d2sum[0] += ( hessian(o,0) = d2sx*sy*sz*weights(o) );
d2sum[1] += ( hessian(o,1) = dsx*dsy*sz*weights(o) );
d2sum[2] += ( hessian(o,2) = dsx*sy*dsz*weights(o) );
d2sum[3] += ( hessian(o,3) = sx*dsy*dsz*weights(o) );
d2sum[4] += ( hessian(o,4) = sx*sy*d2sz*weights(o) );
d2sum[5] += ( hessian(o,5) = sx*d2sy*sz*weights(o) );
}
}
}
sum = 1.0/sum;
dsum[0] *= sum;
dsum[1] *= sum;
dsum[2] *= sum;
d2sum[0] *= sum;
d2sum[1] *= sum;
d2sum[2] *= sum;
d2sum[3] *= sum;
d2sum[4] *= sum;
d2sum[5] *= sum;
for (int o = 0; o < Dof; o++)
{
hessian(o,0) = hessian(o,0)*sum
- 2*du(o,0)*sum*dsum[0]
+ u[o]*sum*(2*dsum[0]*dsum[0] - d2sum[0]);
hessian(o,1) = hessian(o,1)*sum
- du(o,0)*sum*dsum[1]
- du(o,1)*sum*dsum[0]
+ u[o]*sum*(2*dsum[0]*dsum[1] - d2sum[1]);
hessian(o,2) = hessian(o,2)*sum
- du(o,0)*sum*dsum[2]
- du(o,2)*sum*dsum[0]
+ u[o]*sum*(2*dsum[0]*dsum[2] - d2sum[2]);
hessian(o,3) = hessian(o,3)*sum
- du(o,1)*sum*dsum[2]
- du(o,2)*sum*dsum[1]
+ u[o]*sum*(2*dsum[1]*dsum[2] - d2sum[3]);
hessian(o,4) = hessian(o,4)*sum
- 2*du(o,2)*sum*dsum[2]
+ u[o]*sum*(2*dsum[2]*dsum[2] - d2sum[4]);
hessian(o,5) = hessian(o,5)*sum
- 2*du(o,1)*sum*dsum[1]
+ u[o]*sum*(2*dsum[1]*dsum[1] - d2sum[5]);
}
}
// Global object definitions
// Object declared in mesh/triangle.hpp.
// Defined here to ensure it is constructed before 'Geometries'.
Linear2DFiniteElement TriangleFE;
+14 -63
View File
@@ -36,8 +36,7 @@ public:
OpenUniform = 3, ///< Nodes: x_i = (i+1)/(n+1), i=0,...,n-1
ClosedUniform = 4, ///< Nodes: x_i = i/(n-1), i=0,...,n-1
OpenHalfUniform = 5, ///< Nodes: x_i = (i+1/2)/n, i=0,...,n-1
Serendipity = 6, ///< Serendipity basis (squares / cubes)
NumBasisTypes = 7 /**< Keep track of maximum types to prevent
NumBasisTypes = 6 /**< Keep track of maximum types to prevent
hard-coding */
};
/** @brief If the input does not represents a valid BasisType, abort with an
@@ -68,7 +67,6 @@ public:
case OpenUniform: return Quadrature1D::OpenUniform;
case ClosedUniform: return Quadrature1D::ClosedUniform;
case OpenHalfUniform: return Quadrature1D::OpenHalfUniform;
case Serendipity: return Quadrature1D::GaussLobatto;
}
return Quadrature1D::Invalid;
}
@@ -112,7 +110,6 @@ public:
case 'u': return OpenUniform;
case 'U': return ClosedUniform;
case 'o': return OpenHalfUniform;
case 's': return GaussLobatto;
}
MFEM_ABORT("unknown BasisType identifier");
return -1;
@@ -422,29 +419,10 @@ public:
virtual void GetFaceDofs(int face, int **dofs, int *ndofs) const;
/** @brief Evaluate the Hessians of all shape functions of a scalar finite
element in reference space at the given point @a ip. */
/** Each row of the result DenseMatrix @a Hessian contains upper triangular
part of the Hessian of one shape function.
The order in 2D is {u_xx, u_xy, u_yy}.
The size (#Dof x (#Dim (#Dim-1)/2) of @a Hessian must be set in advance.*/
/** each row of h contains the upper triangular part of the hessian
of one shape function; the order in 2D is {u_xx, u_xy, u_yy} */
virtual void CalcHessian (const IntegrationPoint &ip,
DenseMatrix &Hessian) const;
/** @brief Evaluate the Hessian of all shape functions of a scalar finite
element in reference space at the given point @a ip. */
/** The size (#Dof, #Dim*(#Dim+1)/2) of @a Hessian must be set in advance. */
virtual void CalcPhysHessian(ElementTransformation &Trans,
DenseMatrix& Hessian) const;
/** @brief Evaluate the Laplacian of all shape functions of a scalar finite
element in reference space at the given point @a ip. */
/** The size (#Dof) of @a Laplacian must be set in advance. */
virtual void CalcPhysLaplacian(ElementTransformation &Trans,
Vector& Laplacian) const;
virtual void CalcPhysLinLaplacian(ElementTransformation &Trans,
Vector& Laplacian) const;
DenseMatrix &h) const;
/** @brief Return the local interpolation matrix @a I (Dof x Dof) where the
fine element is the image of the base geometry under the given
@@ -1723,6 +1701,9 @@ private:
static void CalcMono(const int p, const double x, double *u);
static void CalcMono(const int p, const double x, double *u, double *d);
static void CalcLegendre(const int p, const double x, double *u);
static void CalcLegendre(const int p, const double x, double *u, double *d);
static void CalcChebyshev(const int p, const double x, double *u);
static void CalcChebyshev(const int p, const double x, double *u, double *d);
static void CalcChebyshev(const int p, const double x, double *u, double *d,
@@ -1811,9 +1792,6 @@ public:
static void CalcBernstein(const int p, const double x, double *u, double *d)
{ CalcBinomTerms(p, x, 1. - x, u, d); }
static void CalcLegendre(const int p, const double x, double *u);
static void CalcLegendre(const int p, const double x, double *u, double *d);
~Poly_1D();
};
@@ -1825,14 +1803,12 @@ protected:
int b_type;
Array<int> dof_map;
Poly_1D::Basis &basis1d;
Array<int> inv_dof_map;
public:
enum DofMapType
{
L2_DOF_MAP = 0,
H1_DOF_MAP = 1,
Sr_DOF_MAP = 2, // Sr = Serendipity
H1_DOF_MAP = 1
};
TensorBasisElement(const int dims, const int p, const int btype,
@@ -1992,18 +1968,6 @@ public:
};
class H1Ser_QuadrilateralElement : public ScalarFiniteElement
{
public:
H1Ser_QuadrilateralElement(const int p);
virtual void CalcShape(const IntegrationPoint &ip, Vector &shape) const;
virtual void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const;
virtual void GetLocalInterpolation(ElementTransformation &Trans,
DenseMatrix &I) const;
using FiniteElement::Project;
};
class H1Pos_HexahedronElement : public PositiveTensorFiniteElement
{
private:
@@ -2924,70 +2888,57 @@ public:
virtual void CalcShape(const IntegrationPoint &ip, Vector &shape) const;
virtual void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const;
virtual void CalcHessian (const IntegrationPoint &ip,
DenseMatrix &hessian) const;
};
class NURBS2DFiniteElement : public NURBSFiniteElement
{
protected:
mutable Vector u, shape_x, shape_y, dshape_x, dshape_y, d2shape_x, d2shape_y;
mutable DenseMatrix du;
mutable Vector u, shape_x, shape_y, dshape_x, dshape_y;
public:
NURBS2DFiniteElement(int p)
: NURBSFiniteElement(2, Geometry::SQUARE, (p + 1)*(p + 1), p,
FunctionSpace::Qk),
u(Dof), shape_x(p + 1), shape_y(p + 1), dshape_x(p + 1),
dshape_y(p + 1), d2shape_x(p + 1), d2shape_y(p + 1), du(Dof,2)
u(Dof), shape_x(p + 1), shape_y(p + 1), dshape_x(p + 1), dshape_y(p + 1)
{ Orders[0] = Orders[1] = p; }
NURBS2DFiniteElement(int px, int py)
: NURBSFiniteElement(2, Geometry::SQUARE, (px + 1)*(py + 1),
std::max(px, py), FunctionSpace::Qk),
u(Dof), shape_x(px + 1), shape_y(py + 1), dshape_x(px + 1),
dshape_y(py + 1), d2shape_x(px + 1), d2shape_y(py + 1), du(Dof,2)
dshape_y(py + 1)
{ Orders[0] = px; Orders[1] = py; }
virtual void SetOrder() const;
virtual void CalcShape(const IntegrationPoint &ip, Vector &shape) const;
virtual void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const;
virtual void CalcHessian (const IntegrationPoint &ip,
DenseMatrix &hessian) const;
};
class NURBS3DFiniteElement : public NURBSFiniteElement
{
protected:
mutable Vector u, shape_x, shape_y, shape_z;
mutable Vector dshape_x, dshape_y, dshape_z;
mutable Vector d2shape_x, d2shape_y, d2shape_z;
mutable DenseMatrix du;
mutable Vector u, shape_x, shape_y, shape_z, dshape_x, dshape_y, dshape_z;
public:
NURBS3DFiniteElement(int p)
: NURBSFiniteElement(3, Geometry::CUBE, (p + 1)*(p + 1)*(p + 1), p,
FunctionSpace::Qk),
u(Dof), shape_x(p + 1), shape_y(p + 1), shape_z(p + 1),
dshape_x(p + 1), dshape_y(p + 1), dshape_z(p + 1),
d2shape_x(p + 1), d2shape_y(p + 1), d2shape_z(p + 1), du(Dof,3)
dshape_x(p + 1), dshape_y(p + 1), dshape_z(p + 1)
{ Orders[0] = Orders[1] = Orders[2] = p; }
NURBS3DFiniteElement(int px, int py, int pz)
: NURBSFiniteElement(3, Geometry::CUBE, (px + 1)*(py + 1)*(pz + 1),
std::max(std::max(px,py),pz), FunctionSpace::Qk),
u(Dof), shape_x(px + 1), shape_y(py + 1), shape_z(pz + 1),
dshape_x(px + 1), dshape_y(py + 1), dshape_z(pz + 1),
d2shape_x(px + 1), d2shape_y(py + 1), d2shape_z(pz + 1), du(Dof,3)
dshape_x(px + 1), dshape_y(py + 1), dshape_z(pz + 1)
{ Orders[0] = px; Orders[1] = py; Orders[2] = pz; }
virtual void SetOrder() const;
virtual void CalcShape(const IntegrationPoint &ip, Vector &shape) const;
virtual void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const;
virtual void CalcHessian (const IntegrationPoint &ip,
DenseMatrix &hessian) const;
};
} // namespace mfem
+12 -73
View File
@@ -162,10 +162,6 @@ FiniteElementCollection *FiniteElementCollection::New(const char *name)
{
fec = new H1Pos_FECollection(atoi(name + 10), atoi(name + 6));
}
else if (!strncmp(name, "H1Ser_", 6))
{
fec = new H1Ser_FECollection(atoi(name + 10), atoi(name + 6));
}
else if (!strncmp(name, "H1@", 3))
{
fec = new H1_FECollection(atoi(name + 9), atoi(name + 5),
@@ -1524,11 +1520,6 @@ H1_FECollection::H1_FECollection(const int p, const int dim, const int btype)
snprintf(h1_name, 32, "H1Pos_%dD_P%d", dim, p);
break;
}
case BasisType::Serendipity:
{
snprintf(h1_name, 32, "H1Ser_%dD_P%d", dim, p);
break;
}
default:
{
MFEM_VERIFY(Quadrature1D::CheckClosed(pt_type) !=
@@ -1591,18 +1582,6 @@ H1_FECollection::H1_FECollection(const int p, const int dim, const int btype)
H1_Elements[Geometry::TRIANGLE] = new H1Pos_TriangleElement(p);
H1_Elements[Geometry::SQUARE] = new H1Pos_QuadrilateralElement(p);
}
else if (b_type == BasisType::Serendipity)
{
// Note: in fe_coll.hpp the DofForGeometry(Geometry::Type) method
// returns H1_dof[GeomType], so we need to fix the value of H1_dof here
// for the serendipity case.
// formula for number of interior serendipity DoFs (when p>1)
H1_dof[Geometry::SQUARE] = (pm3*pm2)/2;
H1_Elements[Geometry::SQUARE] = new H1Ser_QuadrilateralElement(p);
// allows for mixed tri/quad meshes
H1_Elements[Geometry::TRIANGLE] = new H1Pos_TriangleElement(p);
}
else
{
H1_Elements[Geometry::TRIANGLE] = new H1_TriangleElement(p, btype);
@@ -1637,60 +1616,20 @@ H1_FECollection::H1_FECollection(const int p, const int dim, const int btype)
{
QuadDofOrd[i] = QuadDofOrd[i-1] + QuadDof;
}
// For serendipity order >=4, the QuadDofOrd array must be re-defined. We
// do this by computing the corresponding tensor product QuadDofOrd array
// or two orders less, which contains enough DoFs for their serendipity
// basis. This could be optimized.
if (b_type == BasisType::Serendipity)
// see Mesh::GetQuadOrientation in mesh/mesh.cpp
for (int j = 0; j < pm1; j++)
{
if (p < 4)
for (int i = 0; i < pm1; i++)
{
// no face dofs --> don't need to adjust QuadDofOrd
}
else // p >= 4 --> have face dofs
{
// Exactly the same as tensor product case, but with all orders
// reduced by 2 e.g. in case p=5 it builds a 2x2 array, even though
// there are only 3 serendipity dofs.
// In the tensor product case, the i and j index tensor directions,
// and o index from 0 to (pm1)^2,
const int pm4 = pm3 -1;
for (int j = 0; j < pm3; j++) // pm3 instead of pm1, etc
{
for (int i = 0; i < pm3; i++)
{
int o = i + j*pm3;
QuadDofOrd[0][o] = i + j*pm3; // (0,1,2,3)
QuadDofOrd[1][o] = j + i*pm3; // (0,3,2,1)
QuadDofOrd[2][o] = j + (pm4 - i)*pm3; // (1,2,3,0)
QuadDofOrd[3][o] = (pm4 - i) + j*pm3; // (1,0,3,2)
QuadDofOrd[4][o] = (pm4 - i) + (pm4 - j)*pm3; // (2,3,0,1)
QuadDofOrd[5][o] = (pm4 - j) + (pm4 - i)*pm3; // (2,1,0,3)
QuadDofOrd[6][o] = (pm4 - j) + i*pm3; // (3,0,1,2)
QuadDofOrd[7][o] = i + (pm4 - j)*pm3; // (3,2,1,0)
}
}
}
}
else // not serendipity
{
for (int j = 0; j < pm1; j++)
{
for (int i = 0; i < pm1; i++)
{
int o = i + j*pm1;
QuadDofOrd[0][o] = i + j*pm1; // (0,1,2,3)
QuadDofOrd[1][o] = j + i*pm1; // (0,3,2,1)
QuadDofOrd[2][o] = j + (pm2 - i)*pm1; // (1,2,3,0)
QuadDofOrd[3][o] = (pm2 - i) + j*pm1; // (1,0,3,2)
QuadDofOrd[4][o] = (pm2 - i) + (pm2 - j)*pm1; // (2,3,0,1)
QuadDofOrd[5][o] = (pm2 - j) + (pm2 - i)*pm1; // (2,1,0,3)
QuadDofOrd[6][o] = (pm2 - j) + i*pm1; // (3,0,1,2)
QuadDofOrd[7][o] = i + (pm2 - j)*pm1; // (3,2,1,0)
}
int o = i + j*pm1;
QuadDofOrd[0][o] = i + j*pm1; // (0,1,2,3)
QuadDofOrd[1][o] = j + i*pm1; // (0,3,2,1)
QuadDofOrd[2][o] = j + (pm2 - i)*pm1; // (1,2,3,0)
QuadDofOrd[3][o] = (pm2 - i) + j*pm1; // (1,0,3,2)
QuadDofOrd[4][o] = (pm2 - i) + (pm2 - j)*pm1; // (2,3,0,1)
QuadDofOrd[5][o] = (pm2 - j) + (pm2 - i)*pm1; // (2,1,0,3)
QuadDofOrd[6][o] = (pm2 - j) + i*pm1; // (3,0,1,2)
QuadDofOrd[7][o] = i + (pm2 - j)*pm1; // (3,2,1,0)
}
}
-9
View File
@@ -120,15 +120,6 @@ public:
: H1_FECollection(p, dim, BasisType::Positive) { }
};
/** Arbitrary order H1-conforming (continuous) serendipity finite elements;
Current implementation works in 2D only; 3D version is in development. */
class H1Ser_FECollection : public H1_FECollection
{
public:
explicit H1Ser_FECollection(const int p, const int dim = 2)
: H1_FECollection(p, dim, BasisType::Serendipity) { };
};
/** Arbitrary order "H^{1/2}-conforming" trace finite elements defined on the
interface between mesh elements (faces,edges,vertices); these are the trace
FEs of the H1-conforming FEs. */
-2
View File
@@ -18,7 +18,6 @@
#include "fe_coll.hpp"
#include "eltrans.hpp"
#include "coefficient.hpp"
#include "complex_fem.hpp"
#include "lininteg.hpp"
#include "nonlininteg.hpp"
#include "bilininteg.hpp"
@@ -33,7 +32,6 @@
#include "staticcond.hpp"
#include "tmop.hpp"
#include "tmop_tools.hpp"
#include "gslib.hpp"
#ifdef MFEM_USE_MPI
#include "pfespace.hpp"
+52 -62
View File
@@ -325,7 +325,7 @@ void FiniteElementSpace::GetEssentialVDofs(const Array<int> &bdr_attr_is_ess,
// mark possible hidden boundary edges in a non-conforming mesh, also
// local DOFs affected by boundary elements on other processors
if (Nonconforming())
if (mesh->ncmesh)
{
Array<int> bdr_verts, bdr_edges;
mesh->ncmesh->GetBoundaryClosure(bdr_attr_is_ess, bdr_verts, bdr_edges);
@@ -567,16 +567,16 @@ bool FiniteElementSpace::DofFinalizable(int dof, const Array<bool>& finalized,
return true;
}
void FiniteElementSpace::GetDegenerateFaceDofs(int index, Array<int> &dofs,
Geometry::Type master_geom) const
void FiniteElementSpace::GetDegenerateFaceDofs(int index,
Array<int> &dofs) const
{
// In NC meshes with prisms/tets, a special constraint occurs where a
// prism/tet edge is slave to another element's face. Rather than introduce a
// new edge-face constraint type, we handle such cases as degenerate
// face-face constraints, where the point-matrix rectangle has zero height.
// This method returns DOFs for the first edge of the rectangle, duplicated
// in the orthogonal direction, to resemble DOFs for a quadrilateral face.
// The extra DOFs are ignored by FiniteElementSpace::AddDependencies.
// In NC meshes with prisms, a special constraint occurs where a prism edge
// is slave to a quadrilateral face. Rather than introduce a new edge-face
// constraint type, we handle such cases as degenerate face-face constraints,
// where the point-matrix rectangle has zero height. This method returns
// DOFs for the first edge of the rectangle, duplicated in the orthogonal
// direction, to resemble DOFs for a quadrilateral face. The extra DOFs are
// ignored by FiniteElementSpace::AddDependencies.
Array<int> edof;
GetEdgeDofs(-1 - index, edof);
@@ -586,8 +586,6 @@ void FiniteElementSpace::GetDegenerateFaceDofs(int index, Array<int> &dofs,
int nn = 2*nv + ne;
dofs.SetSize(nn*nn);
if (!dofs.Size()) { return; }
dofs = edof[0];
// copy first two vertex DOFs
@@ -597,23 +595,21 @@ void FiniteElementSpace::GetDegenerateFaceDofs(int index, Array<int> &dofs,
dofs[nv+i] = edof[nv+i];
}
// copy first edge DOFs
int face_vert = Geometry::NumVerts[master_geom];
for (int i = 0; i < ne; i++)
{
dofs[face_vert*nv + i] = edof[2*nv + i];
dofs[4*nv + i] = edof[2*nv + i];
}
}
void
FiniteElementSpace::GetEntityDofs(int entity, int index, Array<int> &dofs,
Geometry::Type master_geom) const
FiniteElementSpace::GetEntityDofs(int entity, int index, Array<int> &dofs) const
{
switch (entity)
{
case 0: GetVertexDofs(index, dofs); break;
case 1: GetEdgeDofs(index, dofs); break;
case 2: (index >= 0) ? GetFaceDofs(index, dofs)
/* */ : GetDegenerateFaceDofs(index, dofs, master_geom);
/* */ : GetDegenerateFaceDofs(index, dofs);
}
}
@@ -666,7 +662,7 @@ void FiniteElementSpace::BuildConformingInterpolation() const
for (int si = master.slaves_begin; si < master.slaves_end; si++)
{
const NCMesh::Slave &slave = list.slaves[si];
GetEntityDofs(entity, slave.index, slave_dofs, master.Geom());
GetEntityDofs(entity, slave.index, slave_dofs);
if (!slave_dofs.Size()) { continue; }
slave.OrientedPointMatrix(T.GetPointMat());
@@ -958,7 +954,7 @@ void FiniteElementSpace::GetLocalRefinementMatrices(
const DenseTensor &pmats = rtrans.point_matrices[geom];
int nmat = pmats.SizeK();
int ldof = fe->GetDof();
int ldof = fe->GetDof(); // assuming the same FE everywhere
IsoparametricTransformation isotr;
isotr.SetIdentityTransformation(geom);
@@ -976,8 +972,7 @@ void FiniteElementSpace::GetLocalRefinementMatrices(
SparseMatrix* FiniteElementSpace::RefinementMatrix(int old_ndofs,
const Table* old_elem_dof)
{
MFEM_VERIFY(GetNE() >= old_elem_dof->Size(),
"Previous mesh is not coarser.");
MFEM_VERIFY(ndofs >= old_ndofs, "Previous space is not coarser.");
Mesh::GeometryList elem_geoms(*mesh);
@@ -995,8 +990,10 @@ FiniteElementSpace::RefinementOperator::RefinementOperator
: fespace(fespace)
, old_elem_dof(old_elem_dof)
{
MFEM_VERIFY(fespace->GetNE() >= old_elem_dof->Size(),
"Previous mesh is not coarser.");
const Mesh* mesh = fespace->GetMesh();
MFEM_VERIFY(mesh->ReduceInt(fespace->GetNDofs()) >=
mesh->ReduceInt(old_ndofs),
"Previous space is not coarser.");
width = old_ndofs * fespace->GetVDim();
height = fespace->GetVSize();
@@ -1258,11 +1255,16 @@ SparseMatrix* FiniteElementSpace::DerefinementMatrix(int old_ndofs,
GetLocalDerefinementMatrices(elem_geoms[i], localR[elem_geoms[i]]);
}
SparseMatrix *R = (elem_geoms.Size() != 1)
? new SparseMatrix(ndofs*vdim, old_ndofs*vdim) // variable row size
: new SparseMatrix(ndofs*vdim, old_ndofs*vdim,
localR[elem_geoms[0]].SizeI());
SparseMatrix *R;
if (elem_geoms.Size() == 1)
{
R = new SparseMatrix(ndofs*vdim, old_ndofs*vdim,
localR[elem_geoms[0]].SizeI());
}
else
{
R = new SparseMatrix(ndofs*vdim, old_ndofs*vdim);
}
Array<int> mark(R->Height());
mark = 0;
@@ -1275,7 +1277,7 @@ SparseMatrix* FiniteElementSpace::DerefinementMatrix(int old_ndofs,
for (int k = 0; k < dtrans.embeddings.Size(); k++)
{
const Embedding &emb = dtrans.embeddings[k];
Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
const Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
DenseMatrix &lR = localR[geom](emb.matrix);
elem_dof->GetRow(emb.parent, dofs);
@@ -1288,7 +1290,7 @@ SparseMatrix* FiniteElementSpace::DerefinementMatrix(int old_ndofs,
for (int i = 0; i < lR.Height(); i++)
{
if (!std::isfinite(lR(i, 0))) { continue; }
if (lR(i, 0) == infinity()) { continue; }
int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
@@ -1306,8 +1308,7 @@ SparseMatrix* FiniteElementSpace::DerefinementMatrix(int old_ndofs,
MFEM_VERIFY(num_marked == R->Height(),
"internal error: not all rows of R were set.");
R->Finalize(); // no-op if fixed width
if (elem_geoms.Size() != 1) { R->Finalize(); }
return R;
}
@@ -2690,68 +2691,57 @@ L2ElementRestriction::L2ElementRestriction(const FiniteElementSpace &fes)
void L2ElementRestriction::Mult(const Vector &x, Vector &y) const
{
const int NE = ne;
const int VDIM = vdim;
const int NDOF = ndof;
const bool BYVDIM = byvdim;
auto d_x = x.Read();
auto d_y = y.Write();
MFEM_FORALL(iel, NE,
for (int iel=0; iel<ne; ++iel)
{
for (int vd=0; vd<VDIM; ++vd)
for (int vd=0; vd<vdim; ++vd)
{
for (int idof=0; idof<NDOF; ++idof)
for (int idof=0; idof<ndof; ++idof)
{
// E-vector dimensions (dofs, vdim, elements)
// L-vector dimensions: byVDIM: (vdim, dofs, element)
// byNODES: (dofs, elements, vdim)
int yidx = iel*VDIM*NDOF + vd*NDOF + idof;
int yidx = iel*vdim*ndof + vd*ndof + idof;
int xidx;
if (BYVDIM)
if (byvdim)
{
xidx = iel*NDOF*VDIM + idof*VDIM + vd;
xidx = iel*ndof*vdim + idof*vdim + vd;
}
else
{
xidx = vd*NE*NDOF + iel*NDOF + idof;
xidx = vd*ne*ndof + iel*ndof + idof;
}
d_y[yidx] = d_x[xidx];
y[yidx] = x[xidx];
}
}
});
}
}
void L2ElementRestriction::MultTranspose(const Vector &x, Vector &y) const
{
const int NE = ne;
const int VDIM = vdim;
const int NDOF = ndof;
const bool BYVDIM = byvdim;
auto d_x = x.Read();
auto d_y = y.Write();
// Since this restriction is a permutation, the transpose is the inverse
MFEM_FORALL(iel, NE,
for (int iel=0; iel<ne; ++iel)
{
for (int vd=0; vd<VDIM; ++vd)
for (int vd=0; vd<vdim; ++vd)
{
for (int idof=0; idof<NDOF; ++idof)
for (int idof=0; idof<ndof; ++idof)
{
// E-vector dimensions (dofs, vdim, elements)
// L-vector dimensions: byVDIM: (vdim, dofs, element)
// byNODES: (dofs, elements, vdim)
int xidx = iel*VDIM*NDOF + vd*NDOF + idof;
int xidx = iel*vdim*ndof + vd*ndof + idof;
int yidx;
if (BYVDIM)
if (byvdim)
{
yidx = iel*NDOF*VDIM + idof*VDIM + vd;
yidx = iel*ndof*vdim + idof*vdim + vd;
}
else
{
yidx = vd*NE*NDOF + iel*NDOF + idof;
yidx = vd*ne*ndof + iel*ndof + idof;
}
d_y[yidx] = d_x[xidx];
y[yidx] = x[xidx];
}
}
});
}
}
ElementRestriction::ElementRestriction(const FiniteElementSpace &f,
+2 -4
View File
@@ -145,11 +145,9 @@ protected:
{ return (dof >= 0) ? (sign = 1, dof) : (sign = -1, (-1 - dof)); }
/// Helper to get vertex, edge or face DOFs (entity=0,1,2 resp.).
void GetEntityDofs(int entity, int index, Array<int> &dofs,
Geometry::Type master_geom = Geometry::INVALID) const;
void GetEntityDofs(int entity, int index, Array<int> &dofs) const;
// Get degenerate face DOFs: see explanation in method implementation.
void GetDegenerateFaceDofs(int index, Array<int> &dofs,
Geometry::Type master_geom) const;
void GetDegenerateFaceDofs(int index, Array<int> &dofs) const;
/// Calculate the cP and cR matrices for a nonconforming mesh.
void BuildConformingInterpolation() const;
-95
View File
@@ -475,101 +475,6 @@ const
GetValues(i, ir, vals, vdim);
}
void GridFunction::GetLaplacians(int i, const IntegrationRule &ir, Vector &laps,
int vdim)
const
{
Array<int> dofs;
int n = ir.GetNPoints();
laps.SetSize(n);
fes->GetElementDofs(i, dofs);
fes->DofsToVDofs(vdim-1, dofs);
const FiniteElement *FElem = fes->GetFE(i);
ElementTransformation *ET;
ET = fes->GetElementTransformation(i);
MFEM_ASSERT(FElem->GetMapType() == FiniteElement::VALUE,
"invalid FE map type");
int dof = FElem->GetDof();
Vector DofLap(dof), loc_data(dof);
GetSubVector(dofs, loc_data);
for (int k = 0; k < n; k++)
{
const IntegrationPoint &ip = ir.IntPoint(k);
ET->SetIntPoint(&ip);
FElem->CalcPhysLaplacian(*ET, DofLap);
laps(k) = DofLap * loc_data;
}
}
void GridFunction::GetLaplacians(int i, const IntegrationRule &ir, Vector &laps,
DenseMatrix &tr, int vdim)
const
{
ElementTransformation *ET;
ET = fes->GetElementTransformation(i);
ET->Transform(ir, tr);
GetLaplacians(i, ir, laps, vdim);
}
void GridFunction::GetHessians(int i, const IntegrationRule &ir,
DenseMatrix &hess,
int vdim)
const
{
Array<int> dofs;
int n = ir.GetNPoints();
fes->GetElementDofs(i, dofs);
fes->DofsToVDofs(vdim-1, dofs);
const FiniteElement *FElem = fes->GetFE(i);
ElementTransformation *ET;
ET = fes->GetElementTransformation(i);
int dim = FElem->GetDim();
int size = (dim*(dim+1))/2;
MFEM_ASSERT(FElem->GetMapType() == FiniteElement::VALUE,
"invalid FE map type");
int dof = FElem->GetDof();
DenseMatrix DofHes(dof, size);
hess.SetSize(n, size);
Vector loc_data(dof);
GetSubVector(dofs, loc_data);
hess = 0.0;
for (int k = 0; k < n; k++)
{
const IntegrationPoint &ip = ir.IntPoint(k);
ET->SetIntPoint(&ip);
FElem->CalcPhysHessian(*ET, DofHes);
for (int i = 0; i < size; i++)
{
for (int d = 0; d < dof; d++)
{
hess(k,i) += DofHes(d,i) * loc_data[d];
}
}
}
}
void GridFunction::GetHessians(int i, const IntegrationRule &ir,
DenseMatrix &hess,
DenseMatrix &tr, int vdim)
const
{
ElementTransformation *ET;
ET = fes->GetElementTransformation(i);
ET->Transform(ir, tr);
GetHessians(i, ir, hess, vdim);
}
int GridFunction::GetFaceValues(int i, int side, const IntegrationRule &ir,
Vector &vals, DenseMatrix &tr,
int vdim) const
-12
View File
@@ -152,18 +152,6 @@ public:
void GetValues(int i, const IntegrationRule &ir, Vector &vals,
DenseMatrix &tr, int vdim = 1) const;
void GetLaplacians(int i, const IntegrationRule &ir, Vector &laps,
int vdim = 1) const;
void GetLaplacians(int i, const IntegrationRule &ir, Vector &laps,
DenseMatrix &tr, int vdim = 1) const;
void GetHessians(int i, const IntegrationRule &ir, DenseMatrix &hess,
int vdim = 1) const;
void GetHessians(int i, const IntegrationRule &ir, DenseMatrix &hess,
DenseMatrix &tr, int vdim = 1) const;
int GetFaceValues(int i, int side, const IntegrationRule &ir, Vector &vals,
DenseMatrix &tr, int vdim = 1) const;
-220
View File
@@ -1,220 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "gslib.hpp"
#ifdef MFEM_USE_GSLIB
// Ignore warnings from the gslib header (GCC version)
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-function"
#endif
#include "gslib.h"
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic pop
#endif
namespace mfem
{
FindPointsGSLIB::FindPointsGSLIB()
: mesh(NULL), gsl_mesh(), fdata2D(NULL), fdata3D(NULL), dim(-1)
{
gsl_comm = new comm;
#ifdef MFEM_USE_MPI
MPI_Init(NULL, NULL);
MPI_Comm comm = MPI_COMM_WORLD;;
comm_init(gsl_comm, comm);
#else
comm_init(gsl_comm, 0);
#endif
}
FindPointsGSLIB::~FindPointsGSLIB()
{
delete gsl_comm;
}
#ifdef MFEM_USE_MPI
FindPointsGSLIB::FindPointsGSLIB(MPI_Comm _comm)
: mesh(NULL), gsl_mesh(), fdata2D(NULL), fdata3D(NULL), dim(-1)
{
gsl_comm = new comm;
comm_init(gsl_comm, _comm);
}
#endif
void FindPointsGSLIB::Setup(Mesh &m, double bb_t, double newt_tol, int npt_max)
{
MFEM_VERIFY(m.GetNodes() != NULL, "Mesh nodes are required.");
mesh = &m;
const GridFunction *nodes = mesh->GetNodes();
const FiniteElementSpace *fes = nodes->FESpace();
dim = mesh->Dimension();
const int NE = mesh->GetNE(),
dof_cnt = fes->GetFE(0)->GetDof(),
pts_cnt = NE * dof_cnt;
gsl_mesh.SetSize(dim * pts_cnt);
const TensorBasisElement *tbe =
dynamic_cast<const TensorBasisElement *>(fes->GetFE(0));
const Array<int> &dof_map = tbe->GetDofMap();
DenseMatrix pos(dof_cnt, dim);
Vector posV(pos.Data(), dof_cnt * dim);
Array<int> xdofs(dof_cnt * dim);
int pt_id = 0;
for (int i = 0; i < NE; i++)
{
fes->GetElementVDofs(i, xdofs);
nodes->GetSubVector(xdofs, posV);
for (int j = 0; j < dof_cnt; j++)
{
for (int d = 0; d < dim; d++)
{
gsl_mesh(pts_cnt * d + pt_id) = pos(dof_map[j], d);
}
pt_id++;
}
}
const unsigned dof1D = fes->GetFE(0)->GetOrder() + 1;
if (dim == 2)
{
unsigned nr[2] = {dof1D, dof1D};
unsigned mr[2] = {2*dof1D, 2*dof1D};
double * const elx[2] = { &gsl_mesh(0), &gsl_mesh(pts_cnt) };
fdata2D = findpts_setup_2(gsl_comm, elx, nr, NE, mr, bb_t,
pts_cnt, pts_cnt, npt_max, newt_tol);
}
else
{
unsigned nr[3] = {dof1D, dof1D, dof1D};
unsigned mr[3] = {2*dof1D, 2*dof1D, 2*dof1D};
double * const elx[3] =
{ &gsl_mesh(0), &gsl_mesh(pts_cnt), &gsl_mesh(2*pts_cnt) };
fdata3D = findpts_setup_3(gsl_comm, elx, nr, NE, mr, bb_t,
pts_cnt, pts_cnt, npt_max, newt_tol);
}
}
void FindPointsGSLIB::FindPoints(Vector &point_pos, Array<unsigned int> &codes,
Array<unsigned int> &proc_ids,
Array<unsigned int> &elem_ids,
Vector &ref_pos, Vector &dist)
{
const int points_cnt = point_pos.Size() / dim;
if (dim == 2)
{
const double *xv_base[2];
xv_base[0] = point_pos.GetData();
xv_base[1] = point_pos.GetData() + points_cnt;
unsigned xv_stride[2];
xv_stride[0] = sizeof(double);
xv_stride[1] = sizeof(double);
findpts_2(codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
dist.GetData(), sizeof(double),
xv_base, xv_stride, points_cnt, fdata2D);
}
else
{
const double *xv_base[3];
xv_base[0] = point_pos.GetData();
xv_base[1] = point_pos.GetData() + points_cnt;
xv_base[2] = point_pos.GetData() + 2*points_cnt;
unsigned xv_stride[3];
xv_stride[0] = sizeof(double);
xv_stride[1] = sizeof(double);
xv_stride[2] = sizeof(double);
findpts_3(codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
dist.GetData(), sizeof(double),
xv_base, xv_stride, points_cnt, fdata3D);
}
}
void FindPointsGSLIB::Interpolate(Array<unsigned int> &codes,
Array<unsigned int> &proc_ids,
Array<unsigned int> &elem_ids,
Vector &ref_pos, const GridFunction &field_in,
Vector &field_out)
{
Vector node_vals;
GetNodeValues(field_in, node_vals);
const int points_cnt = ref_pos.Size() / dim;
if (dim==2)
{
findpts_eval_2(field_out.GetData(), sizeof(double),
codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
points_cnt, node_vals.GetData(), fdata2D);
}
else
{
findpts_eval_3(field_out.GetData(), sizeof(double),
codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
points_cnt, node_vals.GetData(), fdata3D);
}
}
void FindPointsGSLIB::FreeData()
{
(dim == 2) ? findpts_free_2(fdata2D) : findpts_free_3(fdata3D);
}
void FindPointsGSLIB::GetNodeValues(const GridFunction &gf_in,
Vector &node_vals)
{
MFEM_ASSERT(gf_in.FESpace()->GetVDim() == 1, "Scalar function expected.");
const GridFunction *nodes = mesh->GetNodes();
const FiniteElementSpace *fes = nodes->FESpace();
const IntegrationRule &ir = fes->GetFE(0)->GetNodes();
const int NE = mesh->GetNE(), dof_cnt = ir.GetNPoints();
node_vals.SetSize(NE * dof_cnt);
const TensorBasisElement *tbe =
dynamic_cast<const TensorBasisElement *>(fes->GetFE(0));
const Array<int> &dof_map = tbe->GetDofMap();
int pt_id = 0;
Vector vals_el;
for (int i = 0; i < NE; i++)
{
gf_in.GetValues(i, ir, vals_el);
for (int j = 0; j < dof_cnt; j++)
{
node_vals(pt_id++) = vals_el(dof_map[j]);
}
}
}
} // namespace mfem
#endif // MFEM_USE_GSLIB
-107
View File
@@ -1,107 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_GSLIB
#define MFEM_GSLIB
#include "../config/config.hpp"
#include "gridfunc.hpp"
#ifdef MFEM_USE_GSLIB
struct comm;
struct findpts_data_2;
struct findpts_data_3;
namespace mfem
{
class FindPointsGSLIB
{
protected:
Mesh *mesh;
Vector gsl_mesh;
struct findpts_data_2 *fdata2D;
struct findpts_data_3 *fdata3D;
int dim;
struct comm *gsl_comm;
void GetNodeValues(const GridFunction &gf_in, Vector &node_vals);
public:
FindPointsGSLIB();
#ifdef MFEM_USE_MPI
FindPointsGSLIB(MPI_Comm _comm);
#endif
~FindPointsGSLIB();
/** Initializes the internal mesh in gslib, by sending the positions of the
Gauss-Lobatto nodes of the input Mesh object @a m.
Note: not tested with periodic (DG meshes).
Note: the input mesh @a m must have Nodes set.
@param[in] m Input mesh.
@param[in] bb_t Relative size of bounding box around each element.
@param[in] newt_tol Newton tolerance for the gslib search methods.
@param[in] npt_max Number of points for simultaneous iteration. This
alters performance and memory footprint. */
void Setup(Mesh &m, double bb_t, double newt_tol, int npt_max);
/** Searches positions given in physical space by @a point_pos. All output
Arrays and Vectors are expected to have the correct size.
@param[in] point_pos Positions to be found. Must by ordered by nodes
(XXX...,YYY...,ZZZ).
@param[out] codes Return codes for each point: inside element (0),
element boundary (1), not found (2).
@param[out] proc_ids MPI proc ids where the points were found.
@param[out] elem_ids Element ids where the points were found.
@param[out] ref_pos Reference coordinates of the found point. Ordered
by vdim (XYZ,XYZ,XYZ...).
Note: the gslib reference frame is [-1,1].
@param[out] dist Distance between the seeked and the found point
in physical space. */
void FindPoints(Vector &point_pos, Array<unsigned int> &codes,
Array<unsigned int> &proc_ids, Array<unsigned int> &elem_ids,
Vector &ref_pos, Vector &dist);
/** Interpolation of field values at prescribed reference space positions.
@param[in] codes Return codes for each point: inside element (0),
element boundary (1), not found (2).
@param[in] proc_ids MPI proc ids where the points were found.
@param[in] elem_ids Element ids where the points were found.
@param[in] ref_pos Reference coordinates of the found point. Ordered
by vdim (XYZ,XYZ,XYZ...).
Note: the gslib reference frame is [-1,1].
@param[in] field_in Function values that will be interpolated on the
reference positions. Note: it is assumed that
@a field_in is in H1 and in the same space as the
mesh that was given to Setup().
@param[out] field_out Interpolated values. */
void Interpolate(Array<unsigned int> &codes, Array<unsigned int> &proc_ids,
Array<unsigned int> &elem_ids, Vector &ref_pos,
const GridFunction &field_in, Vector &field_out);
/** Cleans up memory allocated internally by gslib.
Note that in parallel, this must be called before MPI_Finalize(), as
it calls MPI_Comm_free() for internal gslib communicators. */
void FreeData();
};
} // namespace mfem
#endif //MFEM_USE_GSLIB
#endif //MFEM_GSLIB guard
-12
View File
@@ -48,8 +48,6 @@ IntegrationRule::IntegrationRule(IntegrationRule &irx, IntegrationRule &iry)
ip.weight = ipx.weight * ipy.weight;
}
}
SetPointIndices();
}
IntegrationRule::IntegrationRule(IntegrationRule &irx, IntegrationRule &iry,
@@ -78,8 +76,6 @@ IntegrationRule::IntegrationRule(IntegrationRule &irx, IntegrationRule &iry,
}
}
}
SetPointIndices();
}
const Array<double> &IntegrationRule::GetWeights() const
@@ -95,14 +91,6 @@ const Array<double> &IntegrationRule::GetWeights() const
return weights;
}
void IntegrationRule::SetPointIndices()
{
for (int i = 0; i < Size(); i++)
{
IntPoint(i).index = i;
}
}
void IntegrationRule::GrundmannMollerSimplexRule(int s, int n)
{
// for pow on older compilers
+2 -10
View File
@@ -26,13 +26,8 @@ class IntegrationPoint
{
public:
double x, y, z, weight;
int index;
void Init(int const i)
{
x = y = z = weight = 0.0;
index = i;
}
void Init() { x = y = z = weight = 0.0; }
void Set(const double *p, const int dim)
{
@@ -96,9 +91,6 @@ private:
by request with the method GetWeights(). */
mutable Array<double> weights;
/// Sets the indices of each quadrature point on initialization.
void SetPointIndices();
/// Define n-simplex rule (triangle/tetrahedron for n=2/3) of order (2s+1)
void GrundmannMollerSimplexRule(int s, int n = 3);
@@ -223,7 +215,7 @@ public:
{
for (int i = 0; i < this->Size(); i++)
{
(*this)[i].Init(i);
(*this)[i].Init();
}
}
-175
View File
@@ -1,175 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "ceed.hpp"
#ifdef MFEM_USE_CEED
#include "../../general/device.hpp"
#include "../../fem/gridfunc.hpp"
#include <sys/types.h>
#include <sys/stat.h>
#ifndef _WIN32
typedef struct stat struct_stat;
#else
#define stat(dir, buf) _stat(dir, buf)
#define S_ISDIR(mode) _S_IFDIR(mode)
typedef struct _stat struct_stat;
#endif
namespace mfem
{
namespace internal
{
extern Ceed ceed;
std::string ceed_path;
}
void InitCeedCoeff(Coefficient* Q, CeedData* ptr)
{
if (ConstantCoefficient* coeff = dynamic_cast<ConstantCoefficient*>(Q))
{
CeedConstCoeff* ceedCoeff = new CeedConstCoeff{coeff->constant};
ptr->coeff_type = CeedCoeff::Const;
ptr->coeff = (void*)ceedCoeff;
}
else if (GridFunctionCoefficient* coeff =
dynamic_cast<GridFunctionCoefficient*>(Q))
{
CeedGridCoeff* ceedCoeff = new CeedGridCoeff;
ceedCoeff->coeff = coeff->GetGridFunction();
ptr->coeff_type = CeedCoeff::Grid;
ptr->coeff = (void*)ceedCoeff;
}
else
{
MFEM_ABORT("This type of Coefficient is not supported.");
}
}
void InitCeedTensorBasisAndRestriction(const mfem::FiniteElementSpace &fes,
const mfem::IntegrationRule &ir,
Ceed ceed, CeedBasis *basis,
CeedElemRestriction *restr)
{
mfem::Mesh *mesh = fes.GetMesh();
const mfem::FiniteElement *fe = fes.GetFE(0);
const int order = fes.GetOrder(0);
mfem::Array<int> dof_map;
switch (mesh->Dimension())
{
case 1:
{
const mfem::H1_SegmentElement *h1_fe =
dynamic_cast<const mfem::H1_SegmentElement *>(fe);
MFEM_VERIFY(h1_fe, "invalid FE");
h1_fe->GetDofMap().Copy(dof_map);
break;
}
case 2:
{
const mfem::H1_QuadrilateralElement *h1_fe =
dynamic_cast<const mfem::H1_QuadrilateralElement *>(fe);
MFEM_VERIFY(h1_fe, "invalid FE");
h1_fe->GetDofMap().Copy(dof_map);
break;
}
case 3:
{
const mfem::H1_HexahedronElement *h1_fe =
dynamic_cast<const mfem::H1_HexahedronElement *>(fe);
MFEM_VERIFY(h1_fe, "invalid FE");
h1_fe->GetDofMap().Copy(dof_map);
break;
}
}
const mfem::FiniteElement *fe1d =
fes.FEColl()->FiniteElementForGeometry(mfem::Geometry::SEGMENT);
mfem::DenseMatrix shape1d(fe1d->GetDof(), ir.GetNPoints());
mfem::DenseMatrix grad1d(fe1d->GetDof(), ir.GetNPoints());
mfem::Vector qref1d(ir.GetNPoints()), qweight1d(ir.GetNPoints());
mfem::Vector shape_i(shape1d.Height());
mfem::DenseMatrix grad_i(grad1d.Height(), 1);
const mfem::H1_SegmentElement *h1_fe1d =
dynamic_cast<const mfem::H1_SegmentElement *>(fe1d);
MFEM_VERIFY(h1_fe1d, "invalid FE");
const mfem::Array<int> &dof_map_1d = h1_fe1d->GetDofMap();
for (int i = 0; i < ir.GetNPoints(); i++)
{
const mfem::IntegrationPoint &ip = ir.IntPoint(i);
qref1d(i) = ip.x;
qweight1d(i) = ip.weight;
fe1d->CalcShape(ip, shape_i);
fe1d->CalcDShape(ip, grad_i);
for (int j = 0; j < shape1d.Height(); j++)
{
shape1d(j, i) = shape_i(dof_map_1d[j]);
grad1d(j, i) = grad_i(dof_map_1d[j], 0);
}
}
CeedBasisCreateTensorH1(ceed, mesh->Dimension(), fes.GetVDim(), order + 1,
ir.GetNPoints(), shape1d.GetData(),
grad1d.GetData(), qref1d.GetData(),
qweight1d.GetData(), basis);
const mfem::Table &el_dof = fes.GetElementToDofTable();
mfem::Array<int> tp_el_dof(el_dof.Size_of_connections());
for (int i = 0; i < mesh->GetNE(); i++)
{
const int el_offset = fe->GetDof() * i;
for (int j = 0; j < fe->GetDof(); j++)
{
tp_el_dof[j + el_offset] = el_dof.GetJ()[dof_map[j] + el_offset];
}
}
CeedInterlaceMode imode = CEED_NONINTERLACED;
if (fes.GetOrdering()==Ordering::byVDIM)
{
imode = CEED_INTERLACED;
}
CeedElemRestrictionCreate(ceed, imode, mesh->GetNE(), fe->GetDof(),
fes.GetNDofs(), fes.GetVDim(), CEED_MEM_HOST, CEED_COPY_VALUES,
tp_el_dof.GetData(), restr);
}
const std::string &GetCeedPath()
{
if (internal::ceed_path.empty())
{
const char *install_dir = MFEM_INSTALL_DIR "/include/mfem/fem/libceed";
const char *source_dir = MFEM_SOURCE_DIR "/fem/libceed";
struct_stat m_stat;
if (stat(install_dir, &m_stat) == 0 && S_ISDIR(m_stat.st_mode))
{
internal::ceed_path = install_dir;
}
else if (stat(source_dir, &m_stat) == 0 && S_ISDIR(m_stat.st_mode))
{
internal::ceed_path = source_dir;
}
else
{
MFEM_ABORT("Cannot find libCEED kernels in MFEM_INSTALL_DIR or "
"MFEM_SOURCE_DIR");
}
// Could be useful for debugging:
// mfem::out << "Using libCEED dir: " << internal::ceed_path << std::endl;
}
return internal::ceed_path;
}
} // namespace mfem
#endif // MFEM_USE_CEED
-132
View File
@@ -1,132 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_LIBCEED_HPP
#define MFEM_LIBCEED_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_CEED
#include "../../general/device.hpp"
#include <ceed.h>
namespace mfem
{
class FiniteElementSpace;
class GridFunction;
class IntegrationRule;
class Coefficient;
namespace internal { extern Ceed ceed; } // defined in device.cpp
/// A structure used to pass additional data to f_build_diff and f_apply_diff
struct BuildContext { CeedInt dim, space_dim; CeedScalar coeff; };
enum class CeedCoeff { Const, Grid };
struct CeedConstCoeff
{
double val;
};
struct CeedGridCoeff
{
GridFunction* coeff;
CeedBasis basis;
CeedElemRestriction restr;
CeedVector coeffVector;
};
struct CeedData
{
CeedOperator build_oper, oper;
CeedBasis basis, mesh_basis;
CeedElemRestriction restr, mesh_restr, restr_i, mesh_restr_i;
CeedQFunction apply_qfunc, build_qfunc;
CeedVector node_coords, rho;
CeedCoeff coeff_type;
void* coeff;
BuildContext build_ctx;
CeedVector u, v;
~CeedData()
{
CeedOperatorDestroy(&build_oper);
CeedOperatorDestroy(&oper);
CeedBasisDestroy(&basis);
CeedBasisDestroy(&mesh_basis);
CeedElemRestrictionDestroy(&restr);
CeedElemRestrictionDestroy(&mesh_restr);
CeedElemRestrictionDestroy(&restr_i);
CeedElemRestrictionDestroy(&mesh_restr_i);
CeedQFunctionDestroy(&apply_qfunc);
CeedQFunctionDestroy(&build_qfunc);
CeedVectorDestroy(&node_coords);
CeedVectorDestroy(&rho);
if (coeff_type==CeedCoeff::Grid)
{
CeedGridCoeff* c = (CeedGridCoeff*)coeff;
CeedBasisDestroy(&c->basis);
CeedElemRestrictionDestroy(&c->restr);
CeedVectorDestroy(&c->coeffVector);
delete c;
}
else
{
delete (CeedConstCoeff*)coeff;
}
CeedVectorDestroy(&u);
CeedVectorDestroy(&v);
}
};
/** @brief Identifies the type of coefficient of the Integrator to initialize
accordingly the CeedData. */
void InitCeedCoeff(Coefficient* Q, CeedData* ptr);
/// Initialize a tensor CeedBasis and a CeedElemRestriction
void InitCeedTensorBasisAndRestriction(const FiniteElementSpace &fes,
const IntegrationRule &ir,
Ceed ceed, CeedBasis *basis,
CeedElemRestriction *restr);
/// Return the path to the libCEED q-function headers.
const std::string &GetCeedPath();
/** @brief Function that determines if a CEED kernel should be used, based on
the current mfem::Device configuration. */
inline bool DeviceCanUseCeed()
{
return Device::Allows(Backend::CEED_CUDA) ||
(Device::Allows(Backend::CEED_CPU) &&
!Device::Allows(Backend::DEVICE_MASK|Backend::OMP_MASK));
}
} // namespace mfem
#else // MFEM_USE_CEED
namespace mfem
{
inline bool DeviceCanUseCeed()
{
return false;
}
} // namespace mfem
#endif // MFEM_USE_CEED
#endif // MFEM_LIBCEED_HPP
-153
View File
@@ -1,153 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "diffusion.hpp"
#ifdef MFEM_USE_CEED
#include "../../general/device.hpp"
#include "../../mesh/mesh.hpp"
#include "../../fem/gridfunc.hpp"
#include "ceed.hpp"
#include "diffusion.h"
namespace mfem
{
void CeedPADiffusionAssemble(const FiniteElementSpace &fes,
const mfem::IntegrationRule &irm, CeedData& ceedData)
{
Ceed ceed(internal::ceed);
mfem::Mesh *mesh = fes.GetMesh();
const int ir_order = irm.GetOrder();
const mfem::IntegrationRule &ir =
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir_order);
CeedInt nqpts, nelem = mesh->GetNE(), dim = mesh->SpaceDimension();
mesh->EnsureNodes();
InitCeedTensorBasisAndRestriction(fes, ir, ceed, &ceedData.basis,
&ceedData.restr);
const mfem::FiniteElementSpace *mesh_fes = mesh->GetNodalFESpace();
MFEM_VERIFY(mesh_fes, "the Mesh has no nodal FE space");
InitCeedTensorBasisAndRestriction(*mesh_fes, ir, ceed, &ceedData.mesh_basis,
&ceedData.mesh_restr);
CeedBasisGetNumQuadraturePoints(ceedData.basis, &nqpts);
CeedInterlaceMode imode = CEED_NONINTERLACED;
if (fes.GetOrdering()==Ordering::byVDIM)
{
imode = CEED_INTERLACED;
}
CeedElemRestrictionCreateIdentity(ceed, imode, nelem, nqpts, nqpts * nelem,
dim * (dim + 1) / 2, &ceedData.restr_i);
CeedInterlaceMode mesh_imode = CEED_NONINTERLACED;
if (mesh_fes->GetOrdering()==Ordering::byVDIM)
{
mesh_imode = CEED_INTERLACED;
}
CeedElemRestrictionCreateIdentity(ceed, mesh_imode, nelem, nqpts,
nqpts * nelem, 1, &ceedData.mesh_restr_i);
CeedVectorCreate(ceed, mesh->GetNodes()->Size(), &ceedData.node_coords);
CeedVectorSetArray(ceedData.node_coords, CEED_MEM_HOST, CEED_USE_POINTER,
mesh->GetNodes()->GetData());
CeedVectorCreate(ceed, nelem * nqpts * dim * (dim + 1) / 2, &ceedData.rho);
// Context data to be passed to the 'f_build_diff' Q-function.
ceedData.build_ctx.dim = mesh->Dimension();
ceedData.build_ctx.space_dim = mesh->SpaceDimension();
std::string diff_qf_file = GetCeedPath() + "/diffusion.h";
std::string diff_qf;
// Create the Q-function that builds the diff operator (i.e. computes its
// quadrature data) and set its context data.
switch (ceedData.coeff_type)
{
case CeedCoeff::Const:
diff_qf = diff_qf_file + ":f_build_diff_const";
CeedQFunctionCreateInterior(ceed, 1, f_build_diff_const,
diff_qf.c_str(),
&ceedData.build_qfunc);
ceedData.build_ctx.coeff = ((CeedConstCoeff*)ceedData.coeff)->val;
break;
case CeedCoeff::Grid:
diff_qf = diff_qf_file + ":f_build_diff_grid";
CeedQFunctionCreateInterior(ceed, 1, f_build_diff_grid,
diff_qf.c_str(),
&ceedData.build_qfunc);
CeedQFunctionAddInput(ceedData.build_qfunc, "coeff", 1, CEED_EVAL_INTERP);
break;
default:
MFEM_ABORT("This coeff_type is not handled");
}
CeedQFunctionAddInput(ceedData.build_qfunc, "dx", dim * dim, CEED_EVAL_GRAD);
CeedQFunctionAddInput(ceedData.build_qfunc, "weights", 1, CEED_EVAL_WEIGHT);
CeedQFunctionAddOutput(ceedData.build_qfunc, "rho", dim * (dim + 1) / 2,
CEED_EVAL_NONE);
CeedQFunctionSetContext(ceedData.build_qfunc, &ceedData.build_ctx,
sizeof(ceedData.build_ctx));
// Create the operator that builds the quadrature data for the diff operator.
CeedOperatorCreate(ceed, ceedData.build_qfunc, NULL, NULL,
&ceedData.build_oper);
if (ceedData.coeff_type==CeedCoeff::Grid)
{
CeedGridCoeff* ceedCoeff = (CeedGridCoeff*)ceedData.coeff;
InitCeedTensorBasisAndRestriction(*ceedCoeff->coeff->FESpace(), ir, ceed,
&ceedCoeff->basis,
&ceedCoeff->restr);
CeedVectorCreate(ceed, ceedCoeff->coeff->FESpace()->GetNDofs(),
&ceedCoeff->coeffVector);
CeedVectorSetArray(ceedCoeff->coeffVector, CEED_MEM_HOST, CEED_USE_POINTER,
ceedCoeff->coeff->GetData());
CeedOperatorSetField(ceedData.build_oper, "coeff", ceedCoeff->restr,
ceedCoeff->basis, ceedCoeff->coeffVector);
}
CeedOperatorSetField(ceedData.build_oper, "dx", ceedData.mesh_restr,
ceedData.mesh_basis, CEED_VECTOR_ACTIVE);
CeedOperatorSetField(ceedData.build_oper, "weights", ceedData.mesh_restr_i,
ceedData.mesh_basis, CEED_VECTOR_NONE);
CeedOperatorSetField(ceedData.build_oper, "rho", ceedData.restr_i,
CEED_BASIS_COLLOCATED, CEED_VECTOR_ACTIVE);
// Compute the quadrature data for the diff operator.
CeedOperatorApply(ceedData.build_oper, ceedData.node_coords, ceedData.rho,
CEED_REQUEST_IMMEDIATE);
// Create the Q-function that defines the action of the diff operator.
diff_qf = diff_qf_file + ":f_apply_diff";
CeedQFunctionCreateInterior(ceed, 1, f_apply_diff,
diff_qf.c_str(),
&ceedData.apply_qfunc);
CeedQFunctionAddInput(ceedData.apply_qfunc, "u", dim, CEED_EVAL_GRAD);
CeedQFunctionAddInput(ceedData.apply_qfunc, "rho", dim * (dim + 1) / 2,
CEED_EVAL_NONE);
CeedQFunctionAddOutput(ceedData.apply_qfunc, "v", dim, CEED_EVAL_GRAD);
CeedQFunctionSetContext(ceedData.apply_qfunc, &ceedData.build_ctx,
sizeof(ceedData.build_ctx));
// Create the diff operator.
CeedOperatorCreate(ceed, ceedData.apply_qfunc, NULL, NULL, &ceedData.oper);
CeedOperatorSetField(ceedData.oper, "u", ceedData.restr, ceedData.basis,
CEED_VECTOR_ACTIVE);
CeedOperatorSetField(ceedData.oper, "rho", ceedData.restr_i,
CEED_BASIS_COLLOCATED, ceedData.rho);
CeedOperatorSetField(ceedData.oper, "v", ceedData.restr, ceedData.basis,
CEED_VECTOR_ACTIVE);
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.u);
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.v);
}
} // namespace mfem
#endif // MFEM_USE_CEED
-200
View File
@@ -1,200 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
/// A structure used to pass additional data to f_build_diff and f_apply_diff
struct BuildContext { CeedInt dim, space_dim; CeedScalar coeff; };
/// libCEED Q-function for building quadrature data for a diffusion operator with a constant coefficient
CEED_QFUNCTION(f_build_diff_const)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
BuildContext *bc = (BuildContext*)ctx;
// in[0] is Jacobians with shape [dim, nc=dim, Q]
// in[1] is quadrature weights, size (Q)
//
// At every quadrature point, compute qw/det(J).adj(J).adj(J)^T and store
// the symmetric part of the result.
const CeedScalar coeff = bc->coeff;
const CeedScalar *J = in[0], *qw = in[1];
CeedScalar *qd = out[0];
switch (bc->dim + 10 * bc->space_dim)
{
case 11:
for (CeedInt i = 0; i < Q; i++)
{
qd[i] = coeff * qw[i] / J[i];
}
break;
case 22:
for (CeedInt i = 0; i < Q; i++)
{
// J: 0 2 qd: 0 1 adj(J): J22 -J12
// 1 3 1 2 -J21 J11
const CeedScalar J11 = J[i + Q * 0];
const CeedScalar J21 = J[i + Q * 1];
const CeedScalar J12 = J[i + Q * 2];
const CeedScalar J22 = J[i + Q * 3];
const CeedScalar w = qw[i] / (J11 * J22 - J21 * J12);
qd[i + Q * 0] = coeff * w * (J12 * J12 + J22 * J22);
qd[i + Q * 1] = - coeff * w * (J11 * J12 + J21 * J22);
qd[i + Q * 2] = coeff * w * (J11 * J11 + J21 * J21);
}
break;
case 33:
for (CeedInt i = 0; i < Q; i++)
{
// J: 0 3 6 qd: 0 1 2
// 1 4 7 1 3 4
// 2 5 8 2 4 5
const CeedScalar J11 = J[i + Q * 0];
const CeedScalar J21 = J[i + Q * 1];
const CeedScalar J31 = J[i + Q * 2];
const CeedScalar J12 = J[i + Q * 3];
const CeedScalar J22 = J[i + Q * 4];
const CeedScalar J32 = J[i + Q * 5];
const CeedScalar J13 = J[i + Q * 6];
const CeedScalar J23 = J[i + Q * 7];
const CeedScalar J33 = J[i + Q * 8];
const CeedScalar A11 = J22 * J33 - J23 * J32;
const CeedScalar A12 = J13 * J32 - J12 * J33;
const CeedScalar A13 = J12 * J23 - J13 * J22;
const CeedScalar A21 = J23 * J31 - J21 * J33;
const CeedScalar A22 = J11 * J33 - J13 * J31;
const CeedScalar A23 = J13 * J21 - J11 * J23;
const CeedScalar A31 = J21 * J32 - J22 * J31;
const CeedScalar A32 = J12 * J31 - J11 * J32;
const CeedScalar A33 = J11 * J22 - J12 * J21;
const CeedScalar w = qw[i] / (J11 * A11 + J21 * A12 + J31 * A13);
qd[i + Q * 0] = coeff * w * (A11 * A11 + A12 * A12 + A13 * A13);
qd[i + Q * 1] = coeff * w * (A11 * A21 + A12 * A22 + A13 * A23);
qd[i + Q * 2] = coeff * w * (A11 * A31 + A12 * A32 + A13 * A33);
qd[i + Q * 3] = coeff * w * (A21 * A21 + A22 * A22 + A23 * A23);
qd[i + Q * 4] = coeff * w * (A21 * A31 + A22 * A32 + A23 * A33);
qd[i + Q * 5] = coeff * w * (A31 * A31 + A32 * A32 + A33 * A33);
}
break;
}
return 0;
}
/// libCEED Q-function for building quadrature data for a diffusion operator with a grid function coefficient
CEED_QFUNCTION(f_build_diff_grid)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
BuildContext *bc = (BuildContext *)ctx;
// in[1] is Jacobians with shape [dim, nc=dim, Q]
// in[2] is quadrature weights, size (Q)
//
// At every quadrature point, compute qw/det(J).adj(J).adj(J)^T and store
// the symmetric part of the result.
const CeedScalar *c = in[0], *J = in[1], *qw = in[2];
CeedScalar *qd = out[0];
switch (bc->dim + 10 * bc->space_dim)
{
case 11:
for (CeedInt i = 0; i < Q; i++)
{
qd[i] = c[i] * qw[i] / J[i];
}
break;
case 22:
for (CeedInt i = 0; i < Q; i++)
{
// J: 0 2 qd: 0 1 adj(J): J22 -J12
// 1 3 1 2 -J21 J11
const CeedScalar coeff = c[i];
const CeedScalar J11 = J[i + Q * 0];
const CeedScalar J21 = J[i + Q * 1];
const CeedScalar J12 = J[i + Q * 2];
const CeedScalar J22 = J[i + Q * 3];
const CeedScalar w = qw[i] / (J11 * J22 - J21 * J12);
qd[i + Q * 0] = coeff * w * (J12 * J12 + J22 * J22);
qd[i + Q * 1] = - coeff * w * (J11 * J12 + J21 * J22);
qd[i + Q * 2] = coeff * w * (J11 * J11 + J21 * J21);
}
break;
case 33:
for (CeedInt i = 0; i < Q; i++)
{
// J: 0 3 6 qd: 0 1 2
// 1 4 7 1 3 4
// 2 5 8 2 4 5
const CeedScalar coeff = c[i];
const CeedScalar J11 = J[i + Q * 0];
const CeedScalar J21 = J[i + Q * 1];
const CeedScalar J31 = J[i + Q * 2];
const CeedScalar J12 = J[i + Q * 3];
const CeedScalar J22 = J[i + Q * 4];
const CeedScalar J32 = J[i + Q * 5];
const CeedScalar J13 = J[i + Q * 6];
const CeedScalar J23 = J[i + Q * 7];
const CeedScalar J33 = J[i + Q * 8];
const CeedScalar A11 = J22 * J33 - J23 * J32;
const CeedScalar A12 = J13 * J32 - J12 * J33;
const CeedScalar A13 = J12 * J23 - J13 * J22;
const CeedScalar A21 = J23 * J31 - J21 * J33;
const CeedScalar A22 = J11 * J33 - J13 * J31;
const CeedScalar A23 = J13 * J21 - J11 * J23;
const CeedScalar A31 = J21 * J32 - J22 * J31;
const CeedScalar A32 = J12 * J31 - J11 * J32;
const CeedScalar A33 = J11 * J22 - J12 * J21;
const CeedScalar w = qw[i] / (J11 * A11 + J21 * A12 + J31 * A13);
qd[i + Q * 0] = coeff * w * (A11 * A11 + A12 * A12 + A13 * A13);
qd[i + Q * 1] = coeff * w * (A11 * A21 + A12 * A22 + A13 * A23);
qd[i + Q * 2] = coeff * w * (A11 * A31 + A12 * A32 + A13 * A33);
qd[i + Q * 3] = coeff * w * (A21 * A21 + A22 * A22 + A23 * A23);
qd[i + Q * 4] = coeff * w * (A21 * A31 + A22 * A32 + A23 * A33);
qd[i + Q * 5] = coeff * w * (A31 * A31 + A32 * A32 + A33 * A33);
}
break;
}
return 0;
}
/// libCEED Q-function for applying a diff operator
CEED_QFUNCTION(f_apply_diff)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
BuildContext *bc = (BuildContext *)ctx;
// in[0], out[0] have shape [dim, nc=1, Q]
const CeedScalar *ug = in[0], *qd = in[1];
CeedScalar *vg = out[0];
switch (bc->dim)
{
case 1:
for (CeedInt i = 0; i < Q; i++)
{
vg[i] = ug[i] * qd[i];
}
break;
case 2:
for (CeedInt i = 0; i < Q; i++)
{
const CeedScalar ug0 = ug[i + Q * 0];
const CeedScalar ug1 = ug[i + Q * 1];
vg[i + Q * 0] = qd[i + Q * 0] * ug0 + qd[i + Q * 1] * ug1;
vg[i + Q * 1] = qd[i + Q * 1] * ug0 + qd[i + Q * 2] * ug1;
}
break;
case 3:
for (CeedInt i = 0; i < Q; i++)
{
const CeedScalar ug0 = ug[i + Q * 0];
const CeedScalar ug1 = ug[i + Q * 1];
const CeedScalar ug2 = ug[i + Q * 2];
vg[i + Q * 0] = qd[i + Q * 0] * ug0 + qd[i + Q * 1] * ug1 + qd[i + Q * 2] * ug2;
vg[i + Q * 1] = qd[i + Q * 1] * ug0 + qd[i + Q * 3] * ug1 + qd[i + Q * 4] * ug2;
vg[i + Q * 2] = qd[i + Q * 2] * ug0 + qd[i + Q * 4] * ug1 + qd[i + Q * 5] * ug2;
}
break;
}
return 0;
}
-31
View File
@@ -1,31 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_LIBCEED_DIFF_HPP
#define MFEM_LIBCEED_DIFF_HPP
#include "ceed.hpp"
#ifdef MFEM_USE_CEED
#include "../fespace.hpp"
namespace mfem
{
/// Initialize a Diffusion Integrator using libCEED
void CeedPADiffusionAssemble(const FiniteElementSpace &fes,
const mfem::IntegrationRule &ir, CeedData& ceedData);
}
#endif // MFEM_USE_CEED
#endif // MFEM_LIBCEED_DIFF_HPP
-150
View File
@@ -1,150 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mass.hpp"
#ifdef MFEM_USE_CEED
#include "../../general/device.hpp"
#include "../../mesh/mesh.hpp"
#include "../../fem/gridfunc.hpp"
#include "ceed.hpp"
#include "mass.h"
namespace mfem
{
void CeedPAMassAssemble(const FiniteElementSpace &fes,
const mfem::IntegrationRule &irm, CeedData& ceedData)
{
Ceed ceed(internal::ceed);
mfem::Mesh *mesh = fes.GetMesh();
const int ir_order = irm.GetOrder();
const mfem::IntegrationRule &ir =
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir_order);
CeedInt nqpts, nelem = mesh->GetNE();
InitCeedTensorBasisAndRestriction(fes, ir, ceed, &ceedData.basis,
&ceedData.restr);
mesh->EnsureNodes();
const mfem::FiniteElementSpace *mesh_fes = mesh->GetNodalFESpace();
MFEM_VERIFY(mesh_fes, "the Mesh has no nodal FE space");
InitCeedTensorBasisAndRestriction(*mesh_fes, ir, ceed, &ceedData.mesh_basis,
&ceedData.mesh_restr);
CeedBasisGetNumQuadraturePoints(ceedData.basis, &nqpts);
CeedInterlaceMode imode = CEED_NONINTERLACED;
if (fes.GetOrdering()==Ordering::byVDIM)
{
imode = CEED_INTERLACED;
}
CeedElemRestrictionCreateIdentity(ceed, imode, nelem, nqpts,
nqpts*nelem, 1, &ceedData.restr_i);
CeedInterlaceMode mesh_imode = CEED_NONINTERLACED;
if (mesh_fes->GetOrdering()==Ordering::byVDIM)
{
mesh_imode = CEED_INTERLACED;
}
CeedElemRestrictionCreateIdentity(ceed, mesh_imode, nelem, nqpts,
nqpts*nelem, 1, &ceedData.mesh_restr_i);
CeedVectorCreate(ceed, mesh->GetNodes()->Size(), &ceedData.node_coords);
CeedVectorSetArray(ceedData.node_coords, CEED_MEM_HOST, CEED_USE_POINTER,
mesh->GetNodes()->GetData());
CeedVectorCreate(ceed, nelem*nqpts, &ceedData.rho);
// Context data to be passed to the 'f_build_mass' Q-function.
ceedData.build_ctx.dim = mesh->Dimension();
ceedData.build_ctx.space_dim = mesh->SpaceDimension();
std::string mass_qf_file = GetCeedPath() + "/mass.h";
std::string mass_qf;
// Create the Q-function that builds the mass operator (i.e. computes its
// quadrature data) and set its context data.
switch (ceedData.coeff_type)
{
case CeedCoeff::Const:
mass_qf = mass_qf_file + ":f_build_mass_const";
CeedQFunctionCreateInterior(ceed, 1, f_build_mass_const,
mass_qf.c_str(),
&ceedData.build_qfunc);
ceedData.build_ctx.coeff = ((CeedConstCoeff*)ceedData.coeff)->val;
break;
case CeedCoeff::Grid:
mass_qf = mass_qf_file + ":f_build_mass_grid";
CeedQFunctionCreateInterior(ceed, 1, f_build_mass_grid,
mass_qf.c_str(),
&ceedData.build_qfunc);
CeedQFunctionAddInput(ceedData.build_qfunc, "coeff", 1, CEED_EVAL_INTERP);
break;
default:
MFEM_ABORT("This coeff_type is not handled");
}
CeedQFunctionAddInput(ceedData.build_qfunc, "dx",
mesh->SpaceDimension()*mesh->SpaceDimension(),
CEED_EVAL_GRAD);
CeedQFunctionAddInput(ceedData.build_qfunc, "weights", 1, CEED_EVAL_WEIGHT);
CeedQFunctionAddOutput(ceedData.build_qfunc, "rho", 1, CEED_EVAL_NONE);
CeedQFunctionSetContext(ceedData.build_qfunc, &ceedData.build_ctx,
sizeof(ceedData.build_ctx));
// Create the operator that builds the quadrature data for the mass operator.
CeedOperatorCreate(ceed, ceedData.build_qfunc, NULL, NULL,
&ceedData.build_oper);
if (ceedData.coeff_type==CeedCoeff::Grid)
{
CeedGridCoeff* ceedCoeff = (CeedGridCoeff*)ceedData.coeff;
InitCeedTensorBasisAndRestriction(*ceedCoeff->coeff->FESpace(), ir, ceed,
&ceedCoeff->basis,
&ceedCoeff->restr);
CeedVectorCreate(ceed, ceedCoeff->coeff->FESpace()->GetNDofs(),
&ceedCoeff->coeffVector);
CeedVectorSetArray(ceedCoeff->coeffVector, CEED_MEM_HOST, CEED_USE_POINTER,
ceedCoeff->coeff->GetData());
CeedOperatorSetField(ceedData.build_oper, "coeff", ceedCoeff->restr,
ceedCoeff->basis, ceedCoeff->coeffVector);
}
CeedOperatorSetField(ceedData.build_oper, "dx", ceedData.mesh_restr,
ceedData.mesh_basis, CEED_VECTOR_ACTIVE);
CeedOperatorSetField(ceedData.build_oper, "weights", ceedData.mesh_restr_i,
ceedData.mesh_basis, CEED_VECTOR_NONE);
CeedOperatorSetField(ceedData.build_oper, "rho", ceedData.restr_i,
CEED_BASIS_COLLOCATED, CEED_VECTOR_ACTIVE);
// Compute the quadrature data for the mass operator.
CeedOperatorApply(ceedData.build_oper, ceedData.node_coords, ceedData.rho,
CEED_REQUEST_IMMEDIATE);
// Create the Q-function that defines the action of the mass operator.
mass_qf = mass_qf_file + ":f_apply_mass";
CeedQFunctionCreateInterior(ceed, 1, f_apply_mass,
mass_qf.c_str(), &ceedData.apply_qfunc);
CeedQFunctionAddInput(ceedData.apply_qfunc, "u", 1, CEED_EVAL_INTERP);
CeedQFunctionAddInput(ceedData.apply_qfunc, "rho", 1, CEED_EVAL_NONE);
CeedQFunctionAddOutput(ceedData.apply_qfunc, "v", 1, CEED_EVAL_INTERP);
// Create the mass operator.
CeedOperatorCreate(ceed, ceedData.apply_qfunc, NULL, NULL, &ceedData.oper);
CeedOperatorSetField(ceedData.oper, "u", ceedData.restr,
ceedData.basis, CEED_VECTOR_ACTIVE);
CeedOperatorSetField(ceedData.oper, "rho", ceedData.restr_i,
CEED_BASIS_COLLOCATED, ceedData.rho);
CeedOperatorSetField(ceedData.oper, "v", ceedData.restr,
ceedData.basis, CEED_VECTOR_ACTIVE);
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.u);
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.v);
}
} // namespace mfem
#endif // MFEM_USE_CEED
-107
View File
@@ -1,107 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
/// A structure used to pass additional data to f_build_diff and f_apply_diff
struct BuildContext { CeedInt dim, space_dim; CeedScalar coeff; };
/// libCEED Q-function for building quadrature data for a mass operator with a constant coefficient
CEED_QFUNCTION(f_build_mass_const)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
// in[0] is Jacobians with shape [dim, nc=dim, Q]
// in[1] is quadrature weights, size (Q)
BuildContext *bc = (BuildContext *)ctx;
const CeedScalar coeff = bc->coeff;
const CeedScalar *J = in[0], *qw = in[1];
CeedScalar *rho = out[0];
switch (bc->dim + 10*bc->space_dim)
{
case 11:
for (CeedInt i=0; i<Q; i++)
{
rho[i] = coeff * J[i] * qw[i];
}
break;
case 22:
for (CeedInt i=0; i<Q; i++)
{
// 0 2
// 1 3
rho[i] = coeff * (J[i+Q*0]*J[i+Q*3] - J[i+Q*1]*J[i+Q*2]) * qw[i];
}
break;
case 33:
for (CeedInt i=0; i<Q; i++)
{
// 0 3 6
// 1 4 7
// 2 5 8
rho[i] = (J[i+Q*0]*(J[i+Q*4]*J[i+Q*8] - J[i+Q*5]*J[i+Q*7]) -
J[i+Q*1]*(J[i+Q*3]*J[i+Q*8] - J[i+Q*5]*J[i+Q*6]) +
J[i+Q*2]*(J[i+Q*3]*J[i+Q*7] - J[i+Q*4]*J[i+Q*6])) * coeff * qw[i];
}
break;
}
return 0;
}
/// libCEED Q-function for building quadrature data for a mass operator with a grid function coefficient
CEED_QFUNCTION(f_build_mass_grid)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
// in[0] is Jacobians with shape [dim, nc=dim, Q]
// in[1] is quadrature weights, size (Q)
BuildContext *bc = (BuildContext *)ctx;
const CeedScalar *c = in[0], *J = in[1], *qw = in[2];
CeedScalar *rho = out[0];
switch (bc->dim + 10*bc->space_dim)
{
case 11:
for (CeedInt i=0; i<Q; i++)
{
rho[i] = c[i] * J[i] * qw[i];
}
break;
case 22:
for (CeedInt i=0; i<Q; i++)
{
// 0 2
// 1 3
rho[i] = c[i] * (J[i+Q*0]*J[i+Q*3] - J[i+Q*1]*J[i+Q*2]) * qw[i];
}
break;
case 33:
for (CeedInt i=0; i<Q; i++)
{
// 0 3 6
// 1 4 7
// 2 5 8
rho[i] = (J[i+Q*0]*(J[i+Q*4]*J[i+Q*8] - J[i+Q*5]*J[i+Q*7]) -
J[i+Q*1]*(J[i+Q*3]*J[i+Q*8] - J[i+Q*5]*J[i+Q*6]) +
J[i+Q*2]*(J[i+Q*3]*J[i+Q*7] - J[i+Q*4]*J[i+Q*6])) * c[i] * qw[i];
}
break;
}
return 0;
}
/// libCEED Q-function for applying a mass operator
CEED_QFUNCTION(f_apply_mass)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
const CeedScalar *u = in[0], *w = in[1];
CeedScalar *v = out[0];
for (CeedInt i=0; i<Q; i++)
{
v[i] = w[i] * u[i];
}
return 0;
}
-31
View File
@@ -1,31 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_LIBCEED_MASS_HPP
#define MFEM_LIBCEED_MASS_HPP
#include "ceed.hpp"
#ifdef MFEM_USE_CEED
#include "../fespace.hpp"
namespace mfem
{
/// Initialize a Mass Integrator using libCEED
void CeedPAMassAssemble(const FiniteElementSpace &fes,
const mfem::IntegrationRule &ir, CeedData& ceedData);
}
#endif // MFEM_USE_CEED
#endif // MFEM_LIBCEED_MASS_HPP
-8
View File
@@ -81,14 +81,6 @@ public:
Update(FiniteElementSpace *, Vector &, int). */
LinearForm() { fes = NULL; extern_lfs = 0; UseDevice(true); }
/// Construct a LinearForm using previously allocated array @a data.
/** The LinearForm does not assume ownership of @a data which is assumed to
be of size at least `f->GetVSize()`. Similar to the Vector constructor
for externally allocated array, the pointer @a data can be NULL. The data
array can be replaced later using the method SetData(). */
LinearForm(FiniteElementSpace *f, double *data) : Vector(data, f->GetVSize())
{ fes = f; extern_lfs = 0; }
/// Copy assignment. Only the data of the base class Vector is copied.
/** It is assumed that this object and @a rhs use FiniteElementSpace%s that
have the same size.
+74 -48
View File
@@ -14,25 +14,6 @@
namespace mfem
{
void NonlinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
{
if (ext)
{
MFEM_ABORT("the assembly level has already been set!");
}
assembly = assembly_level;
switch (assembly)
{
case AssemblyLevel::NONE:
// This is the default behavior.
break;
case AssemblyLevel::PARTIAL:
ext = new PANonlinearFormExtension(this);
break;
default:
mfem_error("Unknown assembly level for this form.");
}
}
void NonlinearForm::SetEssentialBC(const Array<int> &bdr_attr_is_ess,
Vector *rhs)
{
@@ -84,6 +65,8 @@ double NonlinearForm::GetGridFunctionEnergy(const Vector &x) const
Vector el_x;
const FiniteElement *fe;
ElementTransformation *T;
Mesh *mesh = fes->GetMesh();
double energy = 0.0;
if (dnfi.Size())
@@ -103,14 +86,81 @@ double NonlinearForm::GetGridFunctionEnergy(const Vector &x) const
if (fnfi.Size())
{
MFEM_ABORT("TODO: add energy contribution from interior face terms");
FaceElementTransformations *tr;
const FiniteElement *fe1, *fe2;
Array<int> vdofs2;
for (int i = 0; i < mesh->GetNumFaces(); i++)
{
tr = mesh->GetInteriorFaceTransformations(i);
if (tr != NULL)
{
fes->GetElementVDofs(tr->Elem1No, vdofs);
fes->GetElementVDofs(tr->Elem2No, vdofs2);
vdofs.Append (vdofs2);
x.GetSubVector(vdofs, el_x);
fe1 = fes->GetFE(tr->Elem1No);
fe2 = fes->GetFE(tr->Elem2No);
for (int k = 0; k < fnfi.Size(); k++)
{
energy += fnfi[k]->GetFaceEnergy(*fe1, *fe2, *tr, el_x);
}
}
}
}
if (bfnfi.Size())
{
MFEM_ABORT("TODO: add energy contribution from boundary face terms");
}
FaceElementTransformations *tr;
const FiniteElement *fe1, *fe2;
// Which boundary attributes need to be processed?
Array<int> bdr_attr_marker(mesh->bdr_attributes.Size() ?
mesh->bdr_attributes.Max() : 0);
bdr_attr_marker = 0;
for (int k = 0; k < bfnfi.Size(); k++)
{
if (bfnfi_marker[k] == NULL)
{
bdr_attr_marker = 1;
break;
}
Array<int> &bdr_marker = *bfnfi_marker[k];
MFEM_ASSERT(bdr_marker.Size() == bdr_attr_marker.Size(),
"invalid boundary marker for boundary face integrator #"
<< k << ", counting from zero");
for (int i = 0; i < bdr_attr_marker.Size(); i++)
{
bdr_attr_marker[i] |= bdr_marker[i];
}
}
for (int i = 0; i < fes -> GetNBE(); i++)
{
const int bdr_attr = mesh->GetBdrAttribute(i);
if (bdr_attr_marker[bdr_attr-1] == 0) { continue; }
tr = mesh->GetBdrFaceTransformations (i);
if (tr != NULL)
{
fes->GetElementVDofs(tr->Elem1No, vdofs);
x.GetSubVector(vdofs, el_x);
fe1 = fes->GetFE(tr->Elem1No);
// The fe2 object is really a dummy and not used on the boundaries,
// but we can't dereference a NULL pointer, and we don't want to
// actually make a fake element.
fe2 = fe1;
for (int k = 0; k < bfnfi.Size(); k++)
{
if (bfnfi_marker[k] &&
(*bfnfi_marker[k])[bdr_attr-1] == 0) { continue; }
energy += bfnfi[k]->GetFaceEnergy(*fe1, *fe2, *tr, el_x);
}
}
}
}
return energy;
}
@@ -128,24 +178,13 @@ const Vector &NonlinearForm::Prolongate(const Vector &x) const
void NonlinearForm::Mult(const Vector &x, Vector &y) const
{
const Vector &px = Prolongate(x);
if (P) { aux2.SetSize(P->Height()); }
// If we are in parallel, ParNonLinearForm::Mult uses the aux2 vector.
// In serial, place the result directly in y.
Vector &py = P ? aux2 : y;
if (ext)
{
ext->Mult(px, py);
return;
}
Array<int> vdofs;
Vector el_x, el_y;
const FiniteElement *fe;
ElementTransformation *T;
Mesh *mesh = fes->GetMesh();
const Vector &px = Prolongate(x);
Vector &py = P ? aux2.SetSize(P->Height()), aux2 : y;
py = 0.0;
@@ -262,11 +301,6 @@ void NonlinearForm::Mult(const Vector &x, Vector &y) const
Operator &NonlinearForm::GetGradient(const Vector &x) const
{
if (ext)
{
MFEM_ABORT("Not yet implemented!");
}
const int skip_zeros = 0;
Array<int> vdofs;
Vector el_x;
@@ -410,8 +444,6 @@ Operator &NonlinearForm::GetGradient(const Vector &x) const
void NonlinearForm::Update()
{
if (ext) { MFEM_ABORT("Not yet implemented!"); }
if (sequence == fes->GetSequence()) { return; }
height = width = fes->GetTrueVSize();
@@ -424,11 +456,6 @@ void NonlinearForm::Update()
cP = dynamic_cast<const SparseMatrix*>(P);
}
void NonlinearForm::Setup()
{
if (ext) { return ext->AssemblePA(); }
}
NonlinearForm::~NonlinearForm()
{
delete cGrad;
@@ -436,7 +463,6 @@ NonlinearForm::~NonlinearForm()
for (int i = 0; i < dnfi.Size(); i++) { delete dnfi[i]; }
for (int i = 0; i < fnfi.Size(); i++) { delete fnfi[i]; }
for (int i = 0; i < bfnfi.Size(); i++) { delete bfnfi[i]; }
delete ext;
}
+2 -22
View File
@@ -14,8 +14,6 @@
#include "../config/config.hpp"
#include "nonlininteg.hpp"
#include "nonlinearform_ext.hpp"
#include "bilinearform.hpp"
#include "gridfunc.hpp"
namespace mfem
@@ -24,13 +22,6 @@ namespace mfem
class NonlinearForm : public Operator
{
protected:
/// The assembly level.
AssemblyLevel assembly;
/** Extension for supporting Partial Assembly (PA) or
Matrix Free assembly (MF). */
NonlinearFormExtension *ext;
/// FE space on which the form lives.
FiniteElementSpace *fes; // not owned
@@ -68,16 +59,11 @@ public:
/** As an Operator, the NonlinearForm has input and output size equal to the
number of true degrees of freedom, i.e. f->GetTrueVSize(). */
NonlinearForm(FiniteElementSpace *f)
: Operator(f->GetTrueVSize()), assembly(AssemblyLevel::NONE),
ext(NULL), fes(f), Grad(NULL), cGrad(NULL),
: Operator(f->GetTrueVSize()), fes(f), Grad(NULL), cGrad(NULL),
sequence(f->GetSequence()), P(f->GetProlongationMatrix()),
cP(dynamic_cast<const SparseMatrix*>(P))
{ }
/// Set the desired assembly level. The default is AssemblyLevel::NONE.
/** This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level);
FiniteElementSpace *FESpace() { return fes; }
const FiniteElementSpace *FESpace() const { return fes; }
@@ -85,9 +71,6 @@ public:
void AddDomainIntegrator(NonlinearFormIntegrator *nlfi)
{ dnfi.Append(nlfi); }
/// Access all integrators added with AddDomainIntegrator().
Array<NonlinearFormIntegrator*> *GetDNFI() { return &dnfi; }
/// Adds new Interior Face Integrator.
void AddInteriorFaceIntegrator(NonlinearFormIntegrator *nlfi)
{ fnfi.Append(nlfi); }
@@ -128,7 +111,7 @@ public:
be fes->GetVSize(). */
double GetGridFunctionEnergy(const Vector &x) const;
/// Compute the enery corresponding to the state @a x.
/// Compute the energy corresponding to the state @a x.
/** In general, @a x may have non-homogeneous essential boundary values.
The state @a x must be a true-dof vector. */
@@ -161,9 +144,6 @@ public:
set again. */
virtual void Update();
/// Setup the NonlinearForm
virtual void Setup();
/// Get the finite element space prolongation matrix
virtual const Operator *GetProlongation() const { return P; }
/// Get the finite element space restriction matrix
-74
View File
@@ -1,74 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
// Implementations of classes FABilinearFormExtension, EABilinearFormExtension,
// PABilinearFormExtension and MFBilinearFormExtension.
#include "nonlinearform.hpp"
namespace mfem
{
NonlinearFormExtension::NonlinearFormExtension(NonlinearForm *form)
: Operator(form->FESpace()->GetTrueVSize()), n(form)
{
// empty
}
PANonlinearFormExtension::PANonlinearFormExtension(NonlinearForm *form):
NonlinearFormExtension(form), fes(*form->FESpace())
{
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
elem_restrict_lex = fes.GetElementRestriction(ordering);
if (elem_restrict_lex)
{
localX.SetSize(elem_restrict_lex->Height(), Device::GetMemoryType());
localY.SetSize(elem_restrict_lex->Height(), Device::GetMemoryType());
localY.UseDevice(true); // ensure 'localY = 0.0' is done on device
}
}
void PANonlinearFormExtension::AssemblePA()
{
Array<NonlinearFormIntegrator*> &integrators = *n->GetDNFI();
const int Ni = integrators.Size();
for (int i = 0; i < Ni; ++i)
{
integrators[i]->AssemblePA(*n->FESpace());
}
}
void PANonlinearFormExtension::Mult(const Vector &x, Vector &y) const
{
Array<NonlinearFormIntegrator*> &integrators = *n->GetDNFI();
const int iSz = integrators.Size();
if (elem_restrict_lex)
{
elem_restrict_lex->Mult(x, localX);
localY = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(localX, localY);
}
elem_restrict_lex->MultTranspose(localY, y);
}
else
{
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(x, y);
}
}
}
}
-44
View File
@@ -1,44 +0,0 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef NONLINEARFORM_EXT_HPP
#define NONLINEARFORM_EXT_HPP
#include "../config/config.hpp"
#include "fespace.hpp"
namespace mfem
{
class NonlinearForm;
class NonlinearFormExtension : public Operator
{
protected:
NonlinearForm *n; ///< Not owned
public:
NonlinearFormExtension(NonlinearForm *form);
virtual void AssemblePA() = 0;
};
/// Data and methods for partially-assembled nonlinear forms
class PANonlinearFormExtension : public NonlinearFormExtension
{
protected:
const FiniteElementSpace &fes; // Not owned
mutable Vector localX, localY;
const Operator *elem_restrict_lex; // Not owned
public:
PANonlinearFormExtension(NonlinearForm*);
void AssemblePA();
void Mult(const Vector &x, Vector &y) const;
};
}
#endif // NONLINEARFORM_EXT_HPP
+8 -140
View File
@@ -10,30 +10,10 @@
// Software Foundation) version 2.1 dated February 1999.
#include "fem.hpp"
#include "../general/forall.hpp"
namespace mfem
{
void NonlinearFormIntegrator::AssemblePA(const FiniteElementSpace&)
{
mfem_error ("NonlinearFormIntegrator::AssemblePA(...)\n"
" is not implemented for this class.");
}
void NonlinearFormIntegrator::AssemblePA(const FiniteElementSpace &,
const FiniteElementSpace &)
{
mfem_error ("NonlinearFormIntegrator::AssemblePA(...)\n"
" is not implemented for this class.");
}
void NonlinearFormIntegrator::AddMultPA(const Vector &, Vector &) const
{
mfem_error ("NonlinearFormIntegrator::AddMultPA(...)\n"
" is not implemented for this class.");
}
void NonlinearFormIntegrator::AssembleElementVector(
const FiniteElement &el, ElementTransformation &Tr,
const Vector &elfun, Vector &elvect)
@@ -75,6 +55,14 @@ double NonlinearFormIntegrator::GetElementEnergy(
return 0.0;
}
double NonlinearFormIntegrator::GetFaceEnergy(
const FiniteElement &el1, const FiniteElement &el2,
FaceElementTransformations &Tr, const Vector &elfun)
{
mfem_error("NonlinearFormIntegrator::GetFaceEnergy"
" is not overloaded!");
return 0.0;
}
void BlockNonlinearFormIntegrator::AssembleElementVector(
const Array<const FiniteElement *> &el,
@@ -693,124 +681,4 @@ void IncompressibleNeoHookeanIntegrator::AssembleElementGrad(
}
const IntegrationRule&
VectorConvectionNLFIntegrator::GetRule(const FiniteElement &fe,
ElementTransformation &T)
{
const int order = 2 * fe.GetOrder() + T.OrderGrad(&fe);
return IntRules.Get(fe.GetGeomType(), order);
}
void VectorConvectionNLFIntegrator::AssembleElementVector(
const FiniteElement &el,
ElementTransformation &T,
const Vector &elfun,
Vector &elvect)
{
const int nd = el.GetDof();
const int dim = el.GetDim();
shape.SetSize(nd);
dshape.SetSize(nd, dim);
elvect.SetSize(nd * dim);
gradEF.SetSize(dim);
EF.UseExternalData(elfun.GetData(), nd, dim);
ELV.UseExternalData(elvect.GetData(), nd, dim);
Vector vec1(dim), vec2(dim);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, T);
ELV = 0.0;
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
T.SetIntPoint(&ip);
el.CalcShape(ip, shape);
el.CalcPhysDShape(T, dshape);
double w = ip.weight * T.Weight();
if (Q) { w *= Q->Eval(T, ip); }
MultAtB(EF, dshape, gradEF);
EF.MultTranspose(shape, vec1);
gradEF.Mult(vec1, vec2);
vec2 *= w;
AddMultVWt(shape, vec2, ELV);
}
}
void VectorConvectionNLFIntegrator::AssembleElementGrad(
const FiniteElement &el,
ElementTransformation &trans,
const Vector &elfun,
DenseMatrix &elmat)
{
int nd = el.GetDof();
int dim = el.GetDim();
shape.SetSize(nd);
dshape.SetSize(nd, dim);
dshapex.SetSize(nd, dim);
elmat.SetSize(nd * dim);
elmat_comp.SetSize(nd);
gradEF.SetSize(dim);
EF.UseExternalData(elfun.GetData(), nd, dim);
double w;
Vector vec1(dim), vec2(dim), vec3(nd);
const IntegrationRule *ir = IntRule;
if (ir == nullptr)
{
int order = 2 * el.GetOrder() + trans.OrderGrad(&el);
ir = &IntRules.Get(el.GetGeomType(), order);
}
elmat = 0.0;
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
trans.SetIntPoint(&ip);
el.CalcShape(ip, shape);
el.CalcDShape(ip, dshape);
Mult(dshape, trans.InverseJacobian(), dshapex);
w = ip.weight;
if (Q)
{
w *= Q->Eval(trans, ip);
}
MultAtB(EF, dshapex, gradEF);
EF.MultTranspose(shape, vec1);
trans.AdjugateJacobian().Mult(vec1, vec2);
vec2 *= w;
dshape.Mult(vec2, vec3);
MultVWt(shape, vec3, elmat_comp);
for (int i = 0; i < dim; i++)
{
elmat.AddMatrix(elmat_comp, i * nd, i * nd);
}
MultVVt(shape, elmat_comp);
w = ip.weight * trans.Weight();
if (Q)
{
w *= Q->Eval(trans, ip);
}
for (int i = 0; i < dim; i++)
{
for (int j = 0; j < dim; j++)
{
elmat.AddMatrix(w * gradEF(i, j), elmat_comp, i * nd, j * nd);
}
}
}
}
}
+6 -57
View File
@@ -15,7 +15,6 @@
#include "../config/config.hpp"
#include "fe.hpp"
#include "coefficient.hpp"
#include "fespace.hpp"
namespace mfem
{
@@ -64,30 +63,16 @@ public:
FaceElementTransformations &Tr,
const Vector &elfun, DenseMatrix &elmat);
/// Compute the local energy
/// Compute the local energy/functional
virtual double GetElementEnergy(const FiniteElement &el,
ElementTransformation &Tr,
const Vector &elfun);
/// Method defining partial assembly.
/** The result of the partial assembly is stored internally so that it can be
used later in the methods AddMultPA(). */
virtual void AssemblePA(const FiniteElementSpace &fes);
/** The result of the partial assembly is stored internally so that it can be
used later in the methods AddMultPA().
Used with BilinearFormIntegrators that have different spaces. */
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
/// Method for partially assembled action.
/** Perform the action of integrator on the input @a x and add the result to
the output @a y. Both @a x and @a y are E-vectors, i.e. they represent
the element-wise discontinuous version of the FE space.
This method can be called only after the method AssemblePA() has been
called. */
virtual void AddMultPA(const Vector &x, Vector &y) const;
/// Compute the face(s) contribution to the energy/functional
virtual double GetFaceEnergy(const FiniteElement &el1,
const FiniteElement &el2,
FaceElementTransformations &Tr,
const Vector &elfun);
virtual ~NonlinearFormIntegrator() { }
};
@@ -306,42 +291,6 @@ public:
const Array2D<DenseMatrix *> &elmats);
};
class VectorConvectionNLFIntegrator : public NonlinearFormIntegrator
{
private:
Coefficient *Q{};
DenseMatrix dshape, dshapex, EF, gradEF, ELV, elmat_comp;
Vector shape;
// PA extension
Vector pa_data;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq;
public:
VectorConvectionNLFIntegrator(Coefficient &q): Q(&q) { }
VectorConvectionNLFIntegrator() = default;
static const IntegrationRule &GetRule(const FiniteElement &fe,
ElementTransformation &T);
virtual void AssembleElementVector(const FiniteElement &el,
ElementTransformation &trans,
const Vector &elfun,
Vector &elvect);
virtual void AssembleElementGrad(const FiniteElement &el,
ElementTransformation &trans,
const Vector &elfun,
DenseMatrix &elmat);
using NonlinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
};
}
#endif
-816
View File
@@ -1,816 +0,0 @@
// Copyright (c) 2019, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../general/forall.hpp"
#include "nonlininteg.hpp"
using namespace std;
namespace mfem
{
void VectorConvectionNLFIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
MFEM_ASSERT(fes.GetOrdering() == Ordering::byNODES,
"PA Only supports Ordering::byNODES!");
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetFE(0);
ElementTransformation &T = *mesh->GetElementTransformation(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, T);
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
pa_data.SetSize(ne * nq * dim * dim, Device::GetMemoryType());
double COEFF = 1.0;
if (Q)
{
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient *>(Q);
MFEM_VERIFY(cQ != NULL, "only ConstantCoefficient is supported!");
COEFF = cQ->constant;
}
const int NE = ne;
const int NQ = nq;
auto W = ir->GetWeights().Read();
if (dim == 1)
{
MFEM_ABORT("dim==1 not supported!");
}
if (dim == 2)
{
auto J = Reshape(geom->J.Read(), NQ, 2, 2, NE);
auto G = Reshape(pa_data.Write(), NQ, 2, 2, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q, 0, 0, e);
const double J12 = J(q, 0, 1, e);
const double J21 = J(q, 1, 0, e);
const double J22 = J(q, 1, 1, e);
// Store wq * Q * adj(J)
G(q, 0, 0, e) = W[q] * COEFF * J22; // 1,1
G(q, 0, 1, e) = W[q] * COEFF * -J12; // 1,2
G(q, 1, 0, e) = W[q] * COEFF * -J21; // 2,1
G(q, 1, 1, e) = W[q] * COEFF * J11; // 2,2
}
});
}
if (dim == 3)
{
auto J = Reshape(geom->J.Read(), NQ, 3, 3, NE);
auto G = Reshape(pa_data.Write(), NQ, 3, 3, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q, 0, 0, e);
const double J21 = J(q, 1, 0, e);
const double J31 = J(q, 2, 0, e);
const double J12 = J(q, 0, 1, e);
const double J22 = J(q, 1, 1, e);
const double J32 = J(q, 2, 1, e);
const double J13 = J(q, 0, 2, e);
const double J23 = J(q, 1, 2, e);
const double J33 = J(q, 2, 2, e);
const double cw = W[q] * COEFF;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// Store wq * Q * adj(J)
G(q, 0, 0, e) = cw * A11; // 1,1
G(q, 0, 1, e) = cw * A12; // 1,2
G(q, 0, 2, e) = cw * A13; // 1,3
G(q, 1, 0, e) = cw * A21; // 2,1
G(q, 1, 1, e) = cw * A22; // 2,2
G(q, 1, 2, e) = cw * A23; // 2,3
G(q, 2, 0, e) = cw * A31; // 3,1
G(q, 2, 1, e) = cw * A32; // 3,2
G(q, 2, 2, e) = cw * A33; // 3,3
}
});
}
}
// PA Convection NL 2D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void PAConvectionNLApply2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Vector &q_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Q = Reshape(q_.Read(), Q1D * Q1D, 2, 2, NE);
auto x = Reshape(x_.Read(), D1D, D1D, 2, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, 2, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double data[max_Q1D][max_Q1D][2];
double grad0[max_Q1D][max_Q1D][2];
double grad1[max_Q1D][max_Q1D][2];
double Z[max_Q1D][max_Q1D][2];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
data[qy][qx][0] = 0.0;
data[qy][qx][1] = 0.0;
grad0[qy][qx][0] = 0.0;
grad0[qy][qx][1] = 0.0;
grad1[qy][qx][0] = 0.0;
grad1[qy][qx][1] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double dataX[max_Q1D][2];
double gradX0[max_Q1D][2];
double gradX1[max_Q1D][2];
for (int qx = 0; qx < Q1D; ++qx)
{
dataX[qx][0] = 0.0;
dataX[qx][1] = 0.0;
gradX0[qx][0] = 0.0;
gradX0[qx][1] = 0.0;
gradX1[qx][0] = 0.0;
gradX1[qx][1] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s0 = x(dx, dy, 0, e);
const double s1 = x(dx, dy, 1, e);
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B(qx, dx);
const double Gx = G(qx, dx);
dataX[qx][0] += s0 * Bx;
dataX[qx][1] += s1 * Bx;
gradX0[qx][0] += s0 * Gx;
gradX0[qx][1] += s0 * Bx;
gradX1[qx][0] += s1 * Gx;
gradX1[qx][1] += s1 * Bx;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double By = B(qy, dy);
const double Gy = G(qy, dy);
for (int qx = 0; qx < Q1D; ++qx)
{
data[qy][qx][0] += dataX[qx][0] * By;
data[qy][qx][1] += dataX[qx][1] * By;
grad0[qy][qx][0] += gradX0[qx][0] * By;
grad0[qy][qx][1] += gradX0[qx][1] * Gy;
grad1[qy][qx][0] += gradX1[qx][0] * By;
grad1[qy][qx][1] += gradX1[qx][1] * Gy;
}
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + qy * Q1D;
const double u1 = data[qy][qx][0];
const double u2 = data[qy][qx][1];
const double grad00 = grad0[qy][qx][0];
const double grad01 = grad0[qy][qx][1];
const double grad10 = grad1[qy][qx][0];
const double grad11 = grad1[qy][qx][1];
const double Dxu1 = grad00 * Q(q, 0, 0, e) + grad01 * Q(q, 1, 0, e);
const double Dyu1 = grad00 * Q(q, 0, 1, e) + grad01 * Q(q, 1, 1, e);
const double Dxu2 = grad10 * Q(q, 0, 0, e) + grad11 * Q(q, 1, 0, e);
const double Dyu2 = grad10 * Q(q, 0, 1, e) + grad11 * Q(q, 1, 1, e);
Z[qy][qx][0] = u1 * Dxu1 + u2 * Dyu1;
Z[qy][qx][1] = u1 * Dxu2 + u2 * Dyu2;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double Y[max_D1D][2];
for (int dx = 0; dx < D1D; ++dx)
{
Y[dx][0] = 0.0;
Y[dx][1] = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const double Btx = Bt(dx, qx);
Y[dx][0] += Btx * Z[qy][qx][0];
Y[dx][1] += Btx * Z[qy][qx][1];
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
const double Bty = Bt(dy, qy);
y(dx, dy, 0, e) += Bty * Y[dx][0];
y(dx, dy, 1, e) += Bty * Y[dx][1];
}
}
}
});
}
// PA Convection NL 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void PAConvectionNLApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Vector &q_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Q = Reshape(q_.Read(), Q1D * Q1D * Q1D, VDIM, VDIM, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
MFEM_FORALL(e, NE,
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double data[max_Q1D][max_Q1D][max_Q1D][VDIM];
double grad0[max_Q1D][max_Q1D][max_Q1D][VDIM];
double grad1[max_Q1D][max_Q1D][max_Q1D][VDIM];
double grad2[max_Q1D][max_Q1D][max_Q1D][VDIM];
double Z[max_Q1D][max_Q1D][max_Q1D][VDIM];
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
data[qz][qy][qx][0] = 0.0;
data[qz][qy][qx][1] = 0.0;
data[qz][qy][qx][2] = 0.0;
grad0[qz][qy][qx][0] = 0.0;
grad0[qz][qy][qx][1] = 0.0;
grad0[qz][qy][qx][2] = 0.0;
grad1[qz][qy][qx][0] = 0.0;
grad1[qz][qy][qx][1] = 0.0;
grad1[qz][qy][qx][2] = 0.0;
grad2[qz][qy][qx][0] = 0.0;
grad2[qz][qy][qx][1] = 0.0;
grad2[qz][qy][qx][2] = 0.0;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
double dataXY[max_Q1D][max_Q1D][VDIM];
double gradXY0[max_Q1D][max_Q1D][VDIM];
double gradXY1[max_Q1D][max_Q1D][VDIM];
double gradXY2[max_Q1D][max_Q1D][VDIM];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
dataXY[qy][qx][0] = 0.0;
dataXY[qy][qx][1] = 0.0;
dataXY[qy][qx][2] = 0.0;
gradXY0[qy][qx][0] = 0.0;
gradXY0[qy][qx][1] = 0.0;
gradXY0[qy][qx][2] = 0.0;
gradXY1[qy][qx][0] = 0.0;
gradXY1[qy][qx][1] = 0.0;
gradXY1[qy][qx][2] = 0.0;
gradXY2[qy][qx][0] = 0.0;
gradXY2[qy][qx][1] = 0.0;
gradXY2[qy][qx][2] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double dataX[max_Q1D][VDIM];
double gradX0[max_Q1D][VDIM];
double gradX1[max_Q1D][VDIM];
double gradX2[max_Q1D][VDIM];
for (int qx = 0; qx < Q1D; ++qx)
{
dataX[qx][0] = 0.0;
dataX[qx][1] = 0.0;
dataX[qx][2] = 0.0;
gradX0[qx][0] = 0.0;
gradX0[qx][1] = 0.0;
gradX0[qx][2] = 0.0;
gradX1[qx][0] = 0.0;
gradX1[qx][1] = 0.0;
gradX1[qx][2] = 0.0;
gradX2[qx][0] = 0.0;
gradX2[qx][1] = 0.0;
gradX2[qx][2] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s0 = x(dx, dy, dz, 0, e);
const double s1 = x(dx, dy, dz, 1, e);
const double s2 = x(dx, dy, dz, 2, e);
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B(qx, dx);
const double Gx = G(qx, dx);
dataX[qx][0] += s0 * Bx;
dataX[qx][1] += s1 * Bx;
dataX[qx][2] += s2 * Bx;
gradX0[qx][0] += s0 * Gx;
gradX0[qx][1] += s0 * Bx;
gradX0[qx][2] += s0 * Bx;
gradX1[qx][0] += s1 * Gx;
gradX1[qx][1] += s1 * Bx;
gradX1[qx][2] += s1 * Bx;
gradX2[qx][0] += s2 * Gx;
gradX2[qx][1] += s2 * Bx;
gradX2[qx][2] += s2 * Bx;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double By = B(qy, dy);
const double Gy = G(qy, dy);
for (int qx = 0; qx < Q1D; ++qx)
{
dataXY[qy][qx][0] += dataX[qx][0] * By;
dataXY[qy][qx][1] += dataX[qx][1] * By;
dataXY[qy][qx][2] += dataX[qx][2] * By;
gradXY0[qy][qx][0] += gradX0[qx][0] * By;
gradXY0[qy][qx][1] += gradX0[qx][1] * Gy;
gradXY0[qy][qx][2] += gradX0[qx][2] * By;
gradXY1[qy][qx][0] += gradX1[qx][0] * By;
gradXY1[qy][qx][1] += gradX1[qx][1] * Gy;
gradXY1[qy][qx][2] += gradX1[qx][2] * By;
gradXY2[qy][qx][0] += gradX2[qx][0] * By;
gradXY2[qy][qx][1] += gradX2[qx][1] * Gy;
gradXY2[qy][qx][2] += gradX2[qx][2] * By;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const double Bz = B(qz, dz);
const double Gz = G(qz, dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
data[qz][qy][qx][0] += dataXY[qy][qx][0] * Bz;
data[qz][qy][qx][1] += dataXY[qy][qx][1] * Bz;
data[qz][qy][qx][2] += dataXY[qy][qx][2] * Bz;
grad0[qz][qy][qx][0] += gradXY0[qy][qx][0] * Bz;
grad0[qz][qy][qx][1] += gradXY0[qy][qx][1] * Bz;
grad0[qz][qy][qx][2] += gradXY0[qy][qx][2] * Gz;
grad1[qz][qy][qx][0] += gradXY1[qy][qx][0] * Bz;
grad1[qz][qy][qx][1] += gradXY1[qy][qx][1] * Bz;
grad1[qz][qy][qx][2] += gradXY1[qy][qx][2] * Gz;
grad2[qz][qy][qx][0] += gradXY2[qy][qx][0] * Bz;
grad2[qz][qy][qx][1] += gradXY2[qy][qx][1] * Bz;
grad2[qz][qy][qx][2] += gradXY2[qy][qx][2] * Gz;
}
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + Q1D * (qy + qz * Q1D);
const double u1 = data[qz][qy][qx][0];
const double u2 = data[qz][qy][qx][1];
const double u3 = data[qz][qy][qx][2];
const double grad00 = grad0[qz][qy][qx][0];
const double grad01 = grad0[qz][qy][qx][1];
const double grad02 = grad0[qz][qy][qx][2];
const double grad10 = grad1[qz][qy][qx][0];
const double grad11 = grad1[qz][qy][qx][1];
const double grad12 = grad1[qz][qy][qx][2];
const double grad20 = grad2[qz][qy][qx][0];
const double grad21 = grad2[qz][qy][qx][1];
const double grad22 = grad2[qz][qy][qx][2];
const double Dxu1 = grad00 * Q(q, 0, 0, e)
+ grad01 * Q(q, 1, 0, e)
+ grad02 * Q(q, 2, 0, e);
const double Dyu1 = grad00 * Q(q, 0, 1, e)
+ grad01 * Q(q, 1, 1, e)
+ grad02 * Q(q, 2, 1, e);
const double Dzu1 = grad00 * Q(q, 0, 2, e)
+ grad01 * Q(q, 1, 2, e)
+ grad02 * Q(q, 2, 2, e);
const double Dxu2 = grad10 * Q(q, 0, 0, e)
+ grad11 * Q(q, 1, 0, e)
+ grad12 * Q(q, 2, 0, e);
const double Dyu2 = grad10 * Q(q, 0, 1, e)
+ grad11 * Q(q, 1, 1, e)
+ grad12 * Q(q, 2, 1, e);
const double Dzu2 = grad10 * Q(q, 0, 2, e)
+ grad11 * Q(q, 1, 2, e)
+ grad12 * Q(q, 2, 2, e);
const double Dxu3 = grad20 * Q(q, 0, 0, e)
+ grad21 * Q(q, 1, 0, e)
+ grad22 * Q(q, 2, 0, e);
const double Dyu3 = grad20 * Q(q, 0, 1, e)
+ grad21 * Q(q, 1, 1, e)
+ grad22 * Q(q, 2, 1, e);
const double Dzu3 = grad20 * Q(q, 0, 2, e)
+ grad21 * Q(q, 1, 2, e)
+ grad22 * Q(q, 2, 2, e);
Z[qz][qy][qx][0] = u1 * Dxu1 + u2 * Dyu1 + u3 * Dzu1;
Z[qz][qy][qx][1] = u1 * Dxu2 + u2 * Dyu2 + u3 * Dzu2;
Z[qz][qy][qx][2] = u1 * Dxu3 + u2 * Dyu3 + u3 * Dzu3;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
double opXY[max_D1D][max_D1D][VDIM];
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
opXY[dy][dx][0] = 0.0;
opXY[dy][dx][1] = 0.0;
opXY[dy][dx][2] = 0.0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double opX[max_D1D][VDIM];
for (int dx = 0; dx < D1D; ++dx)
{
opX[dx][0] = 0.0;
opX[dx][1] = 0.0;
opX[dx][2] = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const double Btx = Bt(dx, qx);
opX[dx][0] += Btx * Z[qz][qy][qx][0];
opX[dx][1] += Btx * Z[qz][qy][qx][1];
opX[dx][2] += Btx * Z[qz][qy][qx][2];
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
const double Bty = Bt(dy, qy);
opXY[dy][dx][0] += Bty * opX[dx][0];
opXY[dy][dx][1] += Bty * opX[dx][1];
opXY[dy][dx][2] += Bty * opX[dx][2];
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
const double Btz = Bt(dz, qz);
y(dx, dy, dz, 0, e) += Btz * opXY[dy][dx][0];
y(dx, dy, dz, 1, e) += Btz * opXY[dy][dx][1];
y(dx, dy, dz, 2, e) += Btz * opXY[dy][dx][2];
}
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, int T_MAX_D1D =0, int T_MAX_Q1D =0>
static void SmemPAConvectionNLApply3D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX_Q1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D * Q1D * Q1D, VDIM, VDIM, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX_Q1D;
MFEM_SHARED double BG[2][MQ1 * MD1];
double(*B)[MD1] = (double(*)[MD1])(BG + 0);
double(*G)[MD1] = (double(*)[MD1])(BG + 1);
double(*Bt)[MQ1] = (double(*)[MQ1])(BG + 0);
MFEM_SHARED double U[2][MQ1][MQ1][MQ1];
MFEM_SHARED double sm0[3][MQ1 * MQ1 * MQ1];
MFEM_SHARED double sm1[3][MQ1 * MQ1 * MQ1];
double(*DDQ0)[MD1][MQ1] = (double(*)[MD1][MQ1])(sm0 + 0);
double(*DDQ1)[MD1][MQ1] = (double(*)[MD1][MQ1])(sm0 + 1);
double(*X)[MD1][MD1] = (double(*)[MD1][MD1])(sm0 + 2);
double(*DQQ0)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm1 + 0);
double(*DQQ1)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm1 + 1);
double(*DQQ2)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm1 + 2);
double(*QQQ0)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm0 + 0);
double(*QQQ1)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm0 + 1);
double(*QQQ2)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm0 + 2);
double(*QQD0)[MQ1][MD1] = (double(*)[MQ1][MD1])(sm1 + 0);
double(*QDD0)[MD1][MD1] = (double(*)[MD1][MD1])(sm0 + 0);
MFEM_SHARED double Z[MQ1][MQ1][MQ1];
for (int cy = 0; cy < VDIM; ++cy)
{
if (tidz == 0)
{
MFEM_FOREACH_THREAD(q, x, Q1D)
{
MFEM_FOREACH_THREAD(d, y, D1D)
{
B[q][d] = b(q, d);
G[q][d] = g(q, d);
}
}
}
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D) { Z[qz][qy][qx] = 0.0; }
}
}
MFEM_SYNC_THREAD;
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
X[dz][dy][dx] = x(dx, dy, dz, cy, e);
U[0][dz][dy][dx] = x(dx, dy, dz, c, e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
double u = 0.0;
double v = 0.0;
double z = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double coord = X[dz][dy][dx];
const double value = U[0][dz][dy][dx];
u += coord * B[qx][dx];
v += coord * G[qx][dx];
z += value * B[qx][dx];
}
DDQ0[dz][dy][qx] = u;
DDQ1[dz][dy][qx] = v;
U[1][dz][dy][qx] = z;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
double z = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u += DDQ1[dz][dy][qx] * B[qy][dy];
v += DDQ0[dz][dy][qx] * G[qy][dy];
w += DDQ0[dz][dy][qx] * B[qy][dy];
z += U[1][dz][dy][qx] * B[qy][dy];
}
DQQ0[dz][qy][qx] = u;
DQQ1[dz][qy][qx] = v;
DQQ2[dz][qy][qx] = w;
U[0][dz][qy][qx] = z;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
double z = 0.0;
for (int dz = 0; dz < D1D; ++dz)
{
u += DQQ0[dz][qy][qx] * B[qz][dz];
v += DQQ1[dz][qy][qx] * B[qz][dz];
w += DQQ2[dz][qy][qx] * G[qz][dz];
z += U[0][dz][qy][qx] * B[qz][dz];
}
QQQ0[qz][qy][qx] = u;
QQQ1[qz][qy][qx] = v;
QQQ2[qz][qy][qx] = w;
U[1][qz][qy][qx] = z;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const double z = U[1][qz][qy][qx];
const double gX = QQQ0[qz][qy][qx];
const double gY = QQQ1[qz][qy][qx];
const double gZ = QQQ2[qz][qy][qx];
const double d = gX * D(q, 0, c, e) + gY * D(q, 1, c, e)
+ gZ * D(q, 2, c, e);
Z[qz][qy][qx] += z * d;
}
}
}
MFEM_SYNC_THREAD;
} // for each conv component
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d, y, D1D)
{
MFEM_FOREACH_THREAD(q, x, Q1D) { Bt[d][q] = b(q, d); }
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
double u = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
u += Z[qz][qy][qx] * Bt[dx][qx];
}
QQD0[qz][qy][dx] = u;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
double u = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += QQD0[qz][qy][dx] * Bt[dy][qy];
}
QDD0[qz][dy][dx] = u;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
double u = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
u += QDD0[qz][dy][dx] * Bt[dz][qz];
}
Y(dx, dy, dz, cy, e) += u;
}
}
}
MFEM_SYNC_THREAD;
}
});
}
void VectorConvectionNLFIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
const int NE = ne;
const int D1D = maps->ndof;
const int Q1D = maps->nqpt;
const Vector &Q = pa_data;
const Array<double> &B = maps->B;
const Array<double> &G = maps->G;
const Array<double> &Bt = maps->Bt;
if (dim == 2)
{
return PAConvectionNLApply2D(NE, B, G, Bt, Q, x, y, D1D, Q1D);
}
if (dim == 3)
{
constexpr int T_MAX_D1D = 8;
constexpr int T_MAX_Q1D = 8;
MFEM_VERIFY(D1D <= T_MAX_D1D && Q1D <= T_MAX_Q1D, "Not yet implemented!");
return SmemPAConvectionNLApply3D<0, 0, T_MAX_D1D, T_MAX_Q1D>
(NE, B, G, Q, x, y, D1D, Q1D);
}
MFEM_ABORT("Not yet implemented!");
}
} // namespace mfem
+67 -71
View File
@@ -98,6 +98,8 @@ void ParFiniteElementSpace::ParInit(ParMesh *pm)
gcomm = NULL;
gfdofs = NULL;
P = NULL;
Pconf = NULL;
R = NULL;
@@ -149,6 +151,7 @@ void ParFiniteElementSpace::Construct()
ConstructTrueDofs();
ngedofs = ngfdofs = 0;
gfdofs = NULL;
// calculate number of ghost DOFs
ngvdofs = pncmesh->GetNGhostVertices()
@@ -162,8 +165,22 @@ void ParFiniteElementSpace::Construct()
if (pmesh->Dimension() > 2)
{
int stride = fec->DofForGeometry(Geometry::SQUARE);
ngfdofs = pncmesh->GetNGhostFaces() * stride;
if (fdofs != NULL) // have mixed faces
{
gfdofs = new int[pncmesh->GetNGhostFaces()+1];
gfdofs[0] = 0;
for (int i = 0; i < pncmesh->GetNGhostFaces(); i++)
{
int ghost = pncmesh->GetNFaces() + i;
ngfdofs += fec->DofForGeometry(pncmesh->GetFaceGeometry(ghost));
gfdofs[i+1] = ngfdofs;
}
}
else
{
ngfdofs = pncmesh->GetNGhostFaces()
* fec->DofForGeometry(pncmesh->GetFaceGeometry(0));
}
}
// total number of ghost DOFs. Ghost DOFs start at index 'ndofs', i.e.,
@@ -746,7 +763,6 @@ void ParFiniteElementSpace::GetEssentialTrueDofs(const Array<int>
GetEssentialVDofs(bdr_attr_is_ess, ess_dofs, component);
GetRestrictionMatrix()->BooleanMult(ess_dofs, true_ess_dofs);
#ifdef MFEM_DEBUG
// Verify that in boolean arithmetic: P^T ess_dofs = R ess_dofs.
Array<int> true_ess_dofs2(true_ess_dofs.Size());
@@ -762,7 +778,6 @@ void ParFiniteElementSpace::GetEssentialTrueDofs(const Array<int>
}
MFEM_VERIFY(counter == 0, "internal MFEM error: counter = " << counter);
#endif
MarkerToList(true_ess_dofs, ess_tdof_list);
}
@@ -770,7 +785,7 @@ int ParFiniteElementSpace::GetLocalTDofNumber(int ldof) const
{
if (Nonconforming())
{
Dof_TrueDof_Matrix(); // make sure P has been built
Dof_TrueDof_Matrix(); // inline method
return ldof_ltdof[ldof]; // NOTE: contains -1 for slaves/DOFs we don't own
}
@@ -861,13 +876,7 @@ const Operator *ParFiniteElementSpace::GetProlongationMatrix() const
{
if (Conforming())
{
if (Pconf) { return Pconf; }
if (NRanks == 1)
{
Pconf = new IdentityOperator(GetTrueVSize());
}
else
if (!Pconf)
{
if (!Device::Allows(Backend::DEVICE_MASK))
{
@@ -875,7 +884,10 @@ const Operator *ParFiniteElementSpace::GetProlongationMatrix() const
}
else
{
Pconf = new DeviceConformingProlongationOperator(*this);
if (NRanks > 1)
{
Pconf = new DeviceConformingProlongationOperator(*this);
}
}
}
return Pconf;
@@ -1336,9 +1348,8 @@ void ParFiniteElementSpace::GetGhostFaceDofs(const MeshId &face_id,
int nv = fec->DofForGeometry(Geometry::POINT);
int ne = fec->DofForGeometry(Geometry::SEGMENT);
int nf_tri = fec->DofForGeometry(Geometry::TRIANGLE);
int nf_quad = fec->DofForGeometry(Geometry::SQUARE);
int nf = (nfv == 3) ? nf_tri : nf_quad;
int nf = fec->DofForGeometry((nfv == 3) ?
Geometry::TRIANGLE : Geometry::SQUARE);
dofs.SetSize(nfv*(nv + ne) + nf);
@@ -1367,7 +1378,8 @@ void ParFiniteElementSpace::GetGhostFaceDofs(const MeshId &face_id,
}
const int ghost_face_index = face_id.index - pncmesh->GetNFaces();
int first = ndofs + ngvdofs + ngedofs + nf_quad*ghost_face_index;
int first = ndofs + ngvdofs + ngedofs;
first += gfdofs ? gfdofs[ghost_face_index] : nf*ghost_face_index;
for (int j = 0; j < nf; j++)
{
@@ -1410,11 +1422,7 @@ void ParFiniteElementSpace::GetBareDofs(int entity, int index,
break;
default:
Geometry::Type geom = pncmesh->GetFaceGeometry(index);
MFEM_ASSERT(geom == Geometry::SQUARE ||
geom == Geometry::TRIANGLE, "");
ned = fec->DofForGeometry(geom);
ned = fec->DofForGeometry(pncmesh->GetFaceGeometry(index));
ghost = pncmesh->GetNFaces();
if (index < ghost) // regular face
@@ -1424,8 +1432,8 @@ void ParFiniteElementSpace::GetBareDofs(int entity, int index,
else // ghost face
{
index -= ghost;
int stride = fec->DofForGeometry(Geometry::SQUARE);
first = ndofs + ngvdofs + ngedofs + index*stride;
first = ndofs + ngvdofs + ngedofs +
(gfdofs ? gfdofs[index] : index*ned);
}
break;
}
@@ -1472,8 +1480,8 @@ int ParFiniteElementSpace::PackDof(int entity, int index, int edof) const
else // ghost face
{
index -= ghost;
int stride = fec->DofForGeometry(Geometry::SQUARE);
return ndofs + ngvdofs + ngedofs + index*stride + edof;
return ndofs + ngvdofs + ngedofs +
(gfdofs ? gfdofs[index] : index*ned) + edof;
}
}
}
@@ -1545,8 +1553,16 @@ void ParFiniteElementSpace::UnpackDof(int dof,
dof -= ngedofs;
if (dof < ngfdofs) // ghost face
{
int stride = fec->DofForGeometry(Geometry::SQUARE);
index = pncmesh->GetNFaces() + dof / stride, edof = dof % stride;
if (gfdofs) // have mixed faces
{
index = bisect(gfdofs+1, pncmesh->GetNGhostFaces(), dof);
edof = dof - gfdofs[index];
}
else // uniform faces
{
int nf = fec->DofForGeometry(pncmesh->GetFaceGeometry(0));
index = pncmesh->GetNFaces() + dof / nf, edof = dof % nf;
}
entity = 2;
return;
}
@@ -1995,7 +2011,7 @@ int ParFiniteElementSpace
const NCMesh::Slave &sf = list.slaves[si];
if (pncmesh->IsGhost(entity, sf.index)) { continue; }
GetEntityDofs(entity, sf.index, slave_dofs, mf.Geom());
GetEntityDofs(entity, sf.index, slave_dofs);
if (!slave_dofs.Size()) { continue; }
sf.OrientedPointMatrix(T.GetPointMat());
@@ -2479,11 +2495,10 @@ ParFiniteElementSpace::RebalanceMatrix(int old_ndofs,
// create the offdiagonal part of the matrix
HYPRE_Int* i_offd = make_i_array(vsize);
for (int i = 0, pos = 0; i < new_elements.Size(); i++)
for (int i = 0; i < new_elements.Size(); i++)
{
GetElementDofs(new_elements[i], dofs);
const long* old_dofs = &old_remote_dofs[pos];
pos += dofs.Size() * vdim;
const long* old_dofs = &old_remote_dofs[i * dofs.Size() * vdim];
for (int vd = 0; vd < vdim; vd++)
{
@@ -2538,35 +2553,24 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
int nrk = HYPRE_AssumedPartitionCheck() ? 2 : NRanks;
MFEM_VERIFY(Nonconforming(), "Not implemented for conforming meshes.");
MFEM_VERIFY(old_dof_offsets[nrk], "Missing previous (finer) space.");
MFEM_VERIFY(pmesh->GetNumGeometries(pmesh->Dimension()) == 1,
"Not implemented for mixed meshes.");
#if 0 // check no longer seems to work with NC tet refinement
MFEM_VERIFY(old_dof_offsets[nrk], "Missing previous (finer) space.");
MFEM_VERIFY(dof_offsets[nrk] <= old_dof_offsets[nrk],
"Previous space is not finer.");
#endif
// Note to the reader: please make sure you first read
// FiniteElementSpace::RefinementMatrix, then
// FiniteElementSpace::DerefinementMatrix, and only then this function.
// You have been warned! :-)
Mesh::GeometryList elem_geoms(*mesh);
Array<int> dofs, old_dofs, old_vdofs;
Vector row;
ParNCMesh* pncmesh = pmesh->pncmesh;
int ldof[Geometry::NumGeom];
for (int i = 0; i < Geometry::NumGeom; i++)
{
ldof[i] = 0;
}
for (int i = 0; i < elem_geoms.Size(); i++)
{
Geometry::Type geom = elem_geoms[i];
ldof[geom] = fec->FiniteElementForGeometry(geom)->GetDof();
}
Geometry::Type geom = pncmesh->GetElementGeometry(0); // TODO mixed meshes
int ldof = fec->FiniteElementForGeometry(geom)->GetDof();
const CoarseFineTransformations &dtrans = pncmesh->GetDerefinementTransforms();
const Array<int> &old_ranks = pncmesh->GetDerefineOldRanks();
@@ -2603,13 +2607,10 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
}
else if (coarse_rank == MyRank && fine_rank != MyRank)
{
MFEM_ASSERT(emb.parent >= 0, "");
Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
DerefDofMessage &msg = messages[k];
msg.dofs.resize(ldof[geom]*vdim);
msg.dofs.resize(ldof*vdim);
MPI_Irecv(&msg.dofs[0], ldof[geom]*vdim, HYPRE_MPI_INT,
MPI_Irecv(&msg.dofs[0], ldof*vdim, HYPRE_MPI_INT,
fine_rank, 291, MyComm, &msg.request);
}
// TODO: coalesce Isends/Irecvs to the same rank. Typically, on uniform
@@ -2617,18 +2618,14 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
// from MyRank+1
}
DenseTensor localR[Geometry::NumGeom];
for (int i = 0; i < elem_geoms.Size(); i++)
{
GetLocalDerefinementMatrices(elem_geoms[i], localR[elem_geoms[i]]);
}
DenseTensor localR;
GetLocalDerefinementMatrices(geom, localR);
// create the diagonal part of the derefinement matrix
SparseMatrix *diag = new SparseMatrix(ndofs*vdim, old_ndofs*vdim);
Array<char> mark(diag->Height());
mark = 0;
for (int k = 0; k < dtrans.embeddings.Size(); k++)
{
const Embedding &emb = dtrans.embeddings[k];
@@ -2639,8 +2636,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
if (coarse_rank == MyRank && fine_rank == MyRank)
{
Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
DenseMatrix &lR = localR[geom](emb.matrix);
DenseMatrix &lR = localR(emb.matrix);
elem_dof->GetRow(emb.parent, dofs);
old_elem_dof->GetRow(k, old_dofs);
@@ -2652,7 +2648,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
for (int i = 0; i < lR.Height(); i++)
{
if (!std::isfinite(lR(i, 0))) { continue; }
if (lR(i, 0) == infinity()) { continue; }
int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
@@ -2670,7 +2666,8 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
diag->Finalize();
// wait for all sends/receives to complete
for (auto it = messages.begin(); it != messages.end(); ++it)
for (std::map<int, DerefDofMessage>::iterator
it = messages.begin(); it != messages.end(); ++it)
{
MPI_Wait(&it->second.request, MPI_STATUS_IGNORE);
}
@@ -2689,8 +2686,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
if (coarse_rank == MyRank && fine_rank != MyRank)
{
Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
DenseMatrix &lR = localR[geom](emb.matrix);
DenseMatrix &lR = localR(emb.matrix);
elem_dof->GetRow(emb.parent, dofs);
@@ -2699,12 +2695,11 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
for (int vd = 0; vd < vdim; vd++)
{
MFEM_ASSERT(ldof[geom], "");
HYPRE_Int* remote_dofs = &msg.dofs[vd*ldof[geom]];
HYPRE_Int* remote_dofs = &msg.dofs[vd*ldof];
for (int i = 0; i < lR.Height(); i++)
{
if (!std::isfinite(lR(i, 0))) { continue; }
if (lR(i, 0) == infinity()) { continue; }
int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
@@ -2712,8 +2707,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
if (!mark[m])
{
lR.GetRow(i, row);
MFEM_ASSERT(ldof[geom] == row.Size(), "");
for (int j = 0; j < ldof[geom]; j++)
for (int j = 0; j < ldof; j++)
{
if (row[j] == 0.0) { continue; } // NOTE: lR thresholded
int &lcol = col_map[remote_dofs[j]];
@@ -2743,7 +2737,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
// sure cmap is determined and sorted before the offd matrix is created
{
int width = offd->Width();
Array<Pair<HYPRE_Int, int> > reorder(width);
Array<Pair<int, int> > reorder(width);
for (int i = 0; i < width; i++)
{
reorder[i].one = cmap[i];
@@ -2799,6 +2793,8 @@ void ParFiniteElementSpace::Destroy()
delete Pconf; Pconf = NULL;
delete R; R = NULL;
delete [] gfdofs; gfdofs = NULL;
delete gcomm; gcomm = NULL;
num_face_nbr_dofs = -1;
+2 -1
View File
@@ -46,6 +46,7 @@ private:
/// Number of vertex/edge/face/total ghost DOFs (nonconforming case).
int ngvdofs, ngedofs, ngfdofs, ngdofs;
int* gfdofs;
/// The group of each local dof.
Array<int> ldof_group;
@@ -71,7 +72,7 @@ private:
/// The matrix P (interpolation from true dof to dof). Owned.
mutable HypreParMatrix *P;
/// Optimized action-only prolongation operator for conforming meshes. Owned.
mutable Operator *Pconf;
mutable class ConformingProlongationOperator *Pconf;
/// The (block-diagonal) matrix R (restriction of dof to true dof). Owned.
mutable SparseMatrix *R;
+46 -34
View File
@@ -404,27 +404,33 @@ void ParGridFunction::ProjectBdrCoefficient(
{
Array<int> values_counter;
AccumulateAndCountBdrValues(coeff, vcoeff, attr, values_counter);
Vector values(Size());
for (int i = 0; i < values.Size(); i++)
if (pfes->Conforming())
{
values(i) = values_counter[i] ? (*this)(i) : 0.0;
}
// Count the values globally.
GroupCommunicator &gcomm = pfes->GroupComm();
gcomm.Reduce<int>(values_counter, GroupCommunicator::Sum);
// Accumulate the values globally.
gcomm.Reduce<double>(values, GroupCommunicator::Sum);
// Only the values in the master are guaranteed to be correct!
for (int i = 0; i < values.Size(); i++)
{
if (values_counter[i])
Vector values(Size());
for (int i = 0; i < values.Size(); i++)
{
(*this)(i) = values(i)/values_counter[i];
values(i) = values_counter[i] ? (*this)(i) : 0.0;
}
// Count the values globally.
GroupCommunicator &gcomm = pfes->GroupComm();
gcomm.Reduce<int>(values_counter, GroupCommunicator::Sum);
// Accumulate the values globally.
gcomm.Reduce<double>(values, GroupCommunicator::Sum);
// Only the values in the master are guaranteed to be correct!
for (int i = 0; i < values.Size(); i++)
{
if (values_counter[i])
{
(*this)(i) = values(i)/values_counter[i];
}
}
}
else
{
// TODO: is this the same as the conforming case (after the merge of
// cut-mesh-groups-dev)?
ComputeMeans(ARITHMETIC, values_counter);
}
#ifdef MFEM_DEBUG
Array<int> ess_vdofs_marker;
pfes->GetEssentialVDofs(attr, ess_vdofs_marker);
@@ -442,27 +448,33 @@ void ParGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient &vcoeff,
{
Array<int> values_counter;
AccumulateAndCountBdrTangentValues(vcoeff, bdr_attr, values_counter);
Vector values(Size());
for (int i = 0; i < values.Size(); i++)
if (pfes->Conforming())
{
values(i) = values_counter[i] ? (*this)(i) : 0.0;
}
// Count the values globally.
GroupCommunicator &gcomm = pfes->GroupComm();
gcomm.Reduce<int>(values_counter, GroupCommunicator::Sum);
// Accumulate the values globally.
gcomm.Reduce<double>(values, GroupCommunicator::Sum);
// Only the values in the master are guaranteed to be correct!
for (int i = 0; i < values.Size(); i++)
{
if (values_counter[i])
Vector values(Size());
for (int i = 0; i < values.Size(); i++)
{
(*this)(i) = values(i)/values_counter[i];
values(i) = values_counter[i] ? (*this)(i) : 0.0;
}
// Count the values globally.
GroupCommunicator &gcomm = pfes->GroupComm();
gcomm.Reduce<int>(values_counter, GroupCommunicator::Sum);
// Accumulate the values globally.
gcomm.Reduce<double>(values, GroupCommunicator::Sum);
// Only the values in the master are guaranteed to be correct!
for (int i = 0; i < values.Size(); i++)
{
if (values_counter[i])
{
(*this)(i) = values(i)/values_counter[i];
}
}
}
else
{
// TODO: is this the same as the conforming case (after the merge of
// cut-mesh-groups-dev)?
ComputeMeans(ARITHMETIC, values_counter);
}
#ifdef MFEM_DEBUG
Array<int> ess_vdofs_marker;
pfes->GetEssentialVDofs(bdr_attr, ess_vdofs_marker);
-9
View File
@@ -45,15 +45,6 @@ public:
/** The pointer @a pf is not owned by the newly constructed object. */
ParLinearForm(ParFiniteElementSpace *pf) : LinearForm(pf) { pfes = pf; }
/// Construct a ParLinearForm using previously allocated array @a data.
/** The ParLinearForm does not assume ownership of @a data which is assumed
to be of size at least `pf->GetVSize()`. Similar to the LinearForm and
Vector constructors for externally allocated array, the pointer @a data
can be NULL. The data array can be replaced later using the method
SetData(). */
ParLinearForm(ParFiniteElementSpace *pf, double *data) :
LinearForm(pf, data), pfes(pf) { }
/** @brief Create a ParLinearForm on the ParFiniteElementSpace @a *pf, using
the same integrators as the ParLinearForm @a *plf.
+3 -4
View File
@@ -46,6 +46,7 @@ double ParNonlinearForm::GetParGridFunctionEnergy(const Vector &x) const
void ParNonlinearForm::Mult(const Vector &x, Vector &y) const
{
NonlinearForm::Mult(x, y); // x --(P)--> aux1 --(A_local)--> aux2
Y.MakeRef(aux2, 0); // aux2 contains A_local.P.x
if (fnfi.Size())
{
@@ -57,7 +58,6 @@ void ParNonlinearForm::Mult(const Vector &x, Vector &y) const
Array<int> vdofs1, vdofs2;
Vector el_x, el_y;
aux1.HostReadWrite();
X.MakeRef(aux1, 0); // aux1 contains P.x
X.ExchangeFaceNbrData();
const int n_shared_faces = pmesh->GetNSharedFaces();
@@ -78,14 +78,13 @@ void ParNonlinearForm::Mult(const Vector &x, Vector &y) const
for (int k = 0; k < fnfi.Size(); k++)
{
fnfi[k]->AssembleFaceVector(*fe1, *fe2, *tr, el_x, el_y);
aux2.AddElementVector(vdofs1, el_y.GetData());
Y.AddElementVector(vdofs1, el_y.GetData());
}
}
}
P->MultTranspose(aux2, y);
P->MultTranspose(Y, y);
y.HostReadWrite();
for (int i = 0; i < ess_tdof_list.Size(); i++)
{
y(ess_tdof_list[i]) = 0.0;

Some files were not shown because too many files have changed in this diff Show More