Compare commits
64
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ef5f729245 | ||
|
|
6fa5a0b096 | ||
|
|
3bd8349909 | ||
|
|
af82ee8560 | ||
|
|
2ac542c720 | ||
|
|
e6f828a5fe | ||
|
|
4d756edd80 | ||
|
|
cf5bd1f5cc | ||
|
|
7173dd2002 | ||
|
|
50182bf440 | ||
|
|
24bfcc5165 | ||
|
|
302f22f297 | ||
|
|
1637fcd933 | ||
|
|
2563506174 | ||
|
|
60640c3f7e | ||
|
|
f221521203 | ||
|
|
1d9e736af6 | ||
|
|
d80dbfd99a | ||
|
|
3f44043e60 | ||
|
|
b218959bca | ||
|
|
aee7bc9d43 | ||
|
|
31cac320d4 | ||
|
|
e7e0fb0a88 | ||
|
|
ba71d13980 | ||
|
|
5c326a5535 | ||
|
|
9457f7e5b6 | ||
|
|
ff030ee970 | ||
|
|
2be9e1f36c | ||
|
|
7671cd9f36 | ||
|
|
f89a633fda | ||
|
|
4ce1cef6b8 | ||
|
|
c667bf3025 | ||
|
|
e1678afe40 | ||
|
|
c0291398ed | ||
|
|
0b4f10d79d | ||
|
|
8dfd0e1547 | ||
|
|
caf239c99a | ||
|
|
44c33aece0 | ||
|
|
ed49856390 | ||
|
|
4fef6ca298 | ||
|
|
d9d809e81c | ||
|
|
51e85ccd84 | ||
|
|
4304159303 | ||
|
|
6276268e52 | ||
|
|
580ae34842 | ||
|
|
97eaf8efbc | ||
|
|
e6621c9b0c | ||
|
|
461246f80e | ||
|
|
a5941ee72f | ||
|
|
c4c2ceab59 | ||
|
|
88b99a1719 | ||
|
|
7aa7b4ee53 | ||
|
|
bcf87fee29 | ||
|
|
2949dc5a46 | ||
|
|
b0dbadd007 | ||
|
|
6ca1f95979 | ||
|
|
39794585c4 | ||
|
|
65a71259f1 | ||
|
|
3f4e8324d4 | ||
|
|
9fca398741 | ||
|
|
a2b8f7a129 | ||
|
|
a367631ce5 | ||
|
|
d7718f5c57 | ||
|
|
fe88c4685d |
@@ -58,7 +58,6 @@ jobs:
|
||||
build-system: [make, cmake]
|
||||
hypre-target: [int32]
|
||||
precision: [fp64]
|
||||
enzyme: [false]
|
||||
exclude:
|
||||
- os: ubuntu-latest
|
||||
build-system: cmake
|
||||
@@ -81,17 +80,15 @@ jobs:
|
||||
codecov: YES
|
||||
- os: ubuntu-latest
|
||||
target: dbg
|
||||
config-opts: "CPPFLAGS+=-Og"
|
||||
config-opts: 'CPPFLAGS+=-Og'
|
||||
- os: macos-latest
|
||||
codecov: NO
|
||||
- os: windows-latest
|
||||
codecov: NO
|
||||
# config-opts: '-G "Ninja Multi-Config"'
|
||||
- os: windows-latest
|
||||
target: opt
|
||||
mpi: par
|
||||
config-opts: "-DBUILD_SHARED_LIBS=ON"
|
||||
# config-opts: '-DBUILD_SHARED_LIBS=ON -G "Ninja Multi-Config"'
|
||||
config-opts: '-DBUILD_SHARED_LIBS=ON'
|
||||
- os: ubuntu-latest
|
||||
target: opt
|
||||
codecov: NO
|
||||
@@ -99,7 +96,7 @@ jobs:
|
||||
build-system: cmake
|
||||
hypre-target: int32
|
||||
precision: fp64
|
||||
config-opts: "-DCMAKE_INSTALL_PREFIX=../cmake-install"
|
||||
config-opts: '-DCMAKE_INSTALL_PREFIX=../cmake-install'
|
||||
# This option can be set to pass additional configuration options to
|
||||
# the MFEM configuration command.
|
||||
# config-opts: '-DCMAKE_VERBOSE_MAKEFILE=ON'
|
||||
@@ -117,24 +114,7 @@ jobs:
|
||||
build-system: make
|
||||
hypre-target: int32
|
||||
precision: fp32
|
||||
- os: macos-latest
|
||||
target: opt
|
||||
codecov: NO
|
||||
mpi: par
|
||||
build-system: make
|
||||
hypre-target: int32
|
||||
precision: fp32
|
||||
- os: macos-latest
|
||||
target: opt
|
||||
codecov: NO
|
||||
mpi: par
|
||||
build-system: make
|
||||
hypre-target: int32
|
||||
precision: fp64
|
||||
enzyme: true
|
||||
config-opts: MFEM_USE_ENZYME=YES ENZYME_DIR=$(brew --prefix enzyme)
|
||||
|
||||
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}-${{ matrix.precision }}${{ matrix.enzyme && '-enzyme' || '' }}
|
||||
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}-${{ matrix.precision }}
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
@@ -144,8 +124,8 @@ jobs:
|
||||
if: matrix.os == 'ubuntu-latest'
|
||||
uses: easimon/maximize-build-space@v8
|
||||
with:
|
||||
overprovision-lvm: "true"
|
||||
remove-android: "true"
|
||||
overprovision-lvm: 'true'
|
||||
remove-android: 'true'
|
||||
|
||||
# Checkout MFEM in "mfem" subdirectory. Final path:
|
||||
# /home/runner/work/mfem/mfem/mfem
|
||||
@@ -157,17 +137,6 @@ jobs:
|
||||
# Fetch the complete history for codecov to access commits ID
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Windows environment - PowerShell [debug]
|
||||
if: matrix.os == 'windows-latest'
|
||||
run: |
|
||||
ls env: | fl
|
||||
|
||||
- name: Windows environment - Bash [debug]
|
||||
if: matrix.os == 'windows-latest'
|
||||
run: |
|
||||
env
|
||||
shell: bash
|
||||
|
||||
- name: Xcode version setup (MacOS)
|
||||
if: matrix.os == 'macos-latest'
|
||||
run: |
|
||||
@@ -185,8 +154,7 @@ jobs:
|
||||
- name: get MPI (Linux)
|
||||
if: matrix.mpi == 'par' && matrix.os == 'ubuntu-latest'
|
||||
run: |
|
||||
sudo apt-get install openmpi-bin libopenmpi-dev
|
||||
export OMPI_MCA_rmaps_base_oversubscribe=1
|
||||
sudo apt-get install mpich libmpich-dev
|
||||
|
||||
- name: get lcov (Linux)
|
||||
if: matrix.codecov == 'YES' && matrix.os == 'ubuntu-latest'
|
||||
@@ -223,7 +191,7 @@ jobs:
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-v2.5
|
||||
key: ${{ runner.os }}-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-v2.5
|
||||
|
||||
- name: get hypre
|
||||
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os != 'windows-latest'
|
||||
@@ -282,18 +250,6 @@ jobs:
|
||||
run: |
|
||||
vcpkg install metis-mfem --triplet=x64-windows-static --overlay-ports=${{ env.MFEM_TOP_DIR }}/config/vcpkg/ports
|
||||
|
||||
# It's usually fine to build the above TPLs with a different compiler.
|
||||
#
|
||||
- name: install Enzyme (macOS w/ Enzyme)
|
||||
if: matrix.enzyme && matrix.os == 'macos-latest'
|
||||
run: |
|
||||
export HOMEBREW_NO_INSTALL_CLEANUP=1
|
||||
brew update
|
||||
brew install llvm@19 enzyme
|
||||
echo "LLVM_PREFIX=$(brew --prefix llvm@19)" >> $GITHUB_ENV
|
||||
echo "OMPI_CC=$(brew --prefix llvm@19)/bin/clang" >> $GITHUB_ENV
|
||||
echo "OMPI_CXX=$(brew --prefix llvm@19)/bin/clang++" >> $GITHUB_ENV
|
||||
|
||||
# MFEM build and test
|
||||
- name: build
|
||||
uses: mfem/github-actions/build-mfem@v2.5
|
||||
|
||||
@@ -45,15 +45,14 @@ jobs:
|
||||
|
||||
- name: Get MPI (Linux)
|
||||
run: |
|
||||
sudo apt-get install openmpi-bin libopenmpi-dev
|
||||
export OMPI_MCA_rmaps_base_oversubscribe=1
|
||||
sudo apt-get install mpich libmpich-dev
|
||||
|
||||
- name: Cache Hypre Install
|
||||
id: hypre-cache
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-v2.5
|
||||
key: ${{ runner.os }}-build-${{ env.HYPRE_TOP_DIR }}-v2.5
|
||||
|
||||
- name: Get Hypre
|
||||
if: steps.hypre-cache.outputs.cache-hit != 'true'
|
||||
|
||||
@@ -49,7 +49,7 @@ jobs:
|
||||
library-only: false
|
||||
config-options:
|
||||
CXX="clang++-18"
|
||||
CXXFLAGS="-g -O1 -std=c++17
|
||||
CXXFLAGS="-g -O1 -std=c++11
|
||||
-fsanitize=address
|
||||
-fno-omit-frame-pointer
|
||||
-fsanitize-address-use-after-scope"
|
||||
|
||||
-10
@@ -80,7 +80,6 @@ examples/sol_p.*
|
||||
examples/sol_r.*
|
||||
examples/sol_i.*
|
||||
examples/ex6p-checkpoint.*
|
||||
examples/order.*
|
||||
examples/ex9.mesh
|
||||
examples/ex9-mesh.*
|
||||
examples/ex9-init.*
|
||||
@@ -201,9 +200,6 @@ examples/superlu/sol.*
|
||||
miniapps/adjoint/cvsRoberts_ASAi_dns
|
||||
miniapps/adjoint/adjoint_advection_diffusion
|
||||
|
||||
miniapps/dfem/dfem-minimal-surface
|
||||
miniapps/dfem/dfem-minimal-surface-output
|
||||
|
||||
miniapps/electromagnetics/volta
|
||||
miniapps/electromagnetics/tesla
|
||||
miniapps/electromagnetics/maxwell
|
||||
@@ -239,8 +235,6 @@ miniapps/meshing/minimal-surface
|
||||
miniapps/meshing/pminimal-surface
|
||||
miniapps/meshing/polar-nc
|
||||
miniapps/meshing/mesh-quality
|
||||
miniapps/meshing/hpref
|
||||
miniapps/meshing/phpref
|
||||
miniapps/meshing/mobius-strip.mesh
|
||||
miniapps/meshing/klein-bottle.mesh
|
||||
miniapps/meshing/toroid-*.mesh
|
||||
@@ -258,10 +252,6 @@ miniapps/meshing/sol.gf
|
||||
miniapps/meshing/optimized*
|
||||
miniapps/meshing/perturbed*
|
||||
miniapps/meshing/polar-nc.mesh
|
||||
miniapps/meshing/mesh.*
|
||||
miniapps/meshing/order.*
|
||||
miniapps/meshing/sol.*
|
||||
miniapps/meshing/refined.mesh
|
||||
|
||||
miniapps/mtop/parheat
|
||||
miniapps/mtop/ParHeat*
|
||||
|
||||
@@ -8,252 +8,159 @@
|
||||
https://mfem.org
|
||||
|
||||
|
||||
Version 4.8.1 (development)
|
||||
Version 4.7.1 (development)
|
||||
===========================
|
||||
|
||||
Starting with this version, MFEM requires a C++17 compiler.
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Introduced dFEM: a new MFEM capability for Automatic Differentiation (AD) of
|
||||
nonlinear finite element operators, based on Enzyme or dual numbers AD at
|
||||
quadrature points. These features are part of the new mfem::future namespace
|
||||
and some of the API can change in the future. See the new dFEM minimal surface
|
||||
miniapp in the miniapps/dfem/ directory for illustration of dFEM's use.
|
||||
|
||||
- Using Enzyme for AD in MFEM is tested with clang v19 and requires clang/LLVM
|
||||
built with plugin support. See INSTALL for more details.
|
||||
|
||||
- In the ParMoonolith integration, added support for variational resampling of
|
||||
H1 vector fields.
|
||||
|
||||
Meshing improvements
|
||||
--------------------
|
||||
- Added support for higher order meshes in Mesh::MakeSimplicial and
|
||||
ParMesh::MakeSimplicial.
|
||||
|
||||
GPU computing
|
||||
-------------
|
||||
- The function Vector::SetSubVector(const Array<int> &, const real_t) now
|
||||
executes on device if either the vector or the array have the device flag
|
||||
set. This is most often used for setting constant essential boundary
|
||||
conditions. A new function Vector::SetSubVectorHost has been added in cases
|
||||
where host execution is always needed (e.g. when the DOFs array is small).
|
||||
|
||||
API changes:
|
||||
-----------
|
||||
- mfem::internal::tensor and mfem::internal::dual have been moved to
|
||||
mfem::future::tensor and mfem::future::dual.
|
||||
|
||||
|
||||
Version 4.8, released on Apr 9, 2025
|
||||
====================================
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Added high-order basis functions on pyramid-shaped elements for all spaces in
|
||||
the de Rham complex based on the paper "Orientation embedded high order shape
|
||||
functions for the exact sequence elements of all shapes" by Fuentes, Keith,
|
||||
Demkowicz and Nagaraj (doi.org/10.1016/j.camwa.2015.04.027). Positive basis
|
||||
functions (Bernstein basis) for H1 and L2 on pyramids were also added.
|
||||
|
||||
- Added support for parallel p- and hp-refinement on quad/hex meshes. For hp, we
|
||||
currently support only isotropic refinement with L2 or H1 spaces. See the new
|
||||
miniapps hpref and phpref in the miniapps/meshing/ directory.
|
||||
|
||||
- Added several improvements for hyperbolic problems:
|
||||
* Assembly of Jacobians in HyperbolicFormIntegrator
|
||||
* Component-wise upwinded flux (ComponentwiseUpwindFlux)
|
||||
* Average fluxes in NumericalFlux (formerly RiemannSolver) and FluxFunction
|
||||
|
||||
- Added convenience methods to class FiniteElementSpace to directly identify all
|
||||
degrees of freedom on the exterior faces of the domain, without referencing
|
||||
boundary element attributes (GetExteriorVDofs and GetExteriorTrueDofs).
|
||||
|
||||
- Refactored ALGOIM cut integration rules. The interface is unified with
|
||||
the interface for moment based cut integration rules.
|
||||
- Altered (Par)GridFunction::Compute*Error functions to ensure they return
|
||||
non-negative values and therefore behave as "norms".
|
||||
|
||||
- SubMesh and ParSubMesh have been extended to work on nonconforming meshes.
|
||||
Extracting volume and exterior surface submeshes are both supported.
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Added NURBS-based H(div) and H(curl) elements in 2D and 3D. Only on single
|
||||
patch meshes. Only implemented for serial computations.
|
||||
|
||||
- LinearFormIntegrator, BilinearFormIntegrator and NonlinearFormIntegrator now
|
||||
inherit from a base class Integrator that centralizes the logic for selecting
|
||||
quadrature rules. This includes a virtual method GetDefaultIntegrationRule,
|
||||
which should be favored over directly defining a default integration rule in
|
||||
the element-level assembly routines. The latter is still possible, by leaving
|
||||
the new virtual method as its default base implementation of returning NULL.
|
||||
- Added support for boundary constraints to the hybridization class.
|
||||
|
||||
- Added NURBS-based H(div) and H(curl) elements in 2D and 3D. Currently only for
|
||||
single patch meshes in serial.
|
||||
- Added support for external boundary submeshes with nonconformal mesh adaptation.
|
||||
|
||||
- Refactored ALGOIM cut integration rules. The interface is unified with the
|
||||
interface for moment based cut integration rules.
|
||||
- Added assembly of Jacobians to `HyperbolicFormIntegrator`.
|
||||
|
||||
- FiniteElementSpace has new methods to directly set the prolongation and
|
||||
restriction operators to user-specified sparse matrices.
|
||||
- Added average fluxes to `NumericalFlux` (formerly `RiemannSolver`)
|
||||
and `FluxFunction`.
|
||||
|
||||
- Added support for H(div) spaces in class QuadratureInterpolator. Currently only
|
||||
- Added component-wise upwinded flux (`ComponentwiseUpwindFlux`).
|
||||
|
||||
- Added support for H(div) spaces in class QuadratureInterpolator. For now, only
|
||||
(vector) VALUES, (vector) PHYSICAL_VALUES, and PHYSICAL_MAGNITUDES evaluations
|
||||
are implemented.
|
||||
|
||||
- Added support for boundary constraints to class Hybridization.
|
||||
|
||||
- Added support for custom interpolation procedure in FindPointsGSLIB.
|
||||
|
||||
- Added GSLIB-based gather-scatter operator.
|
||||
are implemented. [PR #4669]
|
||||
|
||||
Meshing improvements
|
||||
--------------------
|
||||
- Added support for nonuniform anisotropic mesh refinement on serial quad/hex
|
||||
meshes with arbitrary spacing in each direction. This enables in particular
|
||||
3:1 refinement, as demonstrated in the new meshing miniapp ref321.
|
||||
- Added native AD support for numerous TMOP metrics that didn't have first or
|
||||
second derivative implementations.
|
||||
|
||||
- Added capabilities for optimization and adaptation of periodic meshes with
|
||||
TMOP. The internals of TMOP were modified so that the optimization problem is
|
||||
always solved with respect to mesh displacements.
|
||||
- The ExodusII reader now handles pyramid and wedge element types. Mixed meshes
|
||||
are also supported.
|
||||
|
||||
- New convenience methods for manipulating boundary attribute markers in class
|
||||
Mesh: MarkExternalBoundaries, MarkNamedBoundaries, UnmarkInternalBoundaries
|
||||
and UnmarkNamedBoundaries. See Examples 1/1p and 11p for basic usage.
|
||||
- Added support for nonuniform anisotropic (nonconforming) mesh refinement with
|
||||
arbitrary spacing in each direction, for quadrilateral (2D) and hexahedral
|
||||
(3D) meshes. This enables in particular 3:1 refinement, as demonstrated in the
|
||||
new meshing miniapp ref321.
|
||||
|
||||
- Added a new method, GetExteriorFaceMarker, to the serial and parallel mesh
|
||||
classes for identifying faces on the exterior of the mesh irrespective of
|
||||
their presence in the list of "boundary elements".
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added miniapps to demonstrate the H(div) and H(curl) NURBS elements.
|
||||
|
||||
- Added native AD support for computing the derivatives of numerous TMOP metrics
|
||||
that didn't have first or second derivative implementations.
|
||||
- Added an MFEM example for the eikonal equation. This new solver is based on
|
||||
the proximal Galerkin method introduced by Keith and Surowiec.
|
||||
|
||||
- Added ExodusII output capability which can handle in particular pyramid and
|
||||
wedge element types. Mixed meshes are also supported.
|
||||
- Added a new toy miniapp that animates an interesting fidget spiral cone toy.
|
||||
See miniapps/toys/spiral.cpp.
|
||||
|
||||
- Added InverseElementTransformation::InitGuessType::EdgeScan as an alternative
|
||||
initial guess type. This guess type tries solving with multiple initial guesses
|
||||
along the r/s/t=0 edges of the element until a valid solution is found or all
|
||||
initial guess points are exhausted.
|
||||
|
||||
- Added new convenience constructors for NURBS patches and knot vectors.
|
||||
- Added a command line option to all miniapps (`-p` or `--send-port`) for
|
||||
specifying the GLVis server socket port (19916 by default).
|
||||
|
||||
GPU computing
|
||||
-------------
|
||||
- Extended FindPointsGSLIB to support general field interpolation on GPUs. Note
|
||||
that this requires that switch from gslib v1.0.7 to v1.0.9.
|
||||
|
||||
- Added support for GPU-accelerated batched linear algebra (using cuBLAS,
|
||||
hipBLAS, MAGMA, or native MFEM functionality) through the BatchedLinAlg class.
|
||||
|
||||
- A new GPU kernel dispatch mechanism was introduced. Users can instantiate
|
||||
specialized kernels for specific combinations of (for example) polynomial
|
||||
degree and number of quadrature points using method AddSpecialization in
|
||||
classes DiffusionIntegrator and MassIntegrator (this functionality may be
|
||||
added to more integrators in the future).
|
||||
degree and number of quadrature points using
|
||||
`DiffusionIntegrator::AddSpecialization` and
|
||||
`MassIntegrator::AddSpecialization` (this functionality may be added to more
|
||||
integrators in the future).
|
||||
|
||||
- Added BatchInverseElementTransformation to batch InverseElementTransformation
|
||||
searches. Batch searches are currently limited to meshes with a single element
|
||||
geometry type of SEGMENT, SQUARE, or CUBE. Additional element geometry types
|
||||
may be added in the future. Mixed element order meshes are supported. Batch
|
||||
searches can be performed on the CPU (serial per MPI rank) or GPU. Embedded
|
||||
elements (SEGMENT in 2D/3D space or SQUARE in 3D space) are supported, however
|
||||
the existing solvers may struggle to find a valid solution.
|
||||
- Calls to slower fallback kernels can be reported to `mfem::err` by setting
|
||||
the environment variable `MFEM_REPORT_KERNELS` to any value other than `NO`
|
||||
or by explicitly calling `KernelReporter::Enable`. Users can then add
|
||||
specializations for these kernels to achieve higher performance.
|
||||
|
||||
- Calls to slower fallback kernels can be reported to mfem::err by setting the
|
||||
environment variable MFEM_REPORT_KERNELS to any value other than NO or by
|
||||
explicitly calling KernelReporter::Enable. For higher performance, users can
|
||||
then add specializations for these kernels.
|
||||
- Element assembly kernels have been added for low-order refined to
|
||||
high-order transfer operators. New kernels can be offloaded as device
|
||||
kernels. Example usage may be found in lor-transfer.cpp under miniapps/tools.
|
||||
|
||||
- Element assembly kernels have been added for low-order refined -> high-order
|
||||
transfer operators. New kernels can be offloaded as device kernels. Example
|
||||
usage may be found in lor-transfer.cpp in the miniapps/tools/ directory.
|
||||
|
||||
- Added GPU acceleration and element assembly for DivDivIntegrator and
|
||||
NormalTraceJumpIntegrator.
|
||||
- Added support for GPU accelerated FindPointsGSLIB. Note that this will require
|
||||
the users to switch from gslib v1.0.7 to v1.0.9.
|
||||
|
||||
- Allow BlockLowerTriangularPreconditioner to run on GPU.
|
||||
|
||||
- Use device vectors in GMRES, FGMRES and other iterative methods.
|
||||
|
||||
Linear and nonlinear solvers
|
||||
----------------------------
|
||||
- Added GPU acceleration of the algebraic hybridization solver for grad-div
|
||||
problems in H(div). See Example 4.
|
||||
|
||||
- Added a self-contained implementation of the Method of Moving Asymptotes (MMA)
|
||||
for solving optimization problems.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Added support for SUNDIALS v7. See the section "API changes" for some small
|
||||
changes related to this new version.
|
||||
|
||||
- Changed the name of class IterativeSolverMonitor to IterativeSolverController,
|
||||
which now allows for specifying convergence by a user defined criterion. For
|
||||
backward compatibility, the old name is still available.
|
||||
- Refactored the `ARKStepSolver` class (ARKODE interface) to use
|
||||
`TimeDependentOperator::Mult` only when the associated ODE operator is
|
||||
expressed in explicit form (i.e., `TimeDependentOperator::isExplicit()`),
|
||||
otherwise `TimeDependentOperator::ExplicitMult` is used. A check has been
|
||||
added to `ARKStepSolver` to verify that the associated ODE operator is not in
|
||||
explicit form when a mass matrix solver is enabled via a call to either the
|
||||
`UseMFEMMassLinearSolver` or `UseSundialsMassLinearSolver` methods. This is
|
||||
because enabling a mass matrix solver assumes that F(u,k,t) = M k in the
|
||||
associated ODE operator.
|
||||
|
||||
- Refactored the ARKStepSolver class (ARKODE interface) to use the Mult() method
|
||||
of TimeDependentOperator only when the associated ODE operator is expressed in
|
||||
explicit form (i.e., TimeDependentOperator::isExplicit()), otherwise the
|
||||
method ExplicitMult() is used. A check has been added to ARKStepSolver to
|
||||
verify that the associated ODE operator is not in explicit form when a mass
|
||||
matrix solver is enabled via a call to either the UseMFEMMassLinearSolver or
|
||||
UseSundialsMassLinearSolver methods. This is because enabling a mass matrix
|
||||
solver assumes that F(u,k,t) = M k in the associated ODE operator.
|
||||
- Added support for custom interpolation procedure in FindPointsGSLIB.
|
||||
|
||||
- Added ODE solvers selection routines. This creates a uniformity across
|
||||
examples, miniapps and other executables in regard to ODE (time-integrator)
|
||||
selection.
|
||||
- `FiniteElementSpace` has new methods to directly set prolongation and
|
||||
restriction operators to arbitrary sparse matrices.
|
||||
|
||||
- Added new mechanism for retrieving and setting state vectors in ODE solvers.
|
||||
This is relevant for AB/AM and gen-alpha solvers.
|
||||
- There are new convenience constructors for NURBS patches and knot vectors.
|
||||
|
||||
- Added ODEsolver/ODEsolver2 unit tests to verify order of convergence and
|
||||
read/write functionality.
|
||||
- Added convenience methods for manipulating boundary attribute marker arrays;
|
||||
`(Par)Mesh::MarkExternalBoundaries`, `(Par)Mesh::UnmarkInternalBoundaries`,
|
||||
`(Par)Mesh::MarkNamedBoundaries`, and `(Par)Mesh::UnmarkNamedBoundaries`.
|
||||
See examples `ex1.cpp`, `ex1p.cpp`, and `ex11p.cpp` for basic usage.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added an MFEM example for the eikonal equation (examples/ex40) based on the
|
||||
proximal Galerkin method introduced by Keith and Surowiec.
|
||||
- Added `(Par)Mesh::GetExteriorFaceMarker` for identifying faces on the
|
||||
exterior of the mesh irrespective of their presence in the list of "boundary
|
||||
elements".
|
||||
|
||||
- Added miniapps to demonstrate the H(div) and H(curl) NURBS elements.
|
||||
- Added methods to `(Par)FiniteElementSpace` to identify all degrees of freedom
|
||||
located on the exterior of the domain without reference to the list of
|
||||
"boundary elements"; `GetExteriorVDofs` and `GetExteriorTrueDofs`.
|
||||
|
||||
- Fixed element visualization in the Mesh explorer miniapp.
|
||||
- `LinearFormIntegrator` and `NonlinearFormIntegrator` (including
|
||||
`BilinearFormIntegrator`) now all inherit from a base class `Integrator`
|
||||
that combines some logic related to selecting quadrature rules. This includes
|
||||
a virtual method `Integrator::GetDefaultIntegrationRule`, which should be
|
||||
favored over directly defining a default integration rule in the element-level
|
||||
assembly routines (although the latter is still possible, by leaving the new
|
||||
virtual method as its default base implementation of returning `NULL`).
|
||||
|
||||
- Added a command line option to all miniapps (-p or --send-port) for
|
||||
specifying the GLVis server socket port (19916 by default).
|
||||
|
||||
- Added a new toy miniapp that animates an interesting fidget spiral cone toy.
|
||||
See miniapps/toys/spiral.cpp.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Updated the benchmarks (in tests/benchmarks) to work with the latest Google
|
||||
Benchmarks classes (version 1.9.1). Renamed the MFEM_ENABLE_GOOGLE_BENCHMARKS
|
||||
CMake option to just MFEM_ENABLE_BENCHMARKS.
|
||||
|
||||
- Updated the minimum CMake version requirements:
|
||||
* CMake >= 3.12 for CPU builds,
|
||||
* CMake >= 3.17 for CUDA builds, and
|
||||
* CMake >= 3.14 for HIP builds (CMake >= 3.12 may work as well, not tested).
|
||||
|
||||
- Various other simplifications, extensions, and bugfixes in the code.
|
||||
- Changed the name of `IterativeSolverMonitor` to `IterativeSolverController`
|
||||
which now allows for declaring convergence by a user defined criterion. For
|
||||
backward compatibility, the old name is still available.
|
||||
|
||||
API changes
|
||||
-----------
|
||||
- In class GridFunction, fec was renamed to fec_owned.
|
||||
- API change: 'TMOP_Metric_skew2D' has been marked as deprecated.
|
||||
|
||||
- RiemannSolver was renamed to NumericalFlux (the old name has been deprecated
|
||||
through typedef).
|
||||
- API change: in class GridFunction, 'fec' was renamed to 'fec_owned'.
|
||||
|
||||
- API changes due to SUNDIALS v7:
|
||||
* the SUNDIALS types realtype and booleantype are no longer defined by v7
|
||||
and therefore MFEM now uses the new type names sunrealtype and
|
||||
sunbooleantype, respectively, which MFEM defines when using SUNDIALS < v6
|
||||
- API change: `RiemannSolver` was renamed to `NumericalFlux` (the old name has
|
||||
been been deprecated through typedef)
|
||||
|
||||
- API change: support for SUNDIALS v7:
|
||||
* the SUNDIALS types `realtype` and `booleantype` are no longer defined by v7
|
||||
and therefore MFEM now uses the new type names `sunrealtype` and
|
||||
`sunbooleantype`, respectively, which MFEM defines when using SUNDIALS < v6
|
||||
where these types were not defined.
|
||||
* The SUNDIALS macro SUNLS_SUCCESS and some other *_SUCCESS macros were
|
||||
removed and replaced by SUN_SUCCESS in v7, so to avoid tedious checks for
|
||||
SUNDIALS versions, MFEM now defines and uses the constant SUN_SUCCESS when
|
||||
* The SUNDIALS macro `SUNLS_SUCCESS` and some other `*_SUCCESS` macros were
|
||||
removed and replaced by `SUN_SUCCESS` in v7, so to avoid tedious checks for
|
||||
SUNDIALS versions, MFEM now defines and uses the constant `SUN_SUCCESS` when
|
||||
using SUNDIALS < v7.
|
||||
* The constants SUN_PREC_*, introduced by SUNDIALS v6 are now introduced by
|
||||
* The constants `SUN_PREC_*`, introduced by SUNDIALS v6 are now introduced by
|
||||
MFEM when using SUNDIALS < v6 to avoid tedious version checks.
|
||||
|
||||
- TMOP_Metric_skew2D has been marked as deprecated.
|
||||
|
||||
|
||||
Version 4.7, released on May 7, 2024
|
||||
====================================
|
||||
@@ -278,6 +185,9 @@ Meshing improvements
|
||||
|
||||
- Added support for internal boundary elements in nonconforming meshes.
|
||||
|
||||
- Added ExodusII output capability. The writer can handle first-order (Pyramid5,
|
||||
Wedge6, Hex8, Tet4) and second-order FE types (Pyramid14, Wedge18, Hex27, Tet10).
|
||||
|
||||
- The ReadCubit Genesis mesh importer has been rewritten to improve readability.
|
||||
|
||||
Discretization improvements
|
||||
@@ -337,6 +247,15 @@ New and updated examples and miniapps
|
||||
- Added two new example codes: 38 and 39/39p described above. Substantially
|
||||
updated Example 18/18p.
|
||||
|
||||
- Added ODE solvers selection routines. This creates a uniformity across examples,
|
||||
miniapps and other executables in regard to ODE(time-integrator) selection.
|
||||
|
||||
- Added new mechanism for retrieving and setting state vectors in ODE solvers.
|
||||
This is relevant for AB/AM and gen-alpha solvers.
|
||||
|
||||
- Added ODEsolver/ODEsolver2 unit tests to verify order of convergence and
|
||||
read/write functionality.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Updated the Doxygen documentation style, which now requires Doxygen version
|
||||
@@ -350,6 +269,8 @@ Miscellaneous
|
||||
|
||||
- Various other simplifications, extensions, and bugfixes in the code.
|
||||
|
||||
- Added GSLIB-based gather-scatter operator.
|
||||
|
||||
|
||||
Version 4.6, released on September 27, 2023
|
||||
===========================================
|
||||
|
||||
+29
-56
@@ -12,14 +12,13 @@
|
||||
# The variable CMAKE_CXX_STANDARD and related were introduced in CMake v3.1
|
||||
# Version 3.8 fixes the handling of CMAKE_CXX_STANDARD for try_compile.
|
||||
# Version 3.8 or newer is required for direct CUDA support.
|
||||
# Version 3.12 or newer is required for setting maximum policy version.
|
||||
cmake_minimum_required(VERSION 3.12.0...4.0.0)
|
||||
cmake_minimum_required(VERSION 3.8)
|
||||
message(STATUS "CMake version: ${CMAKE_VERSION}")
|
||||
set(USER_CONFIG "${CMAKE_CURRENT_SOURCE_DIR}/config/user.cmake" CACHE PATH
|
||||
"Path to optional user configuration file.")
|
||||
|
||||
# Require C++17 and disable compiler-specific extensions
|
||||
set(CMAKE_CXX_STANDARD 17 CACHE STRING "C++ standard to use.")
|
||||
# Require C++11 and disable compiler-specific extensions
|
||||
set(CMAKE_CXX_STANDARD 11 CACHE STRING "C++ standard to use.")
|
||||
set(CMAKE_CXX_STANDARD_REQUIRED ON CACHE BOOL
|
||||
"Force the use of the chosen C++ standard.")
|
||||
set(CMAKE_CXX_EXTENSIONS OFF CACHE BOOL "Enable C++ standard extensions.")
|
||||
@@ -59,7 +58,7 @@ project(mfem NONE)
|
||||
# Current version of MFEM, see also `makefile`.
|
||||
# mfem_VERSION = (string)
|
||||
# MFEM_VERSION = (int) [automatically derived from mfem_VERSION]
|
||||
set(${PROJECT_NAME}_VERSION 4.8.1)
|
||||
set(${PROJECT_NAME}_VERSION 4.7.1)
|
||||
|
||||
# Prohibit in-source build
|
||||
if (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
|
||||
@@ -105,13 +104,6 @@ endif()
|
||||
# Include xSDK default CMake file.
|
||||
include("${CMAKE_CURRENT_SOURCE_DIR}/config/XSDKDefaults.cmake")
|
||||
|
||||
# Path to MFEM's CMake modules and utilities.
|
||||
set(MFEM_CMAKE_PATH ${PROJECT_SOURCE_DIR}/config)
|
||||
set(CMAKE_MODULE_PATH ${MFEM_CMAKE_PATH}/cmake/modules)
|
||||
|
||||
# Load MFEM CMake utilities.
|
||||
include(MfemCmakeUtilities)
|
||||
|
||||
# Enable languages.
|
||||
enable_language(CXX)
|
||||
if (MINGW)
|
||||
@@ -124,15 +116,17 @@ if (MFEM_USE_CUDA)
|
||||
if (MFEM_USE_HIP)
|
||||
message(FATAL_ERROR " *** MFEM_USE_HIP cannot be combined with MFEM_USE_CUDA.")
|
||||
endif()
|
||||
# CUDAToolkit was added in CMake 3.17, so we require at least CMake 3.17 when
|
||||
# CUDA is enabled:
|
||||
if (CMAKE_VERSION VERSION_LESS 3.17.0)
|
||||
message(FATAL_ERROR "CUDA support requires CMake >= 3.17")
|
||||
endif()
|
||||
# Use ${CMAKE_CXX_COMPILER} as the cuda host compiler.
|
||||
if (NOT CMAKE_CUDA_HOST_COMPILER)
|
||||
set(CMAKE_CUDA_HOST_COMPILER ${CMAKE_CXX_COMPILER})
|
||||
endif()
|
||||
enable_language(CUDA)
|
||||
set(CMAKE_CUDA_STANDARD ${CMAKE_CXX_STANDARD} CACHE STRING
|
||||
"CUDA standard to use.")
|
||||
set(CMAKE_CUDA_STANDARD_REQUIRED ON CACHE BOOL
|
||||
"Force the use of the chosen CUDA standard.")
|
||||
set(CMAKE_CUDA_EXTENSIONS OFF CACHE BOOL "Enable CUDA standard extensions.")
|
||||
set(CUDA_FLAGS "--expt-extended-lambda")
|
||||
if (CMAKE_VERSION VERSION_LESS 3.18.0)
|
||||
set(CUDA_FLAGS "-arch=${CUDA_ARCH} ${CUDA_FLAGS}")
|
||||
elseif (NOT CMAKE_CUDA_ARCHITECTURES)
|
||||
@@ -146,33 +140,18 @@ if (MFEM_USE_CUDA)
|
||||
set(CUDA_ARCH "CMAKE_CUDA_ARCHITECTURES: ${CMAKE_CUDA_ARCHITECTURES}")
|
||||
endif()
|
||||
message(STATUS "Using CUDA architecture: ${CUDA_ARCH}")
|
||||
enable_language(CUDA)
|
||||
if (CMAKE_VERSION VERSION_LESS 3.18.0)
|
||||
# backup try to detect if this is clang or nvcc
|
||||
if(CMAKE_CUDA_COMPILER MATCHES "nvcc$")
|
||||
# nvcc
|
||||
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
|
||||
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
|
||||
endif()
|
||||
else()
|
||||
if (CMAKE_CUDA_COMPILER_ID STREQUAL "NVIDIA")
|
||||
# nvcc
|
||||
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
|
||||
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
|
||||
endif()
|
||||
if (CMAKE_VERSION VERSION_LESS 3.12.0)
|
||||
# CMake versions 3.8 and 3.9 require this to work; 3.10 and 3.11 are not
|
||||
# tested and may not actually need this (but should be ok to keep).
|
||||
set(CUDA_FLAGS "-ccbin=${CMAKE_CXX_COMPILER} ${CUDA_FLAGS}")
|
||||
set(CMAKE_CUDA_HOST_LINK_LAUNCHER ${CMAKE_CXX_COMPILER})
|
||||
endif()
|
||||
set(CMAKE_CUDA_STANDARD ${CMAKE_CXX_STANDARD} CACHE STRING
|
||||
"CUDA standard to use.")
|
||||
set(CMAKE_CUDA_STANDARD_REQUIRED ON CACHE BOOL
|
||||
"Force the use of the chosen CUDA standard.")
|
||||
set(CMAKE_CUDA_EXTENSIONS OFF CACHE BOOL "Enable CUDA standard extensions.")
|
||||
set(CMAKE_CUDA_FLAGS "${CMAKE_CUDA_FLAGS} ${CUDA_FLAGS}")
|
||||
find_package(CUDAToolkit REQUIRED)
|
||||
set(CUSPARSE_FOUND TRUE)
|
||||
set(CUBLAS_FOUND TRUE)
|
||||
# Initialize CUSPARSE_LIBRARIES and CUBLAS_LIBRARIES:
|
||||
mfem_culib_set_libraries(CUSPARSE cusparse)
|
||||
mfem_culib_set_libraries(CUBLAS cublas)
|
||||
get_target_property(CUSPARSE_LIBRARIES CUDA::cusparse LOCATION)
|
||||
get_target_property(CUBLAS_LIBRARIES CUDA::cublas LOCATION)
|
||||
endif()
|
||||
|
||||
if (XSDK_ENABLE_C)
|
||||
@@ -185,6 +164,13 @@ endif()
|
||||
# Suppress warnings about MACOSX_RPATH
|
||||
set(CMAKE_MACOSX_RPATH OFF CACHE BOOL "")
|
||||
|
||||
# CMake needs to know where to find things
|
||||
set(MFEM_CMAKE_PATH ${PROJECT_SOURCE_DIR}/config)
|
||||
set(CMAKE_MODULE_PATH ${MFEM_CMAKE_PATH}/cmake/modules)
|
||||
|
||||
# Load MFEM CMake utilities.
|
||||
include(MfemCmakeUtilities)
|
||||
|
||||
string(TOUPPER "${PROJECT_NAME}" PROJECT_NAME_UC)
|
||||
mfem_version_to_int(${${PROJECT_NAME}_VERSION} ${PROJECT_NAME_UC}_VERSION)
|
||||
set(${PROJECT_NAME_UC}_VERSION_STRING ${${PROJECT_NAME}_VERSION})
|
||||
@@ -269,11 +255,7 @@ if (MFEM_USE_OPENMP OR MFEM_USE_LEGACY_OPENMP)
|
||||
if (OPENMP_FOUND)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} ${OpenMP_CXX_FLAGS}")
|
||||
if (MFEM_USE_CUDA)
|
||||
if(MFEM_CUDA_COMPILER_IS_NVCC)
|
||||
set(CMAKE_CUDA_FLAGS "${CMAKE_CUDA_FLAGS} -Xcompiler=${OpenMP_CXX_FLAGS}")
|
||||
else()
|
||||
set(CMAKE_CUDA_FLAGS "${CMAKE_CUDA_FLAGS} ${OpenMP_CXX_FLAGS}")
|
||||
endif()
|
||||
set(CMAKE_CUDA_FLAGS "${CMAKE_CUDA_FLAGS} -Xcompiler=${OpenMP_CXX_FLAGS}")
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
@@ -405,11 +387,6 @@ if (MFEM_USE_GSLIB)
|
||||
find_package(GSLIB REQUIRED)
|
||||
endif()
|
||||
|
||||
# HDF5
|
||||
if (MFEM_USE_HDF5)
|
||||
find_package(HDF5 REQUIRED)
|
||||
endif()
|
||||
|
||||
# NetCDF
|
||||
if (MFEM_USE_NETCDF)
|
||||
find_package(NetCDF REQUIRED)
|
||||
@@ -549,10 +526,9 @@ if (MFEM_USE_TRIBOL)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Enzyme
|
||||
if (MFEM_USE_ENZYME)
|
||||
find_package(Enzyme REQUIRED HINTS ${ENZYME_DIR})
|
||||
message(STATUS "Enzyme found in ${ENZYME_DIR}.")
|
||||
set(ENZYME_INCLUDE_DIRS ${ENZYME_DIR}/include)
|
||||
find_package(ENZYME REQUIRED)
|
||||
endif()
|
||||
|
||||
# MFEM_TIMER_TYPE
|
||||
@@ -592,7 +568,7 @@ find_package(Threads REQUIRED)
|
||||
# integers, the METIS header (with 32-bit indices, as used by mfem) needs to
|
||||
# be before SuiteSparse.
|
||||
set(MFEM_TPLS OPENMP HYPRE LAPACK BLAS SuperLUDist STRUMPACK METIS SuiteSparse
|
||||
SUNDIALS PETSC SLEPC MUMPS AXOM FMS CONDUIT Ginkgo GNUTLS GSLIB HDF5
|
||||
SUNDIALS PETSC SLEPC MUMPS AXOM FMS CONDUIT Ginkgo GNUTLS GSLIB
|
||||
NETCDF MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE
|
||||
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CALIPER CODIPACK
|
||||
BENCHMARK PARELAG TRIBOL MPI_CXX HIP HIPBLAS HIPSPARSE MOONOLITH BLITZ
|
||||
@@ -704,9 +680,6 @@ if (MFEM_USE_MPI)
|
||||
target_link_libraries(mfem PUBLIC ${MPI_CXX_LINK_FLAGS})
|
||||
endif()
|
||||
endif()
|
||||
if (MFEM_USE_ENZYME)
|
||||
target_link_libraries(mfem PUBLIC ClangEnzymeFlags)
|
||||
endif()
|
||||
|
||||
set_target_properties(mfem PROPERTIES VERSION "${mfem_VERSION}")
|
||||
set_target_properties(mfem PROPERTIES SOVERSION "${mfem_VERSION}")
|
||||
|
||||
+1
-8
@@ -120,10 +120,7 @@ The MFEM source code has the following structure:
|
||||
| └── superlu
|
||||
├── fem
|
||||
│ ├── ceed
|
||||
│ ├── dfem
|
||||
│ ├── eltrans
|
||||
│ ├── fe
|
||||
│ ├── gslib
|
||||
│ ├── integ
|
||||
│ ├── lor
|
||||
│ ├── moonolith
|
||||
@@ -131,7 +128,6 @@ The MFEM source code has the following structure:
|
||||
│ └── tmop
|
||||
├── general
|
||||
├── linalg
|
||||
│ ├── batched
|
||||
│ └── simd
|
||||
├── mesh
|
||||
│ └── submesh
|
||||
@@ -139,7 +135,6 @@ The MFEM source code has the following structure:
|
||||
│ ├── adjoint
|
||||
│ ├── autodiff
|
||||
│ ├── common
|
||||
│ ├── dfem
|
||||
│ ├── dpg
|
||||
│ ├── electromagnetics
|
||||
│ ├── gslib
|
||||
@@ -551,8 +546,6 @@ Before a PR can be merged, it should satisfy the following:
|
||||
- [ ] Add a short description of the example in the "Extensive Examples" section of `features.md`.
|
||||
- [ ] New miniapps:
|
||||
- [ ] All sample runs at the top of the miniapp source file work.
|
||||
- [ ] Add to internal testing repo, if sample runs should be included in nightly tests [internally](#tests-at-llnl).
|
||||
- [ ] Exclude long sample runs from automated testing, with `* ` (one space) before the command.
|
||||
- [ ] Update top-level `makefile` and `makefile` in corresponding miniapp directory.
|
||||
- [ ] Add the miniapp binary and any files generated by it to the top-level `.gitignore` file.
|
||||
- [ ] Update CMake build system:
|
||||
@@ -747,7 +740,7 @@ and debug build is performed with a simple run of `ex1` to verify the executable
|
||||
- We mirror the `master` and `next` branches internally (to `gh-master` and
|
||||
`gh-next`) and run longer nightly tests via cron. On the weekends, a more
|
||||
extensive test is run which extracts and executes all the different sample
|
||||
runs from each example and most miniapps.
|
||||
runs from each example.
|
||||
|
||||
- We also mirror PRs on the LLNL GitLab instance. PR mirroring can only be
|
||||
triggered by _LLNL developers_, but test status is publicly available. Only
|
||||
|
||||
@@ -122,14 +122,9 @@ Parallel build:
|
||||
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
|
||||
|
||||
CUDA build:
|
||||
(this build requires CMake 3.17 or newer)
|
||||
(this build requires CMake 3.8 or newer)
|
||||
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_CUDA=YES -DCUDA_ARCH=sm_70
|
||||
make -j 4
|
||||
|
||||
HIP build:
|
||||
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 -DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
|
||||
cmake <mfem-source-dir> -DMFEM_USE_CUDA=YES
|
||||
make -j 4
|
||||
|
||||
Example codes (serial/parallel, depending on the build):
|
||||
@@ -263,7 +258,7 @@ See the configuration file config/defaults.mk for the default settings.
|
||||
Compilers:
|
||||
CXX - C++ compiler, serial build
|
||||
MPICXX - MPI C++ compiler, parallel build
|
||||
CUDA_CXX - The CUDA compiler, 'nvcc' or 'clang++'
|
||||
CUDA_CXX - The CUDA compiler, 'nvcc'
|
||||
|
||||
Compiler options:
|
||||
OPTIM_FLAGS - Options for optimized build
|
||||
@@ -423,10 +418,6 @@ MFEM_USE_GNUTLS = YES/NO
|
||||
When MFEM_USE_GNUTLS is enabled, the additional build options, GNUTLS_*, are
|
||||
also used, see below.
|
||||
|
||||
MFEM_USE_HDF5 = YES/NO
|
||||
The HDF5 library is used for input and output of HDF5 files, for example
|
||||
Cubit mesh files or VTKHDF files for ParaView.
|
||||
|
||||
MFEM_USE_NETCDF = YES/NO
|
||||
NetCDF is the library that is used by the SNL Cubit mesh generator to create
|
||||
Genesis mesh files. This option enables a reader for these files, which
|
||||
@@ -608,12 +599,11 @@ MFEM_USE_TRIBOL = YES/NO
|
||||
|
||||
MFEM_USE_ENZYME = YES/NO
|
||||
Enables automatic differentiation support through the LLVM plugin Enzyme.
|
||||
This requires the compiler to be set to clang (>=14.0.0). We also advise the
|
||||
use of the link time optimization (LTO) plugin, so functions defined over
|
||||
multiple files (compilation units) can be differentiated automatically. This
|
||||
requires to also use LLVM/LLD for linking. The recommended options are in
|
||||
config/defaults.mk. For more detailed instructions, see the section "Specific
|
||||
options for Enzyme" below.
|
||||
This requires the compiler to be set to clang (>=14.0.0). We also advise to
|
||||
use the link time optimization (LTO) plugin, to enable functions that you
|
||||
define over multiple files (compilation units) and want to be differentiated
|
||||
automatically, to work. This requires to also use LLVM/LLD for linking.
|
||||
Recommended options are in config/defaults.mk.
|
||||
|
||||
MFEM_BUILD_TAG = (any value)
|
||||
An optional tag to characterize the build. Exported to config/config.mk.
|
||||
@@ -738,9 +728,6 @@ The specific libraries and their options are:
|
||||
Options: GNUTLS_OPT, GNUTLS_LIB.
|
||||
Versions: GnuTLS >= 2.12.0, older versions may work too.
|
||||
|
||||
- HDF5 (optional), used when MFEM_USE_HDF5 = YES, required for reading and
|
||||
writing files in VTKHDF format.
|
||||
|
||||
- NetCDF (optional), used when MFEM_USE_NETCDF = YES, required for reading Cubit
|
||||
mesh files. Also requires installation of HDF5 and ZLIB, as explained at the
|
||||
NetCDF web site. Note that we use the plain vanilla "C" version of NetCDF, you
|
||||
@@ -836,7 +823,7 @@ The specific libraries and their options are:
|
||||
|
||||
- CUDA (optional), used when MFEM_USE_CUDA = YES.
|
||||
URL: https://developer.nvidia.com/cuda-toolkit
|
||||
Options: CUDA_CXX, CUDA_ARCH, CUDA_OPT, CUDA_LIB, CUDA_DIR (when CUDA_CXX=clang++).
|
||||
Options: CUDA_CXX, CUDA_ARCH, CUDA_OPT, CUDA_LIB.
|
||||
Versions: CUDA >= 10.1.168.
|
||||
|
||||
- HIP (optional), used when MFEM_USE_HIP = YES.
|
||||
@@ -912,7 +899,7 @@ The specific libraries and their options are:
|
||||
- Enzyme, used when MFEM_USE_ENZYME = YES. Requires LLVM/Clang >= 14.0.0.
|
||||
URL: https://github.com/EnzymeAD/Enzyme
|
||||
Options: ENZYME_DIR, ENZYME_OPT, ENZYME_LIB.
|
||||
Versions: Enzyme >= v0.0.176.
|
||||
Versions: Enzyme >= v0.0.33.
|
||||
|
||||
|
||||
Building with CMake
|
||||
@@ -1015,7 +1002,7 @@ Configuration variables (CMake)
|
||||
===============================
|
||||
See the configuration file config/defaults.cmake for the default settings.
|
||||
|
||||
Note: the option MFEM_USE_CUDA requires CMake version 3.17 or newer!
|
||||
Note: the option MFEM_USE_CUDA requires CMake version 3.8 or newer!
|
||||
|
||||
Non-standard CMake variables for compilers:
|
||||
CXX - If set, overwrite the auto-detected C++ compiler, serial build
|
||||
@@ -1046,7 +1033,6 @@ MFEM_USE_STRUMPACK
|
||||
MFEM_USE_GINKGO
|
||||
MFEM_USE_AMGX
|
||||
MFEM_USE_GNUTLS
|
||||
MFEM_USE_HDF5
|
||||
MFEM_USE_NETCDF
|
||||
MFEM_USE_MPFR
|
||||
MFEM_USE_ZLIB
|
||||
@@ -1175,89 +1161,6 @@ larger problems, there are two options:
|
||||
1. Building hypre with '--enable-bigint' defines the local and global indices to
|
||||
be 64-bit. This is convenient, but requires more memory than necessary.
|
||||
|
||||
2. Building hypre with '--enable-mixedint' defines the local indices to be
|
||||
2. Building hypre with '--enable-mixedint' defines the local indiced to be
|
||||
32-bit, while using a 64-bit storage for global indices. This option is
|
||||
currently tested only in ex1p, and may not work in more general settings.
|
||||
|
||||
Specific options for HIP
|
||||
========================
|
||||
MFEM expects the `ROCM_PATH` environment variable to be set to the path of the
|
||||
ROCM install, as well as having `$ROCM_PATH/bin` in `PATH`.
|
||||
|
||||
Specific options for RAJA+HIP+MPI
|
||||
=================================
|
||||
RAJA uses CMake's built-in HIP support (added in CMake 3.21), while MFEM uses
|
||||
the older HIP C++ library build/linkage. To ensure proper build and linkage
|
||||
check that `CMAKE_CXX_COMPILER` and `CMAKE_HIP_COMPILER` are set to the same
|
||||
compiler. This is especially important when using an MPI compiler (for example
|
||||
crayCC) where some linker flags may get dropped if these two are not identical.
|
||||
|
||||
Specific options for Enzyme
|
||||
===========================
|
||||
To work properly, MFEM and Enzyme need to use the same LLVM/Clang configuration.
|
||||
For example, on macOS this can be done by using Homebrew: first install Enzyme,
|
||||
which in turn installs LLVM as a dependency (as of May 2025, this is LLVM 19):
|
||||
|
||||
brew install enzyme
|
||||
|
||||
In order to ensure the correct compiler choice for the MFEM makefile build, set
|
||||
|
||||
CXX = $(shell brew --prefix llvm@19)/bin/clang++
|
||||
|
||||
in the user.mk file (adapted from config/defaults.mk, see the section "Building
|
||||
with GNU make" above). With MPI, it is convenient to set
|
||||
|
||||
MPICXX = OMPI_CXX=$(CXX) mpicxx
|
||||
|
||||
for OpenMPI and
|
||||
|
||||
MPICXX = MPICH_CXX=$(CXX) mpicxx
|
||||
|
||||
for MPICH.
|
||||
|
||||
Additionally, the Enzyme directory needs to be set in user.mk as follows:
|
||||
|
||||
ENZYME_DIR = $(shell brew --prefix enzyme)
|
||||
|
||||
Specifically, a full build on a Mac can be tested by adding the following
|
||||
user.mk file in the config/ directory
|
||||
|
||||
MFEM_USE_ENZYME = YES
|
||||
ENZYME_DIR = $(shell brew --prefix enzyme)
|
||||
LLVM_DIR = $(shell brew --prefix llvm@19)
|
||||
CXX = $(LLVM_DIR)/bin/clang++
|
||||
MFEM_USE_MPI = YES
|
||||
MPICXX = OMPI_CXX=$(CXX) mpicxx
|
||||
|
||||
and running
|
||||
|
||||
make config
|
||||
make -j
|
||||
cd miniapps/dfem
|
||||
make
|
||||
./dfem-minimal-surface
|
||||
|
||||
On Linux systems, for example Ubuntu 24.04, use the package manager to install
|
||||
the Enzyme dependencies
|
||||
|
||||
sudo apt install libclang-dev libzstd-dev llvm-dev clang
|
||||
|
||||
and then clone and build Enzyme
|
||||
|
||||
cd $HOME
|
||||
git clone https://github.com/EnzymeAD/Enzyme.git
|
||||
cd Enzyme/enzyme && mkdir build && cd build
|
||||
CC=clang CXX=clang++ cmake .. -DLLVM_DIR=/usr/lib/llvm-18/lib/cmake -DCMAKE_INSTALL_PREFIX=$HOME/Enzyme/enzyme/build
|
||||
make -j
|
||||
make install
|
||||
|
||||
From here, one can proceed in the same way using the following user.mk settings
|
||||
|
||||
MFEM_USE_ENZYME = YES
|
||||
ENZYME_DIR = $(HOME)/Enzyme/enzyme/build
|
||||
CXX = clang++
|
||||
MFEM_USE_MPI = YES
|
||||
MPICXX = OMPI_CXX=$(CXX) mpicxx
|
||||
|
||||
On other Linux systems the LLVM packages may have different names, for example
|
||||
on RHEL9, one needs to "sudo yum install llvm-devel libzstd clang-devel".
|
||||
|
||||
@@ -41,7 +41,6 @@ set(MFEM_USE_MAGMA @MFEM_USE_MAGMA@)
|
||||
set(MFEM_USE_HIOP @MFEM_USE_HIOP@)
|
||||
set(MFEM_USE_GNUTLS @MFEM_USE_GNUTLS@)
|
||||
set(MFEM_USE_GSLIB @MFEM_USE_GSLIB@)
|
||||
set(MFEM_USE_HDF5 @MFEM_USE_HDF5@)
|
||||
set(MFEM_USE_NETCDF @MFEM_USE_NETCDF@)
|
||||
set(MFEM_USE_PETSC @MFEM_USE_PETSC@)
|
||||
set(MFEM_USE_SLEPC @MFEM_USE_SLEPC@)
|
||||
|
||||
@@ -132,9 +132,6 @@
|
||||
// Enable Conduit support.
|
||||
#cmakedefine MFEM_USE_CONDUIT
|
||||
|
||||
// Enable functionality based on the HDF5 library (reading VTKHDF files).
|
||||
#cmakedefine MFEM_USE_HDF5
|
||||
|
||||
// Enable functionality based on the NetCDF library (reading CUBIT files).
|
||||
#cmakedefine MFEM_USE_NETCDF
|
||||
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
message(STATUS "Looking for ENZYME ...")
|
||||
message(STATUS " in ENZYME_DIR = ${ENZYME_DIR}")
|
||||
|
||||
# Make sure the directory and version combination works. Do nothing otherwise.
|
||||
if(EXISTS "${ENZYME_DIR}/ClangEnzyme-${ENZYME_VERSION}.so")
|
||||
message(STATUS "Found ENZYME: ${ENZYME_DIR}/ClangEnzyme-${ENZYME_VERSION}.so")
|
||||
|
||||
# Set ENZYME_FOUND
|
||||
set(ENZYME_FOUND TRUE CACHE BOOL "ENZYME was found." FORCE)
|
||||
|
||||
# Set CXX flags to accommodate the Enzyme Clang plugin
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -Xclang -load -Xclang ${ENZYME_DIR}/ClangEnzyme-${ENZYME_VERSION}.so -mllvm -enzyme-loose-types=1")
|
||||
set(MFEM_USE_ENZYME YES)
|
||||
else()
|
||||
|
||||
endif()
|
||||
@@ -25,12 +25,7 @@ if (HYPRE_FOUND)
|
||||
find_package(rocsparse REQUIRED)
|
||||
find_package(rocrand REQUIRED)
|
||||
endif()
|
||||
if (HYPRE_LIBRARIES AND HYPRE_INCLUDE_DIRS AND HYPRE_VERSION)
|
||||
find_package_handle_standard_args(HYPRE
|
||||
REQUIRED_VARS HYPRE_LIBRARIES HYPRE_INCLUDE_DIRS HYPRE_VERSION
|
||||
)
|
||||
return()
|
||||
endif()
|
||||
return()
|
||||
endif()
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
@@ -82,13 +77,11 @@ endif()
|
||||
|
||||
if (HYPRE_FOUND AND HYPRE_USING_CUDA)
|
||||
find_package(CUDAToolkit REQUIRED)
|
||||
# Initialize CUSPARSE_LIBRARIES, CURAND_LIBRARIES, and CUBLAS_LIBRARIES:
|
||||
mfem_culib_set_libraries(CUSPARSE cusparse)
|
||||
mfem_culib_set_libraries(CURAND curand)
|
||||
mfem_culib_set_libraries(CUBLAS cublas)
|
||||
mfem_culib_set_libraries(CUSOLVER cusolver)
|
||||
get_target_property(CUSPARSE_LIBRARIES CUDA::cusparse LOCATION)
|
||||
get_target_property(CURAND_LIBRARIES CUDA::curand LOCATION)
|
||||
get_target_property(CUBLAS_LIBRARIES CUDA::cublas LOCATION)
|
||||
list(APPEND HYPRE_LIBRARIES ${CUSPARSE_LIBRARIES} ${CURAND_LIBRARIES}
|
||||
${CUBLAS_LIBRARIES} ${CUSOLVER_LIBRARIES})
|
||||
${CUBLAS_LIBRARIES})
|
||||
set(HYPRE_LIBRARIES ${HYPRE_LIBRARIES} CACHE STRING
|
||||
"HYPRE libraries + dependencies." FORCE)
|
||||
message(STATUS "Updated HYPRE_LIBRARIES: ${HYPRE_LIBRARIES}")
|
||||
@@ -102,7 +95,3 @@ if (HYPRE_FOUND AND HYPRE_USING_HIP)
|
||||
"HYPRE libraries + dependencies." FORCE)
|
||||
message(STATUS "Updated HYPRE_LIBRARIES: ${HYPRE_LIBRARIES}")
|
||||
endif()
|
||||
|
||||
find_package_handle_standard_args(HYPRE
|
||||
REQUIRED_VARS HYPRE_LIBRARIES HYPRE_INCLUDE_DIRS HYPRE_VERSION
|
||||
)
|
||||
|
||||
@@ -19,10 +19,8 @@ mfem_find_package(MAGMA MAGMA MAGMA_DIR "include" "magma.h" "lib" "magma"
|
||||
"Paths to headers required by MAGMA." "Libraries required by MAGMA.")
|
||||
|
||||
if (MAGMA_FOUND AND MFEM_USE_CUDA)
|
||||
find_package(CUDAToolkit REQUIRED)
|
||||
# Initialize CUSPARSE_LIBRARIES and CUBLAS_LIBRARIES:
|
||||
mfem_culib_set_libraries(CUSPARSE cusparse)
|
||||
mfem_culib_set_libraries(CUBLAS cublas)
|
||||
get_target_property(CUSPARSE_LIBRARIES CUDA::cusparse LOCATION)
|
||||
get_target_property(CUBLAS_LIBRARIES CUDA::cublas LOCATION)
|
||||
list(APPEND MAGMA_LIBRARIES ${CUSPARSE_LIBRARIES} ${CUBLAS_LIBRARIES})
|
||||
set(MAGMA_LIBRARIES ${MAGMA_LIBRARIES} CACHE STRING
|
||||
"MAGMA libraries + dependencies." FORCE)
|
||||
|
||||
@@ -123,10 +123,15 @@ macro(add_mfem_miniapp MFEM_EXE_NAME)
|
||||
|
||||
# If CUDA is enabled, tag source files to be compiled with nvcc.
|
||||
if (MFEM_USE_CUDA)
|
||||
set_source_files_properties(${MAIN_LIST} ${EXTRA_SOURCES_LIST}
|
||||
PROPERTIES LANGUAGE CUDA)
|
||||
if (MFEM_CUDA_COMPILER_IS_NVCC)
|
||||
set_source_files_properties(${MAIN_LIST} ${EXTRA_SOURCES_LIST} PROPERTIES LANGUAGE CUDA)
|
||||
if (CMAKE_VERSION VERSION_GREATER_EQUAL 3.12.0)
|
||||
list(TRANSFORM EXTRA_OPTIONS_LIST PREPEND "-Xcompiler=")
|
||||
else()
|
||||
set(LIST_)
|
||||
foreach(item IN LISTS EXTRA_OPTIONS_LIST)
|
||||
list(APPEND LIST_ "-Xcompiler=${item}")
|
||||
endforeach()
|
||||
set(EXTRA_OPTIONS_LIST ${LIST_})
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -152,21 +157,6 @@ macro(add_mfem_miniapp MFEM_EXE_NAME)
|
||||
endif()
|
||||
endmacro()
|
||||
|
||||
# Macro for setting variables like '<culib>_LIBRARIES' where <culib> is a CUDA
|
||||
# library like cublas. This macro assumes that the CUDAToolkit module was loaded
|
||||
# successfully. Example usage:
|
||||
# mfem_culib_set_libraries(CUBLAS cublas)
|
||||
macro(mfem_culib_set_libraries _CULIB _culib)
|
||||
# The following command does not work with older CMake versions, e.g. 3.20:
|
||||
# get_target_property(${_CULIB}_LIBRARIES CUDA::${_culib} LOCATION)
|
||||
# Therefore, we use the respective internal variable:
|
||||
set(${_CULIB}_LIBRARIES ${CUDA_${_culib}_LIBRARY})
|
||||
if (NOT ${_CULIB}_LIBRARIES)
|
||||
message(FATAL_ERROR
|
||||
"Error setting ${_CULIB}_LIBRARIES: ${${_CULIB}_LIBRARIES}")
|
||||
endif()
|
||||
endmacro()
|
||||
|
||||
|
||||
# Auxiliary function, used in mfem_find_package().
|
||||
function(mfem_find_component Prefix DirVar IncSuffixes Header LibSuffixes Lib
|
||||
@@ -879,8 +869,7 @@ function(mfem_export_mk_files)
|
||||
MFEM_USE_OCCA MFEM_USE_CEED MFEM_USE_CALIPER MFEM_USE_UMPIRE MFEM_USE_SIMD
|
||||
MFEM_USE_ADIOS2 MFEM_USE_MKL_CPARDISO MFEM_USE_MKL_PARDISO
|
||||
MFEM_USE_ADFORWARD MFEM_USE_CODIPACK MFEM_USE_BENCHMARK MFEM_USE_PARELAG
|
||||
MFEM_USE_TRIBOL MFEM_USE_MOONOLITH MFEM_USE_ALGOIM MFEM_USE_ENZYME
|
||||
MFEM_USE_HDF5)
|
||||
MFEM_USE_TRIBOL MFEM_USE_MOONOLITH MFEM_USE_ALGOIM MFEM_USE_ENZYME)
|
||||
foreach(var ${CONFIG_MK_BOOL_VARS})
|
||||
if (${var})
|
||||
set(${var} YES)
|
||||
|
||||
@@ -93,17 +93,17 @@ macro (RESOLVE_LIBRARIES LIBS LINK_LINE)
|
||||
set (_directory_list ${_directory_list} ${libpath})
|
||||
set (token ${libname})
|
||||
endif (token MATCHES "^/")
|
||||
set (_lib "NOTFOUND")
|
||||
set (_lib "NOTFOUND" CACHE FILEPATH "Cleared" FORCE)
|
||||
find_library (_lib ${token} HINTS ${_directory_list} ${_root})
|
||||
if (_lib)
|
||||
string (REPLACE "//" "/" _lib ${_lib})
|
||||
string (REPLACE "//" "/" _lib ${_lib})
|
||||
list (APPEND _libs_found ${_lib})
|
||||
else (_lib)
|
||||
message (STATUS "Unable to find library ${token}")
|
||||
endif (_lib)
|
||||
unset(_lib CACHE)
|
||||
endif (token MATCHES "-L([^\" ]+|\"[^\"]+\")")
|
||||
endforeach (token)
|
||||
set (_lib "NOTFOUND" CACHE INTERNAL "Scratch variable" FORCE)
|
||||
# only the LAST occurrence of each library is required since there should be no circular dependencies
|
||||
if (_libs_found)
|
||||
list (REVERSE _libs_found)
|
||||
|
||||
@@ -132,9 +132,6 @@
|
||||
// Enable Conduit support.
|
||||
// #define MFEM_USE_CONDUIT
|
||||
|
||||
// Enable functionality based on the HDF5 library
|
||||
// #define MFEM_USE_HDF5
|
||||
|
||||
// Enable functionality based on the NetCDF library (reading CUBIT files).
|
||||
// #define MFEM_USE_NETCDF
|
||||
|
||||
|
||||
+1
-2
@@ -40,7 +40,6 @@ MFEM_USE_GINKGO = @MFEM_USE_GINKGO@
|
||||
MFEM_USE_AMGX = @MFEM_USE_AMGX@
|
||||
MFEM_USE_MAGMA = @MFEM_USE_MAGMA@
|
||||
MFEM_USE_GNUTLS = @MFEM_USE_GNUTLS@
|
||||
MFEM_USE_HDF5 = @MFEM_USE_HDF5@
|
||||
MFEM_USE_NETCDF = @MFEM_USE_NETCDF@
|
||||
MFEM_USE_PETSC = @MFEM_USE_PETSC@
|
||||
MFEM_USE_SLEPC = @MFEM_USE_SLEPC@
|
||||
@@ -98,7 +97,7 @@ MFEM_MPIEXEC_NP = @MFEM_MPIEXEC_NP@
|
||||
MFEM_MPI_NP = @MFEM_MPI_NP@
|
||||
|
||||
# The NVCC compiler cannot link with -x=cu
|
||||
MFEM_LINK_FLAGS := $(filter-out -x=cu -xcuda -xhip, $(MFEM_FLAGS))
|
||||
MFEM_LINK_FLAGS := $(filter-out -x=cu -xhip, $(MFEM_FLAGS))
|
||||
|
||||
# Optional extra configuration
|
||||
@MFEM_CONFIG_EXTRA@
|
||||
|
||||
@@ -43,7 +43,6 @@ option(MFEM_USE_AMGX "Enable AmgX usage" OFF)
|
||||
option(MFEM_USE_MAGMA "Enable MAGMA usage" OFF)
|
||||
option(MFEM_USE_GNUTLS "Enable GNUTLS usage" OFF)
|
||||
option(MFEM_USE_GSLIB "Enable GSLIB usage" OFF)
|
||||
option(MFEM_USE_HDF5 "Enable HDF5 usage" OFF)
|
||||
option(MFEM_USE_NETCDF "Enable NETCDF usage" OFF)
|
||||
option(MFEM_USE_PETSC "Enable PETSc support." OFF)
|
||||
option(MFEM_USE_SLEPC "Enable SLEPc support." OFF)
|
||||
@@ -268,8 +267,6 @@ set(TRIBOL_DIR "${MFEM_DIR}/../tribol" CACHE PATH "Path to Tribol")
|
||||
set(Tribol_REQUIRED_PACKAGES "Axom/core/mint/slam/slic" CACHE STRING
|
||||
"Additional packages required by Tribol")
|
||||
|
||||
set(ENZYME_DIR "${MFEM_DIR}/../enzyme" CACHE PATH "Path to Enzyme")
|
||||
|
||||
set(BLAS_INCLUDE_DIRS "" CACHE STRING "Path to BLAS headers.")
|
||||
set(BLAS_LIBRARIES "" CACHE STRING "The BLAS library.")
|
||||
set(LAPACK_INCLUDE_DIRS "" CACHE STRING "Path to LAPACK headers.")
|
||||
|
||||
+33
-48
@@ -24,7 +24,7 @@ EGREP_BIN = $(shell command -v egrep 2> /dev/null)
|
||||
CXX = g++
|
||||
MPICXX = mpicxx
|
||||
|
||||
BASE_FLAGS = -std=c++17
|
||||
BASE_FLAGS = -std=c++11
|
||||
OPTIM_FLAGS = -O3 $(BASE_FLAGS)
|
||||
DEBUG_FLAGS = -g $(XCOMPILER)-Wall $(BASE_FLAGS)
|
||||
|
||||
@@ -43,23 +43,12 @@ SHARED = NO
|
||||
|
||||
# CUDA configuration options
|
||||
#
|
||||
# If you set MFEM_USE_ENZYME=YES, must use CUDA_CXX=clang++
|
||||
# If you set MFEM_USE_ENZYME=YES, CUDA_CXX has to be configured to use cuda with
|
||||
# clang as its host compiler.
|
||||
CUDA_CXX = nvcc
|
||||
CUDA_ARCH = sm_60
|
||||
# Base CUDA install directory, only needed if building with clang+cuda:
|
||||
# The default setting is:
|
||||
# 1. If CUDA_HOME is defined and non-empty, use that.
|
||||
# 2. If nvcc is in the path, use the directory two levels up from that.
|
||||
# 3. Use /usr/local/cuda
|
||||
CUDA_DIR = $(or $(CUDA_HOME),$(patsubst %/,%,$(dir \
|
||||
$(patsubst %/,%,$(dir $(shell command -v nvcc))))),/usr/local/cuda)
|
||||
# flags for clang+cuda
|
||||
CLANG_CUDA_FLAGS = -xcuda --cuda-path=$(CUDA_DIR) --cuda-gpu-arch=$(CUDA_ARCH)
|
||||
# flags for nvcc
|
||||
NVCC_FLAGS = -x=cu --expt-extended-lambda --expt-relaxed-constexpr \
|
||||
-arch=$(CUDA_ARCH)
|
||||
# Prefixes for passing flags to the host compiler and linker when using
|
||||
# CUDA_CXX=nvcc
|
||||
CUDA_FLAGS = -x=cu --expt-extended-lambda -arch=$(CUDA_ARCH)
|
||||
# Prefixes for passing flags to the host compiler and linker when using CUDA_CXX
|
||||
CUDA_XCOMPILER = -Xcompiler=
|
||||
CUDA_XLINKER = -Xlinker=
|
||||
|
||||
@@ -156,7 +145,6 @@ MFEM_USE_GINKGO = NO
|
||||
MFEM_USE_AMGX = NO
|
||||
MFEM_USE_MAGMA = NO
|
||||
MFEM_USE_GNUTLS = NO
|
||||
MFEM_USE_HDF5 = NO
|
||||
MFEM_USE_NETCDF = NO
|
||||
MFEM_USE_PETSC = NO
|
||||
MFEM_USE_SLEPC = NO
|
||||
@@ -238,7 +226,7 @@ HYPRE_OPT = -I$(HYPRE_DIR)/include
|
||||
HYPRE_LIB = -L$(HYPRE_DIR)/lib -lHYPRE
|
||||
ifeq (YES,$(MFEM_USE_CUDA))
|
||||
# This is only necessary when hypre is built with cuda:
|
||||
HYPRE_LIB += -lcusolver -lcusparse -lcurand -lcublas
|
||||
HYPRE_LIB += -lcusparse -lcurand -lcublas
|
||||
endif
|
||||
ifeq (YES,$(MFEM_USE_HIP))
|
||||
# This is only necessary when hypre is built with hip:
|
||||
@@ -253,7 +241,7 @@ ifeq ($(MFEM_USE_SUPERLU)$(MFEM_USE_STRUMPACK)$(MFEM_USE_MUMPS),NONONO)
|
||||
METIS_OPT =
|
||||
METIS_LIB = -L$(METIS_DIR) -lmetis
|
||||
else
|
||||
METIS_DIR = @MFEM_DIR@/../metis-5.1.0
|
||||
METIS_DIR = @MFEM_DIR@/../metis-5.0
|
||||
METIS_OPT = -I$(METIS_DIR)/include
|
||||
METIS_LIB = -L$(METIS_DIR)/lib -lmetis
|
||||
endif
|
||||
@@ -413,14 +401,9 @@ MAGMA_LIB = -L$(MAGMA_DIR)/lib -l:libmagma.a -lcublas -lcusparse $(LAPACK_LIB)
|
||||
GNUTLS_OPT =
|
||||
GNUTLS_LIB = -lgnutls
|
||||
|
||||
# HDF5 library configuration
|
||||
HDF5_DIR = $(HOME)/local
|
||||
HDF5_OPT = -I$(HDF5_DIR)/include
|
||||
HDF5_LIB = $(XLINKER)-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib -lhdf5_hl -lhdf5 \
|
||||
$(ZLIB_LIB)
|
||||
|
||||
# NetCDF library configuration
|
||||
NETCDF_DIR = $(HOME)/local
|
||||
HDF5_DIR = $(HOME)/local
|
||||
NETCDF_OPT = -I$(NETCDF_DIR)/include -I$(HDF5_DIR)/include $(ZLIB_OPT)
|
||||
NETCDF_LIB = $(XLINKER)-rpath,$(NETCDF_DIR)/lib -L$(NETCDF_DIR)/lib\
|
||||
$(XLINKER)-rpath,$(HDF5_DIR)/lib -L$(HDF5_DIR)/lib\
|
||||
@@ -501,8 +484,8 @@ SIDRE_LIB = \
|
||||
# Note that PUMI_DIR is needed -- it is used to check for gmi_sim.h
|
||||
PUMI_DIR = @MFEM_DIR@/../pumi-2.1.0
|
||||
PUMI_OPT = -I$(PUMI_DIR)/include
|
||||
PUMI_LIB = -L$(PUMI_DIR)/lib64 -L$(PUMI_DIR)/lib -lpumi -lcrv -lma -lmds -lapf\
|
||||
-lpcu -lgmi -lparma -llion -lmth -lapf_zoltan -lspr
|
||||
PUMI_LIB = -L$(PUMI_DIR)/lib -lpumi -lcrv -lma -lmds -lapf -lpcu -lgmi -lparma\
|
||||
-llion -lmth -lapf_zoltan -lspr
|
||||
|
||||
# HIOP
|
||||
HIOP_DIR = @MFEM_DIR@/../hiop/install
|
||||
@@ -522,9 +505,6 @@ GSLIB_LIB = -L$(GSLIB_DIR)/lib -lgs
|
||||
# CUDA library configuration
|
||||
CUDA_OPT =
|
||||
CUDA_LIB = -lcusparse -lcublas
|
||||
CLANG_CUDA_LIB = -L$(CUDA_DIR)/lib64 -L$(CUDA_DIR)/lib \
|
||||
$(XLINKER)-rpath,$(CUDA_DIR)/lib64,-rpath,$(CUDA_DIR)/lib \
|
||||
-lcudart -ldl -lrt -pthread
|
||||
|
||||
# HIP library configuration
|
||||
HIP_OPT =
|
||||
@@ -588,11 +568,7 @@ RAJA_LIB = $(XLINKER)-rpath,$(RAJA_DIR)/lib -L$(RAJA_DIR)/lib -lRAJA $(CAMP_LIB)
|
||||
# UMPIRE library configuration
|
||||
UMPIRE_DIR = @MFEM_DIR@/../umpire
|
||||
UMPIRE_OPT = -I$(UMPIRE_DIR)/include $(if $(CAMP_DIR), -I$(CAMP_DIR)/include)
|
||||
UMPIRE_LIB = -L$(UMPIRE_DIR)/lib -L$(UMPIRE_DIR)/lib64 -lumpire $(CAMP_LIB)
|
||||
ifdef FMT_DIR
|
||||
UMPIRE_OPT += -I$(FMT_DIR)/include
|
||||
UMPIRE_LIB += -L$(FMT_DIR)/lib -L$(FMT_DIR)/lib64 -lfmt
|
||||
endif
|
||||
UMPIRE_LIB = -L$(UMPIRE_DIR)/lib -lumpire $(CAMP_LIB)
|
||||
|
||||
# MKL CPardiso library configuration
|
||||
MKL_CPARDISO_DIR ?=
|
||||
@@ -624,20 +600,29 @@ TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -lredecomp -L$(AXOM_DIR)/lib -laxom_mi
|
||||
-laxom_slam -laxom_slic -laxom_core
|
||||
|
||||
# Enzyme configuration
|
||||
ENZYME_DIR = @MFEM_DIR@/../enzyme
|
||||
ENZYME_PLUGIN = $(abspath $(wildcard $(subst \
|
||||
@MFEM_DIR@,$(MFEM_DIR),$(ENZYME_DIR))/lib/ClangEnzyme-*.$(SO_EXT)))
|
||||
ifeq ($(MAKECMDGOALS)-$(MFEM_USE_ENZYME),config-YES)
|
||||
ifeq ($(ENZYME_PLUGIN),)
|
||||
$(error Unable to find the Enzyme pluging! Please set ENZYME_DIR)
|
||||
endif
|
||||
ifneq ($(words $(ENZYME_PLUGIN)),1)
|
||||
$(error Multiple versions of the Enzyme pluging found! \
|
||||
Please set ENZYME_PLUGIN directly)
|
||||
endif
|
||||
|
||||
# If you want to enable automatic differentiation at compile time, use the
|
||||
# options below, adapted to your configuration. To be more flexible, we
|
||||
# recommend using the Enzyme plugin during link time optimization. One option is
|
||||
# to add your options to the global compiler/linker flags like
|
||||
#
|
||||
# BASE_FLAGS += -flto
|
||||
# CXX_XLINKER += -fuse-ld=lld -Wl,--lto-legacy-pass-manager\
|
||||
# -Wl,-mllvm=-load=$(ENZYME_DIR)/LLDEnzyme-$(ENZYME_VERSION).so -Wl,
|
||||
#
|
||||
ENZYME_DIR ?= @MFEM_DIR@/../enzyme
|
||||
ENZYME_VERSION ?= 14
|
||||
ENZYME_OPT = -fno-experimental-new-pass-manager -Xclang -load -Xclang $(ENZYME_DIR)/ClangEnzyme-$(ENZYME_VERSION).so
|
||||
ENZYME_LIB = ""
|
||||
|
||||
# Google Benchmark, SUNDIALS >= 6.4.0, STRUMPACK, RAJA, UMPIRE, and Tribol require C++14:
|
||||
ifneq ($(filter YES,$(MFEM_USE_BENCHMARK) $(MFEM_USE_SUNDIALS) $(MFEM_USE_STRUMPACK) $(MFEM_USE_RAJA) $(MFEM_USE_UMPIRE) $(MFEM_USE_TRIBOL)),)
|
||||
BASE_FLAGS = -std=c++14
|
||||
endif
|
||||
# Ginkgo requires C++17:
|
||||
ifeq ($(MFEM_USE_GINKGO),YES)
|
||||
BASE_FLAGS = -std=c++17
|
||||
endif
|
||||
ENZYME_OPT = -fplugin=$(ENZYME_PLUGIN)
|
||||
ENZYME_LIB =
|
||||
|
||||
# If YES, enable some informational messages
|
||||
VERBOSE = NO
|
||||
|
||||
@@ -17,7 +17,7 @@ We provide two containers, which you can either build or use directly from
|
||||
|
||||
In the above, "ghcr.io" means "GitHub Container Registry" and
|
||||
is the [GitHub packages](https://github.com/features/packages) registry that supports
|
||||
Docker images and other OCI artifacts.
|
||||
Docker images and other OCI artifacts.
|
||||
|
||||
### Ubuntu
|
||||
|
||||
@@ -132,7 +132,7 @@ examples.
|
||||
|
||||
> Use this build for a development environment with spack and mfem
|
||||
|
||||
This container is also [provided on GitHub packages](https://github.com/mfem/mfem/pkgs/container/mfem-ubuntu-base),
|
||||
This container is also [provided on GitHub packages](https://github.com/mfem/mfem/pkgs/container/mfem-ubuntu-base),
|
||||
however you can build it locally too:
|
||||
|
||||
```bash
|
||||
@@ -197,7 +197,7 @@ Average reduction factor = 0.140201
|
||||
|
||||
This container is likely ideal for someone that wants to develop mfem itself.
|
||||
For other use cases, we recommend using the slimmer image. As an example,
|
||||
if you want to develop with your own code base (and mfem as is in the container)
|
||||
if you want to develop with your own code base (and mfem as is in the container)
|
||||
you can bind to somewhere else in the container (e.g., src)
|
||||
|
||||
```bash
|
||||
|
||||
@@ -1,20 +0,0 @@
|
||||
diff --git a/CMakeLists.txt b/CMakeLists.txt
|
||||
index 186a320..39e5356 100644
|
||||
--- a/CMakeLists.txt
|
||||
+++ b/CMakeLists.txt
|
||||
@@ -1,4 +1,4 @@
|
||||
-cmake_minimum_required(VERSION 2.8)
|
||||
+cmake_minimum_required(VERSION 3.12.0...4.0.0)
|
||||
project(METIS)
|
||||
|
||||
set(GKLIB_PATH "${CMAKE_SOURCE_DIR}/GKlib" CACHE PATH "path to GKlib")
|
||||
diff --git a/GKlib/CMakeLists.txt b/GKlib/CMakeLists.txt
|
||||
index 67b600a..44321a6 100644
|
||||
--- a/GKlib/CMakeLists.txt
|
||||
+++ b/GKlib/CMakeLists.txt
|
||||
@@ -1,4 +1,4 @@
|
||||
-cmake_minimum_required(VERSION 2.8)
|
||||
+cmake_minimum_required(VERSION 3.12.0...4.0.0)
|
||||
project(GKlib)
|
||||
|
||||
get_filename_component(abs "." ABSOLUTE)
|
||||
@@ -22,7 +22,6 @@ vcpkg_extract_source_archive_ex(
|
||||
fix-linux-build-error.patch
|
||||
install-metisConfig.patch
|
||||
fix-INT_MIN_define.patch
|
||||
cmake4.patch
|
||||
)
|
||||
|
||||
vcpkg_configure_cmake(
|
||||
|
||||
@@ -1,38 +0,0 @@
|
||||
MFEM mesh v1.0
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see fem/geom.hpp):
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
# PYRAMID = 7
|
||||
#
|
||||
|
||||
dimension
|
||||
3
|
||||
|
||||
elements
|
||||
1
|
||||
1 7 0 1 2 3 4
|
||||
|
||||
boundary
|
||||
5
|
||||
1 3 3 2 1 0
|
||||
2 2 0 1 4
|
||||
3 2 1 2 4
|
||||
4 2 2 3 4
|
||||
5 2 3 0 4
|
||||
|
||||
vertices
|
||||
5
|
||||
3
|
||||
-0.5 -0.5 0
|
||||
0.5 -0.5 0
|
||||
0.5 0.5 0
|
||||
-0.5 0.5 0
|
||||
0 0 0.7071067811865475
|
||||
@@ -1,108 +0,0 @@
|
||||
MFEM mesh v1.0
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see fem/geom.hpp):
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
# PYRAMID = 7
|
||||
#
|
||||
|
||||
dimension
|
||||
3
|
||||
|
||||
elements
|
||||
16
|
||||
1 5 12 13 16 15 21 22 25 24
|
||||
1 6 9 12 8 18 21 17
|
||||
1 6 11 8 12 20 17 21
|
||||
1 6 3 2 12 6 5 15
|
||||
1 6 11 12 2 14 15 5
|
||||
1 6 3 12 0 4 13 1
|
||||
1 6 9 0 12 10 1 13
|
||||
1 7 12 13 22 21 19
|
||||
1 7 15 16 13 12 7
|
||||
1 7 12 21 24 15 23
|
||||
1 7 9 12 21 18 19
|
||||
1 7 11 20 21 12 23
|
||||
1 7 9 10 13 12 19
|
||||
1 7 11 12 15 14 23
|
||||
1 7 3 6 15 12 7
|
||||
1 7 3 12 13 4 7
|
||||
|
||||
boundary
|
||||
39
|
||||
1 3 5 6 3 2
|
||||
2 2 6 7 3
|
||||
2 2 7 4 3
|
||||
3 3 3 4 1 0
|
||||
4 2 11 12 8
|
||||
4 2 9 8 12
|
||||
5 2 3 12 2
|
||||
5 2 11 2 12
|
||||
6 3 0 1 10 9
|
||||
7 2 10 19 9
|
||||
7 2 18 9 19
|
||||
8 3 8 9 18 17
|
||||
9 2 4 13 1
|
||||
9 2 10 1 13
|
||||
10 2 4 7 13
|
||||
10 2 16 13 7
|
||||
11 3 13 16 25 22
|
||||
12 2 10 13 19
|
||||
12 2 22 19 13
|
||||
13 2 6 15 7
|
||||
13 2 16 7 15
|
||||
14 2 6 5 15
|
||||
14 2 14 15 5
|
||||
15 2 14 23 15
|
||||
15 2 24 15 23
|
||||
16 3 16 15 24 25
|
||||
17 3 5 2 11 14
|
||||
18 2 3 0 12
|
||||
18 2 9 12 0
|
||||
19 3 11 8 17 20
|
||||
20 2 14 11 23
|
||||
20 2 20 23 11
|
||||
21 2 18 21 17
|
||||
21 2 20 17 21
|
||||
22 2 18 19 21
|
||||
22 2 22 21 19
|
||||
23 3 21 22 25 24
|
||||
24 2 20 21 23
|
||||
24 2 24 23 21
|
||||
|
||||
vertices
|
||||
26
|
||||
3
|
||||
0 -1 -1
|
||||
1 -1 -1
|
||||
-1 0 -1
|
||||
0 0 -1
|
||||
1 0 -1
|
||||
-1 1 -1
|
||||
0 1 -1
|
||||
1 1 -1
|
||||
-1 -1 0
|
||||
0 -1 0
|
||||
1 -1 0
|
||||
-1 0 0
|
||||
0 0 0
|
||||
1 0 0
|
||||
-1 1 0
|
||||
0 1 0
|
||||
1 1 0
|
||||
-1 -1 1
|
||||
0 -1 1
|
||||
1 -1 1
|
||||
-1 0 1
|
||||
0 0 1
|
||||
1 0 1
|
||||
-1 1 1
|
||||
0 1 1
|
||||
1 1 1
|
||||
@@ -1,9 +1,9 @@
|
||||
MFEM INLINE mesh v1.0
|
||||
|
||||
type = pyramid
|
||||
nx = 2
|
||||
ny = 2
|
||||
nz = 2
|
||||
nx = 4
|
||||
ny = 4
|
||||
nz = 4
|
||||
sx = 1.0
|
||||
sy = 1.0
|
||||
sz = 1.0
|
||||
|
||||
@@ -48,7 +48,7 @@ PROJECT_NAME = MFEM
|
||||
# could be handy for archiving the generated documentation or if some version
|
||||
# control system is used.
|
||||
|
||||
PROJECT_NUMBER = v4.8.1
|
||||
PROJECT_NUMBER = v4.7.1
|
||||
|
||||
# Using the PROJECT_BRIEF tag one can provide an optional one line description
|
||||
# for a project that appears at the top of each page and should give viewer a
|
||||
@@ -951,11 +951,7 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
|
||||
@MFEM_SOURCE_DIR@/fem/ceed/integrators/nlconvection \
|
||||
@MFEM_SOURCE_DIR@/fem/ceed/interface \
|
||||
@MFEM_SOURCE_DIR@/fem/ceed/solvers \
|
||||
@MFEM_SOURCE_DIR@/fem/dfem \
|
||||
@MFEM_SOURCE_DIR@/fem/eltrans \
|
||||
@MFEM_SOURCE_DIR@/fem/fe \
|
||||
@MFEM_SOURCE_DIR@/fem/gslib \
|
||||
@MFEM_SOURCE_DIR@/fem/integ \
|
||||
@MFEM_SOURCE_DIR@/fem/lor \
|
||||
@MFEM_SOURCE_DIR@/fem/moonolith \
|
||||
@MFEM_SOURCE_DIR@/fem/qinterp \
|
||||
@@ -973,9 +969,6 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
|
||||
@MFEM_SOURCE_DIR@/miniapps/adjoint \
|
||||
@MFEM_SOURCE_DIR@/miniapps/autodiff \
|
||||
@MFEM_SOURCE_DIR@/miniapps/common \
|
||||
@MFEM_SOURCE_DIR@/miniapps/dfem \
|
||||
@MFEM_SOURCE_DIR@/miniapps/dpg \
|
||||
@MFEM_SOURCE_DIR@/miniapps/dpg/util \
|
||||
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
|
||||
@MFEM_SOURCE_DIR@/miniapps/gslib \
|
||||
@MFEM_SOURCE_DIR@/miniapps/hdiv-linear-solver \
|
||||
@@ -993,10 +986,12 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
|
||||
@MFEM_SOURCE_DIR@/miniapps/performance \
|
||||
@MFEM_SOURCE_DIR@/miniapps/shifted \
|
||||
@MFEM_SOURCE_DIR@/miniapps/solvers \
|
||||
@MFEM_SOURCE_DIR@/miniapps/spde \
|
||||
@MFEM_SOURCE_DIR@/miniapps/tools \
|
||||
@MFEM_SOURCE_DIR@/miniapps/toys \
|
||||
@MFEM_SOURCE_DIR@/miniapps/tribol
|
||||
@MFEM_SOURCE_DIR@/miniapps/tribol \
|
||||
@MFEM_SOURCE_DIR@/miniapps/spde \
|
||||
@MFEM_SOURCE_DIR@/miniapps/dpg \
|
||||
@MFEM_SOURCE_DIR@/miniapps/dpg/util
|
||||
|
||||
# This tag can be used to specify the character encoding of the source files
|
||||
# that doxygen parses. Internally doxygen uses the UTF-8 encoding. Doxygen uses
|
||||
|
||||
+16
-16
@@ -42,10 +42,10 @@ namespace mfem {
|
||||
* - mfem::forall functions in forall.hpp
|
||||
*
|
||||
* <H3>Example codes</H3>
|
||||
* - <a class="el" href="ex0_8cpp_source.html">Example 0</a>: simplest example, nodal H1 FEM for the Poisson problem
|
||||
* - <a class="el" href="ex0p_8cpp_source.html">Example 0p</a>: simplest parallel example, nodal H1 FEM for the Poisson problem
|
||||
* - <a class="el" href="examples_2ex1_8cpp_source.html">Example 1</a>: nodal H1 FEM for the Poisson problem (same discretization as ex0 but with more sophisticated options)
|
||||
* - <a class="el" href="examples_2ex1p_8cpp_source.html">Example 1p</a>: parallel nodal H1 FEM for the Poisson problem (same discretization as ex0p but with more sophisticated options)
|
||||
* - <a class="el" href="ex0_8cpp_source.html">Example 0</a>: simplest example, nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="ex0p_8cpp_source.html">Example 0p</a>: simplest parallel example, nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="examples_2ex1_8cpp_source.html">Example 1</a>: nodal H1 FEM for the Laplace problem (same discretization as ex0 but with more sophisticated options)
|
||||
* - <a class="el" href="examples_2ex1p_8cpp_source.html">Example 1p</a>: parallel nodal H1 FEM for the Laplace problem (same discretization as ex0p but with more sophisticated options)
|
||||
* - <a class="el" href="ex2_8cpp_source.html">Example 2</a>: vector FEM for linear elasticity
|
||||
* - <a class="el" href="ex2p_8cpp_source.html">Example 2p</a>: parallel vector FEM for linear elasticity
|
||||
* - <a class="el" href="ex3_8cpp_source.html">Example 3</a>: Nedelec H(curl) FEM for the definite Maxwell problem
|
||||
@@ -54,12 +54,12 @@ namespace mfem {
|
||||
* - <a class="el" href="ex4p_8cpp_source.html">Example 4p</a>: parallel Raviart-Thomas H(div) FEM for the grad-div problem
|
||||
* - <a class="el" href="ex5_8cpp_source.html">Example 5</a>: mixed pressure-velocity FEM for the Darcy problem
|
||||
* - <a class="el" href="ex5p_8cpp_source.html">Example 5p</a>: parallel mixed pressure-velocity FEM for the Darcy problem
|
||||
* - <a class="el" href="ex6_8cpp_source.html">Example 6</a>: non-conforming adaptive mesh refinement for the Poisson problem
|
||||
* - <a class="el" href="ex6p_8cpp_source.html">Example 6p</a>: parallel non-conforming adaptive mesh refinement for the Poisson problem
|
||||
* - <a class="el" href="ex7_8cpp_source.html">Example 7</a>: screened Poisson problem on a surface (the unit sphere)
|
||||
* - <a class="el" href="ex7p_8cpp_source.html">Example 7p</a>: parallel screened Poisson problem on a surface (the unit sphere)
|
||||
* - <a class="el" href="ex8_8cpp_source.html">Example 8</a>: Discontinuous Petrov-Galerkin (DPG) for the Poisson problem
|
||||
* - <a class="el" href="ex8p_8cpp_source.html">Example 8p</a>: parallel Discontinuous Petrov-Galerkin (DPG) for the Poisson problem
|
||||
* - <a class="el" href="ex6_8cpp_source.html">Example 6</a>: non-conforming adaptive mesh refinement for the Laplace problem
|
||||
* - <a class="el" href="ex6p_8cpp_source.html">Example 6p</a>: parallel non-conforming adaptive mesh refinement for the Laplace problem
|
||||
* - <a class="el" href="ex7_8cpp_source.html">Example 7</a>: Laplace problem on a surface (the unit sphere)
|
||||
* - <a class="el" href="ex7p_8cpp_source.html">Example 7p</a>: parallel Laplace problem on a surface (the unit sphere)
|
||||
* - <a class="el" href="ex8_8cpp_source.html">Example 8</a>: Discontinuous Petrov-Galerkin (DPG) for the Laplace problem
|
||||
* - <a class="el" href="ex8p_8cpp_source.html">Example 8p</a>: parallel Discontinuous Petrov-Galerkin (DPG) for the Laplace problem
|
||||
* - <a class="el" href="ex9_8cpp_source.html">Example 9</a>: Discontinuous Galerkin (DG) time-dependent advection
|
||||
* - <a class="el" href="ex9p_8cpp_source.html">Example 9p</a>: parallel Discontinuous Galerkin (DG) time-dependent advection
|
||||
* - <a class="el" href="ex10_8cpp_source.html">Example 10</a>: time-dependent implicit nonlinear elasticity
|
||||
@@ -67,8 +67,8 @@ namespace mfem {
|
||||
* - <a class="el" href="ex11p_8cpp_source.html">Example 11p</a>: parallel Laplace eigensolver
|
||||
* - <a class="el" href="ex12p_8cpp_source.html">Example 12p</a>: parallel linear elasticity eigensolver
|
||||
* - <a class="el" href="ex13p_8cpp_source.html">Example 13p</a>: parallel Maxwell eigensolver
|
||||
* - <a class="el" href="ex14_8cpp_source.html">Example 14</a>: Discontinuous Galerkin (DG) for the Poisson problem
|
||||
* - <a class="el" href="ex14p_8cpp_source.html">Example 14p</a>: parallel Discontinuous Galerkin (DG) for the Poisson problem
|
||||
* - <a class="el" href="ex14_8cpp_source.html">Example 14</a>: Discontinuous Galerkin (DG) for the Laplace problem
|
||||
* - <a class="el" href="ex14p_8cpp_source.html">Example 14p</a>: parallel Discontinuous Galerkin (DG) for the Laplace problem
|
||||
* - <a class="el" href="ex15_8cpp_source.html">Example 15</a>: dynamic AMR for Laplace with prescribed time-dependent source
|
||||
* - <a class="el" href="ex15p_8cpp_source.html">Example 15p</a>: parallel dynamic AMR for Laplace with prescribed time-dependent source
|
||||
* - <a class="el" href="ex16_8cpp_source.html">Example 16</a>: time-dependent nonlinear heat equation
|
||||
@@ -90,8 +90,8 @@ namespace mfem {
|
||||
* - <a class="el" href="ex24p_8cpp_source.html">Example 24p</a>: parallel mixed finite element spaces and interpolators
|
||||
* - <a class="el" href="ex25_8cpp_source.html">Example 25</a>: simulation of electromagnetic wave propagation using a Perfectly Matched Layer (PML)
|
||||
* - <a class="el" href="ex25p_8cpp_source.html">Example 25p</a>: parallel simulation of electromagnetic wave propagation using a Perfectly Matched Layer (PML)
|
||||
* - <a class="el" href="ex26_8cpp_source.html">Example 26</a>: multigrid preconditioner for the Poisson problem using nodal H1 FEM
|
||||
* - <a class="el" href="ex26p_8cpp_source.html">Example 26p</a>: parallel multigrid preconditioner for the Poisson problem using nodal H1 FEM
|
||||
* - <a class="el" href="ex26_8cpp_source.html">Example 26</a>: multigrid preconditioner for the Laplace problem using nodal H1 FEM
|
||||
* - <a class="el" href="ex26p_8cpp_source.html">Example 26p</a>: parallel multigrid preconditioner for the Laplace problem using nodal H1 FEM
|
||||
* - <a class="el" href="ex27_8cpp_source.html">Example 27</a>: boundary conditions for the Laplace problem
|
||||
* - <a class="el" href="ex27p_8cpp_source.html">Example 27p</a>: parallel boundary conditions for the Laplace problem
|
||||
* - <a class="el" href="ex28_8cpp_source.html">Example 28</a>: sliding contact in elasticity
|
||||
@@ -230,8 +230,8 @@ namespace mfem {
|
||||
* - <a class="el" href="parheat_8cpp_source.html">Optimization gradients</a>: Gradients of PDE-constrained function
|
||||
* - <a class="el" href="par__example_8cpp_source.html">Parallel AD</a>: Parallel p-Laplacian example
|
||||
* - <a class="el" href="seq__example_8cpp_source.html">Serial AD</a>: Serial p-Laplacian example
|
||||
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Poisson problem
|
||||
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Poisson problem
|
||||
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="generate__random__field_8cpp_source.html">SPDE Solvers</a>: SPDE solver random field generation
|
||||
* - <a class="el" href="contact-patch-test_8cpp_source.html">Contact</a>: mortar contact patch test for elasticity
|
||||
* - <a class="el" href="multidomain_8cpp_source.html">Multidomain miniapp</a>: Multidomain and Submesh demonstration miniapp
|
||||
|
||||
@@ -12,7 +12,7 @@
|
||||
// ex1 --amgx-file precon.json --amgx-preconditioner -d cuda
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
// Specifically, we discretize using a FE space of the specified
|
||||
// order, or if order < 1 using an isoparametric/isogeometric
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
// mpirun -np 4 ex1p --amgx-file amg_pcg.json
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
// Specifically, we discretize using a FE space of the specified
|
||||
// order, or if order < 1 using an isoparametric/isogeometric
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -7,7 +7,7 @@
|
||||
// ex0 -m ../data/square-disc.mesh -o 2
|
||||
//
|
||||
// Description: This example code demonstrates the most basic usage of MFEM to
|
||||
// define a simple finite element discretization of the Poisson
|
||||
// define a simple finite element discretization of the Laplace
|
||||
// problem -Delta u = 1 with zero Dirichlet boundary conditions.
|
||||
// General 2D/3D mesh files and finite element polynomial degrees
|
||||
// can be specified by command line options.
|
||||
|
||||
+1
-1
@@ -8,7 +8,7 @@
|
||||
//
|
||||
// Description: This example code demonstrates the most basic parallel usage of
|
||||
// MFEM to define a simple finite element discretization of the
|
||||
// Poisson problem -Delta u = 1 with zero Dirichlet boundary
|
||||
// Laplace problem -Delta u = 1 with zero Dirichlet boundary
|
||||
// conditions. General 2D/3D serial mesh files and finite element
|
||||
// polynomial degrees can be specified by command line options.
|
||||
|
||||
|
||||
+1
-1
@@ -50,7 +50,7 @@
|
||||
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cuda:/gpu/cuda/ref
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
// Specifically, we discretize using a FE space of the specified
|
||||
// order, or if order < 1 using an isoparametric/isogeometric
|
||||
|
||||
+1
-1
@@ -27,7 +27,7 @@
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// discontinuous Galerkin (DG) finite element discretization of
|
||||
// the Poisson problem -Delta u = 1 with homogeneous Dirichlet
|
||||
// the Laplace problem -Delta u = 1 with homogeneous Dirichlet
|
||||
// boundary conditions. Finite element spaces of any order,
|
||||
// including zero on regular grids, are supported. The example
|
||||
// highlights the use of discontinuous spaces and DG-specific face
|
||||
|
||||
+1
-1
@@ -26,7 +26,7 @@
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// discontinuous Galerkin (DG) finite element discretization of
|
||||
// the Poisson problem -Delta u = 1 with homogeneous Dirichlet
|
||||
// the Laplace problem -Delta u = 1 with homogeneous Dirichlet
|
||||
// boundary conditions. Finite element spaces of any order,
|
||||
// including zero on regular grids, are supported. The example
|
||||
// highlights the use of discontinuous spaces and DG-specific face
|
||||
|
||||
+2
-2
@@ -159,7 +159,7 @@ int main(int argc, char *argv[])
|
||||
FiniteElementSpace fespace(&mesh, &fec);
|
||||
|
||||
// 6. As in Example 1p, we set up bilinear and linear forms corresponding to
|
||||
// the Poisson problem -\Delta u = 1. We don't assemble the discrete
|
||||
// the Laplace problem -\Delta u = 1. We don't assemble the discrete
|
||||
// problem yet, this will be done in the inner loop.
|
||||
BilinearForm a(&fespace);
|
||||
LinearForm b(&fespace);
|
||||
@@ -446,7 +446,7 @@ real_t bdr_func(const Vector &pt, real_t t)
|
||||
return composite_func(pt, t, front, ball);
|
||||
}
|
||||
|
||||
// Laplacian of the exact solution, used for the right hand side.
|
||||
// Laplace of the exact solution, used for the right hand side.
|
||||
real_t rhs_func(const Vector &pt, real_t t)
|
||||
{
|
||||
return composite_func(pt, t, front_laplace, ball_laplace);
|
||||
|
||||
+2
-2
@@ -181,7 +181,7 @@ int main(int argc, char *argv[])
|
||||
ParFiniteElementSpace fespace(&pmesh, &fec);
|
||||
|
||||
// 7. As in Example 1p, we set up bilinear and linear forms corresponding to
|
||||
// the Poisson problem -\Delta u = 1. We don't assemble the discrete
|
||||
// the Laplace problem -\Delta u = 1. We don't assemble the discrete
|
||||
// problem yet, this will be done in the inner loop.
|
||||
ParBilinearForm a(&fespace);
|
||||
ParLinearForm b(&fespace);
|
||||
@@ -507,7 +507,7 @@ real_t bdr_func(const Vector &pt, real_t t)
|
||||
return composite_func(pt, t, front, ball);
|
||||
}
|
||||
|
||||
// Laplacian of the exact solution, used for the right hand side.
|
||||
// Laplace of the exact solution, used for the right hand side.
|
||||
real_t rhs_func(const Vector &pt, real_t t)
|
||||
{
|
||||
return composite_func(pt, t, front_laplace, ball_laplace);
|
||||
|
||||
+1
-1
@@ -45,7 +45,7 @@
|
||||
// mpirun -np 4 ex1p -m ../data/beam-tet.mesh -pa -d ceed-cpu
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
// Specifically, we discretize using a FE space of the specified
|
||||
// order, or if order < 1 using an isoparametric/isogeometric
|
||||
|
||||
+1
-1
@@ -94,7 +94,7 @@ WaveOperator::WaveOperator(FiniteElementSpace &f,
|
||||
M->AddDomainIntegrator(new MassIntegrator());
|
||||
M->Assemble();
|
||||
|
||||
// Apply BCs
|
||||
// Apply Bcs
|
||||
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
K->FormSystemMatrix(ess_tdof_list, Kmat);
|
||||
M->FormSystemMatrix(ess_tdof_list, Mmat);
|
||||
|
||||
+1
-1
@@ -17,7 +17,7 @@
|
||||
// ex26 -m ../data/beam-hex.mesh -d cuda
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions
|
||||
// as in Example 1.
|
||||
//
|
||||
|
||||
+1
-1
@@ -14,7 +14,7 @@
|
||||
// mpirun -np 4 ex26p -d ceed-cuda
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions
|
||||
// as in Example 1.
|
||||
//
|
||||
|
||||
+1
-1
@@ -10,7 +10,7 @@
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// finite element discretization of a PDE on a 2 dimensional
|
||||
// surface embedded in a 3 dimensional domain. In this case we
|
||||
// solve the Poisson problem -Div(sigma Grad u) = 1, with
|
||||
// solve the Laplace problem -Div(sigma Grad u) = 1, with
|
||||
// homogeneous Dirichlet boundary conditions, where sigma is an
|
||||
// anisotropic diffusion constant defined as a 3x3 matrix
|
||||
// coefficient.
|
||||
|
||||
+1
-1
@@ -10,7 +10,7 @@
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// finite element discretization of a PDE on a 2 dimensional
|
||||
// surface embedded in a 3 dimensional domain. In this case we
|
||||
// solve the Poisson problem -Div(sigma Grad u) = 1, with
|
||||
// solve the Laplace problem -Div(sigma Grad u) = 1, with
|
||||
// homogeneous Dirichlet boundary conditions, where sigma is an
|
||||
// anisotropic diffusion constant defined as a 3x3 matrix
|
||||
// coefficient.
|
||||
|
||||
+3
-3
@@ -390,7 +390,7 @@ public:
|
||||
|
||||
|
||||
/**
|
||||
@brief Class for surface linear form integrator
|
||||
@brief Class for surface linearform integrator
|
||||
|
||||
Integrator to demonstrate the use of the surface integration rule on an
|
||||
implicit surface defined by a level-set.
|
||||
@@ -460,7 +460,7 @@ public:
|
||||
};
|
||||
|
||||
/**
|
||||
@brief Class for subdomain linear form integrator
|
||||
@brief Class for subdomain linearform integrator
|
||||
|
||||
Integrator to demonstrate the use of the subdomain integration rule within
|
||||
an area defined by an implicit surface defined by a level-set.
|
||||
@@ -546,7 +546,7 @@ int main(int argc, char *argv[])
|
||||
args.AddOption(&ref_levels, "-r", "--refine", "Number of meh refinements");
|
||||
args.AddOption(&method, "-m", "--method",
|
||||
"Cut integration method: 0 for moments-based, 1 for Algoim.");
|
||||
args.AddOption(&inttype, "-i", "--integration-type",
|
||||
args.AddOption(&inttype, "-i", "--integrationtype",
|
||||
"IntegrationType to demonstrate");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
|
||||
+1
-1
@@ -16,7 +16,7 @@
|
||||
//
|
||||
// The particular problem being solved here is nearly the same
|
||||
// as that in example 1 i.e. a simple finite element
|
||||
// discretization of the Poisson problem -Delta u = 1 with
|
||||
// discretization of the Laplace problem -Delta u = 1 with
|
||||
// homogeneous Dirichlet boundary conditions and, in this case,
|
||||
// an inhomogeneous diffusion coefficient. The diffusion
|
||||
// coefficient is given a small default value throughout the
|
||||
|
||||
+1
-1
@@ -16,7 +16,7 @@
|
||||
//
|
||||
// The particular problem being solved here is nearly the same
|
||||
// as that in example 1 i.e. a simple finite element
|
||||
// discretization of the Poisson problem -Delta u = 1 with
|
||||
// discretization of the Laplace problem -Delta u = 1 with
|
||||
// homogeneous Dirichlet boundary conditions and, in this case,
|
||||
// an inhomogeneous diffusion coefficient. The diffusion
|
||||
// coefficient is given a small default value throughout the
|
||||
|
||||
+7
-5
@@ -65,7 +65,6 @@ int main(int argc, char *argv[])
|
||||
bool static_cond = false;
|
||||
bool hybridization = false;
|
||||
bool pa = false;
|
||||
bool ea = false;
|
||||
const char *device_config = "cpu";
|
||||
bool visualization = 1;
|
||||
|
||||
@@ -84,14 +83,18 @@ int main(int argc, char *argv[])
|
||||
"--no-hybridization", "Enable hybridization.");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&ea, "-ea", "--element-assembly", "-no-ea",
|
||||
"--no-element-assembly", "Enable Element Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.ParseCheck();
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
return 1;
|
||||
}
|
||||
args.PrintOptions(cout);
|
||||
kappa = freq * M_PI;
|
||||
|
||||
// 2. Enable hardware devices such as GPUs, and programming models such as
|
||||
@@ -163,7 +166,6 @@ int main(int argc, char *argv[])
|
||||
Coefficient *beta = new ConstantCoefficient(1.0);
|
||||
BilinearForm *a = new BilinearForm(fespace);
|
||||
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
if (ea) { a->SetAssemblyLevel(AssemblyLevel::ELEMENT); }
|
||||
a->AddDomainIntegrator(new DivDivIntegrator(*alpha));
|
||||
a->AddDomainIntegrator(new VectorFEMassIntegrator(*beta));
|
||||
|
||||
|
||||
+13
-5
@@ -71,7 +71,6 @@ int main(int argc, char *argv[])
|
||||
bool static_cond = false;
|
||||
bool hybridization = false;
|
||||
bool pa = false;
|
||||
bool ea = false;
|
||||
const char *device_config = "cpu";
|
||||
bool visualization = 1;
|
||||
|
||||
@@ -90,14 +89,24 @@ int main(int argc, char *argv[])
|
||||
"--no-hybridization", "Enable hybridization.");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&ea, "-ea", "--element-assembly", "-no-ea",
|
||||
"--no-element-assembly", "Enable Element Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.ParseCheck();
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
kappa = freq * M_PI;
|
||||
|
||||
// 3. Enable hardware devices such as GPUs, and programming models such as
|
||||
@@ -185,7 +194,6 @@ int main(int argc, char *argv[])
|
||||
Coefficient *beta = new ConstantCoefficient(1.0);
|
||||
ParBilinearForm *a = new ParBilinearForm(fespace);
|
||||
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
if (ea) { a->SetAssemblyLevel(AssemblyLevel::ELEMENT); }
|
||||
a->AddDomainIntegrator(new DivDivIntegrator(*alpha));
|
||||
a->AddDomainIntegrator(new VectorFEMassIntegrator(*beta));
|
||||
|
||||
|
||||
+2
-2
@@ -25,7 +25,7 @@
|
||||
// ex6 -pa -d ceed-cuda:/gpu/cuda/shared
|
||||
//
|
||||
// Description: This is a version of Example 1 with a simple adaptive mesh
|
||||
// refinement loop. The problem being solved is again the Poisson
|
||||
// refinement loop. The problem being solved is again the Laplace
|
||||
// equation -Delta u = 1 with homogeneous Dirichlet boundary
|
||||
// conditions. The problem is solved on a sequence of meshes which
|
||||
// are locally refined in a conforming (triangles, tetrahedrons)
|
||||
@@ -113,7 +113,7 @@ int main(int argc, char *argv[])
|
||||
FiniteElementSpace fespace(&mesh, &fec);
|
||||
|
||||
// 6. As in Example 1, we set up bilinear and linear forms corresponding to
|
||||
// the Poisson problem -\Delta u = 1. We don't assemble the discrete
|
||||
// the Laplace problem -\Delta u = 1. We don't assemble the discrete
|
||||
// problem yet, this will be done in the main loop.
|
||||
BilinearForm a(&fespace);
|
||||
if (pa)
|
||||
|
||||
+7
-104
@@ -3,7 +3,6 @@
|
||||
// Compile with: make ex6p
|
||||
//
|
||||
// Sample runs: mpirun -np 4 ex6p -m ../data/star-hilbert.mesh -o 2
|
||||
// mpirun -np 4 ex6p -m ../data/star-hilbert.mesh -pref
|
||||
// mpirun -np 4 ex6p -m ../data/square-disc.mesh -rm 1 -o 1
|
||||
// mpirun -np 4 ex6p -m ../data/square-disc.mesh -rm 1 -o 2 -h1
|
||||
// mpirun -np 4 ex6p -m ../data/square-disc.mesh -o 2 -cs
|
||||
@@ -29,7 +28,7 @@
|
||||
// mpirun -np 4 ex6p -pa -d ceed-cuda:/gpu/cuda/shared
|
||||
//
|
||||
// Description: This is a version of Example 1 with a simple adaptive mesh
|
||||
// refinement loop. The problem being solved is again the Poisson
|
||||
// refinement loop. The problem being solved is again the Laplace
|
||||
// equation -Delta u = 1 with homogeneous Dirichlet boundary
|
||||
// conditions. The problem is solved on a sequence of meshes which
|
||||
// are locally refined in a conforming (triangles, tetrahedrons)
|
||||
@@ -42,12 +41,6 @@
|
||||
// from coarse to fine meshes, restarting from a checkpoint, as
|
||||
// well as persistent GLVis visualization are also illustrated.
|
||||
//
|
||||
// There is also the option to use hp-refinement. Real
|
||||
// applications should use some problem-dependent criteria for
|
||||
// selecting between h- and p-refinement, but in this example, we
|
||||
// simply alternate between refinement types to demonstrate the
|
||||
// capabilities.
|
||||
//
|
||||
// We recommend viewing Example 1 before viewing this example.
|
||||
|
||||
#include "mfem.hpp"
|
||||
@@ -76,8 +69,6 @@ int main(int argc, char *argv[])
|
||||
bool smooth_rt = true;
|
||||
bool restart = false;
|
||||
bool visualization = true;
|
||||
bool rebalance = true;
|
||||
bool usePRefinement = false;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
@@ -99,10 +90,6 @@ int main(int argc, char *argv[])
|
||||
"Stop after reaching this many degrees of freedom.");
|
||||
args.AddOption(&smooth_rt, "-rt", "--smooth-rt", "-h1", "--smooth-h1",
|
||||
"Represent the smooth flux in RT or vector H1 space.");
|
||||
args.AddOption(&usePRefinement, "-pref", "--p-refine", "-no-pref",
|
||||
"--no-p-refine", "Alternate between h- and p-refinement.");
|
||||
args.AddOption(&rebalance, "-reb", "--rebalance", "-no-reb",
|
||||
"--no-rebalance", "Load balance the nonconforming mesh.");
|
||||
args.AddOption(&restart, "-res", "--restart", "-no-res", "--no-restart",
|
||||
"Restart computation from the last checkpoint.");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
@@ -122,15 +109,6 @@ int main(int argc, char *argv[])
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
if (usePRefinement && rebalance)
|
||||
{
|
||||
rebalance = false;
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Load balancing is not performed with p-refinements.\n";
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
Device device(device_config);
|
||||
@@ -208,7 +186,7 @@ int main(int argc, char *argv[])
|
||||
ParFiniteElementSpace fespace(pmesh, &fec);
|
||||
|
||||
// 11. As in Example 1p, we set up bilinear and linear forms corresponding to
|
||||
// the Poisson problem -\Delta u = 1. We don't assemble the discrete
|
||||
// the Laplace problem -\Delta u = 1. We don't assemble the discrete
|
||||
// problem yet, this will be done in the main loop.
|
||||
ParBilinearForm a(&fespace);
|
||||
if (pa)
|
||||
@@ -343,15 +321,7 @@ int main(int argc, char *argv[])
|
||||
if (visualization)
|
||||
{
|
||||
sout << "parallel " << num_procs << " " << myid << "\n";
|
||||
if (usePRefinement)
|
||||
{
|
||||
std::unique_ptr<GridFunction> vis_x = x.ProlongateToMaxOrder();
|
||||
sout << "solution\n" << *pmesh << *vis_x << flush;
|
||||
}
|
||||
else
|
||||
{
|
||||
sout << "solution\n" << *pmesh << x << flush;
|
||||
}
|
||||
sout << "solution\n" << *pmesh << x << flush;
|
||||
}
|
||||
|
||||
if (global_dofs >= max_dofs)
|
||||
@@ -367,31 +337,8 @@ int main(int argc, char *argv[])
|
||||
// estimator to obtain element errors, then it selects elements to be
|
||||
// refined and finally it modifies the mesh. The Stop() method can be
|
||||
// used to determine if a stopping criterion was met.
|
||||
|
||||
// Simply alternate between h- and p-refinement.
|
||||
const bool pRefine = usePRefinement && ((it % 2) == 1);
|
||||
bool stop = false;
|
||||
Array<pRefinement> prefinements;
|
||||
if (pRefine)
|
||||
{
|
||||
Array<Refinement> refinements;
|
||||
refiner.MarkWithoutRefining(*pmesh, refinements);
|
||||
stop = pmesh->ReduceInt(refinements.Size()) == 0LL;
|
||||
|
||||
prefinements.SetSize(refinements.Size());
|
||||
for (int i=0; i<refinements.Size(); ++i)
|
||||
{
|
||||
prefinements[i].index = refinements[i].index;
|
||||
prefinements[i].delta = 1; // Increase the element order by 1
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
refiner.Apply(*pmesh);
|
||||
stop = refiner.Stop();
|
||||
}
|
||||
|
||||
if (stop)
|
||||
refiner.Apply(*pmesh);
|
||||
if (refiner.Stop())
|
||||
{
|
||||
if (myid == 0)
|
||||
{
|
||||
@@ -405,20 +352,12 @@ int main(int argc, char *argv[])
|
||||
// to any GridFunctions over the space. In this case, the update
|
||||
// matrix is an interpolation matrix so the updated GridFunction will
|
||||
// still represent the same function as before refinement.
|
||||
if (pRefine)
|
||||
{
|
||||
fespace.PRefineAndUpdate(prefinements);
|
||||
}
|
||||
else
|
||||
{
|
||||
fespace.Update();
|
||||
}
|
||||
|
||||
fespace.Update();
|
||||
x.Update();
|
||||
|
||||
// 25. Load balance the mesh, and update the space and solution. Currently
|
||||
// available only for nonconforming meshes.
|
||||
if (pmesh->Nonconforming() && rebalance)
|
||||
if (pmesh->Nonconforming())
|
||||
{
|
||||
pmesh->Rebalance();
|
||||
|
||||
@@ -450,42 +389,6 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
}
|
||||
|
||||
// Save result
|
||||
if (usePRefinement)
|
||||
{
|
||||
L2_FECollection fecL2(0, dim);
|
||||
ParFiniteElementSpace l2fespace(pmesh, &fecL2);
|
||||
ParGridFunction xo(&l2fespace); // Element order field
|
||||
xo = 0.0;
|
||||
|
||||
for (int e=0; e<pmesh->GetNE(); ++e)
|
||||
{
|
||||
const int p_elem = fespace.GetElementOrder(e);
|
||||
Array<int> dofs;
|
||||
l2fespace.GetElementDofs(e, dofs);
|
||||
xo[dofs[0]] = p_elem;
|
||||
}
|
||||
|
||||
ostringstream mesh_name, sol_name, order_name;
|
||||
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
|
||||
sol_name << "sol." << setfill('0') << setw(6) << myid;
|
||||
order_name << "order." << setfill('0') << setw(6) << myid;
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
pmesh->ParPrint(mesh_ofs);
|
||||
|
||||
ofstream sol_ofs(sol_name.str().c_str());
|
||||
sol_ofs.precision(8);
|
||||
|
||||
std::unique_ptr<ParGridFunction> vis_x = x.ProlongateToMaxOrder();
|
||||
vis_x->Save(sol_ofs);
|
||||
|
||||
ofstream order_ofs(order_name.str().c_str());
|
||||
order_ofs.precision(8);
|
||||
xo.Save(order_ofs);
|
||||
}
|
||||
|
||||
delete smooth_flux_fes;
|
||||
delete smooth_flux_fec;
|
||||
delete pmesh;
|
||||
|
||||
+2
-2
@@ -9,8 +9,8 @@
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// triangulation of a unit sphere and a simple isoparametric
|
||||
// finite element discretization of the screened Poisson problem,
|
||||
// -Delta u + u = f.
|
||||
// finite element discretization of the Laplace problem with mass
|
||||
// term, -Delta u + u = f.
|
||||
//
|
||||
// The example highlights mesh generation, the use of mesh
|
||||
// refinement, high-order meshes and finite elements, as well as
|
||||
|
||||
+2
-2
@@ -9,8 +9,8 @@
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// triangulation of a unit sphere and a simple isoparametric
|
||||
// finite element discretization of the screened Poisson problem,
|
||||
// -Delta u + u = f.
|
||||
// finite element discretization of the Laplace problem with mass
|
||||
// term, -Delta u + u = f.
|
||||
//
|
||||
// The example highlights mesh generation, the use of mesh
|
||||
// refinement, high-order meshes and finite elements, as well as
|
||||
|
||||
+1
-1
@@ -15,7 +15,7 @@
|
||||
//
|
||||
// Description: This example code demonstrates the use of the Discontinuous
|
||||
// Petrov-Galerkin (DPG) method in its primal 2x2 block form as a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = f with homogeneous Dirichlet boundary conditions. We
|
||||
// use high-order continuous trial space, a high-order interfacial
|
||||
// (trace) space, and a high-order discontinuous test space
|
||||
|
||||
+1
-1
@@ -14,7 +14,7 @@
|
||||
//
|
||||
// Description: This example code demonstrates the use of the Discontinuous
|
||||
// Petrov-Galerkin (DPG) method in its primal 2x2 block form as a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = f with homogeneous Dirichlet boundary conditions. We
|
||||
// use high-order continuous trial space, a high-order interfacial
|
||||
// (trace) space, and a high-order discontinuous test space
|
||||
|
||||
@@ -35,7 +35,7 @@
|
||||
// ex1 -m ../../data/beam-hex.mesh -pa -d cuda
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
// Specifically, we discretize using a FE space of the specified
|
||||
// order, or if order < 1 using an isoparametric/isogeometric
|
||||
|
||||
@@ -33,7 +33,7 @@
|
||||
"id": "public-white",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"This is the simplest MFEM example and a good starting point for new users. The example demonstrates the use of MFEM to define and solve an $H^1$ finite element discretization of the Poisson problem\n",
|
||||
"This is the simplest MFEM example and a good starting point for new users. The example demonstrates the use of MFEM to define and solve an $H^1$ finite element discretization of the Laplace problem\n",
|
||||
"\n",
|
||||
"$$\n",
|
||||
"-\\Delta u = 1\n",
|
||||
|
||||
+1
-1
@@ -189,7 +189,7 @@ clean-build:
|
||||
clean-exec:
|
||||
@rm -f refined.mesh displaced.mesh mesh.* ex5.mesh ex6p-checkpoint.*
|
||||
@rm -rf Example5* Example9* Example15* Example16* Example23* ParaView
|
||||
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.* order.*
|
||||
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.*
|
||||
@rm -f ex9.mesh ex9-mesh.* ex9-init.* ex9-final.*
|
||||
@rm -f deformed.* velocity.* elastic_energy.* mode_* mode_deriv_* flux.*
|
||||
@rm -f ex5-p-*.bp ex9-p-*.bp ex12-p-*.bp ex16-p-*.bp
|
||||
|
||||
@@ -16,16 +16,10 @@
|
||||
// multi-physics applications.
|
||||
//
|
||||
// This particular example is only for serial runtimes.
|
||||
// For non-conforming meshes please have a look at example
|
||||
// "ex2p.cpp".
|
||||
|
||||
#include "example_utils.hpp"
|
||||
#include "mfem.hpp"
|
||||
|
||||
#ifndef MFEM_USE_MOONOLITH
|
||||
#error This example requires that MFEM is built with MFEM_USE_MOONOLITH=YES
|
||||
#endif
|
||||
|
||||
using namespace mfem;
|
||||
using namespace std;
|
||||
|
||||
@@ -221,8 +215,8 @@ int main(int argc, char *argv[])
|
||||
mfem::out << "l2 error: src: " << src_err << ", dest: " << dest_err
|
||||
<< std::endl;
|
||||
|
||||
plot(*src_mesh, src_fun, "source", 0);
|
||||
plot(*dest_mesh, dest_fun, "destination", 1);
|
||||
plot(*src_mesh, src_fun, "source");
|
||||
plot(*dest_mesh, dest_fun, "destination");
|
||||
}
|
||||
}
|
||||
else
|
||||
|
||||
+16
-54
@@ -8,23 +8,18 @@
|
||||
// mpirun -np 4 ex1p -s ../../data/inline-hex.mesh -d ../../data/inline-tet.mesh
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM for transferring
|
||||
// discrete fields from one conforming finite element mesh to another. The
|
||||
// discrete fields from one finite element mesh to another. The
|
||||
// meshes can be of arbitrary shape and completely unrelated with
|
||||
// each other. This feature can be used for implementing immersed
|
||||
// domain methods for fluid-structure interaction or general
|
||||
// multi-physics applications.
|
||||
//
|
||||
// This particular example is for parallel runtimes. Vector FE is
|
||||
// an experimental feature in parallel. For non-conforming meshes
|
||||
// please have a look at example "ex2p.cpp".
|
||||
// an experimental feature in parallel.
|
||||
|
||||
#include "example_utils.hpp"
|
||||
#include "mfem.hpp"
|
||||
|
||||
#ifndef MFEM_USE_MOONOLITH
|
||||
#error This example requires that MFEM is built with MFEM_USE_MOONOLITH=YES
|
||||
#endif
|
||||
|
||||
using namespace mfem;
|
||||
using namespace std;
|
||||
|
||||
@@ -55,8 +50,6 @@ int main(int argc, char *argv[])
|
||||
int dest_fe_order = 1;
|
||||
bool visualization = true;
|
||||
bool use_vector_fe = false;
|
||||
bool use_h1 = true;
|
||||
bool use_vector_space = false;
|
||||
bool verbose = false;
|
||||
bool assemble_mass_and_coupling_together = true;
|
||||
|
||||
@@ -79,28 +72,14 @@ int main(int argc, char *argv[])
|
||||
args.AddOption(&verbose, "-verb", "--verbose", "--no-verb", "--no-verbose",
|
||||
"Enable/Disable verbose output");
|
||||
args.AddOption(&use_vector_fe, "-vfe", "--use_vector_fe", "-no-vfe",
|
||||
"--no-vector_fe",
|
||||
"Use RT|ND vector finite elements (Experimental)");
|
||||
args.AddOption(&use_vector_space, "-vfs", "--use_vector_space", "-no-vfs",
|
||||
"--no-vector_space",
|
||||
"Use Lagrange vector finite elements (Experimental)");
|
||||
args.AddOption(&use_h1, "-h1", "--use-h1", "-nh1", "--no-h1",
|
||||
"Use H1 collection");
|
||||
"--no-vector_fe", "Use vector finite elements (Experimental)");
|
||||
args.AddOption(&assemble_mass_and_coupling_together, "-act",
|
||||
"--assemble_mass_and_coupling_together", "-no-act",
|
||||
"--no-assemble_mass_and_coupling_together",
|
||||
"Assemble mass and coupling operators together (better for "
|
||||
"non-affine elements)");
|
||||
"Assemble mass and coupling operators together (better for non-affine elements)");
|
||||
args.Parse();
|
||||
check_options(args);
|
||||
|
||||
if (use_vector_fe && use_vector_space)
|
||||
{
|
||||
mfem::err <<
|
||||
"WARNING: use_vector_fe and use_vector_space options"
|
||||
"are both true, ignoring use_vector_fe\n";
|
||||
}
|
||||
|
||||
shared_ptr<Mesh> src_mesh, dest_mesh;
|
||||
|
||||
ifstream imesh;
|
||||
@@ -190,30 +169,17 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
else
|
||||
{
|
||||
|
||||
if (use_h1)
|
||||
{
|
||||
src_fe_coll =
|
||||
make_shared<H1_FECollection>(source_fe_order, src_mesh->Dimension());
|
||||
dest_fe_coll =
|
||||
make_shared<H1_FECollection>(dest_fe_order, dest_mesh->Dimension());
|
||||
}
|
||||
else
|
||||
{
|
||||
src_fe_coll =
|
||||
make_shared<L2_FECollection>(source_fe_order, src_mesh->Dimension());
|
||||
dest_fe_coll =
|
||||
make_shared<L2_FECollection>(dest_fe_order, dest_mesh->Dimension());
|
||||
}
|
||||
src_fe_coll =
|
||||
make_shared<L2_FECollection>(source_fe_order, src_mesh->Dimension());
|
||||
dest_fe_coll =
|
||||
make_shared<L2_FECollection>(dest_fe_order, dest_mesh->Dimension());
|
||||
}
|
||||
|
||||
auto src_fe = make_shared<ParFiniteElementSpace>(
|
||||
p_src_mesh.get(), src_fe_coll.get(),
|
||||
use_vector_space ? src_mesh->Dimension() : 1);
|
||||
auto src_fe =
|
||||
make_shared<ParFiniteElementSpace>(p_src_mesh.get(), src_fe_coll.get());
|
||||
|
||||
auto dest_fe = make_shared<ParFiniteElementSpace>(
|
||||
p_dest_mesh.get(), dest_fe_coll.get(),
|
||||
use_vector_space ? dest_mesh->Dimension() : 1);
|
||||
auto dest_fe =
|
||||
make_shared<ParFiniteElementSpace>(p_dest_mesh.get(), dest_fe_coll.get());
|
||||
|
||||
ParGridFunction src_fun(src_fe.get());
|
||||
|
||||
@@ -223,7 +189,7 @@ int main(int argc, char *argv[])
|
||||
// To be used with vector fe
|
||||
VectorFunctionCoefficient vector_coeff(dim, &vector_fun);
|
||||
|
||||
if (use_vector_fe || use_vector_space)
|
||||
if (use_vector_fe)
|
||||
{
|
||||
src_fun.ProjectCoefficient(vector_coeff);
|
||||
src_fun.Update();
|
||||
@@ -243,11 +209,7 @@ int main(int argc, char *argv[])
|
||||
assemble_mass_and_coupling_together);
|
||||
assembler.SetVerbose(verbose);
|
||||
|
||||
if (use_vector_space)
|
||||
{
|
||||
assembler.AddMortarIntegrator(make_shared<LagrangeVectorL2MortarIntegrator>());
|
||||
}
|
||||
else if (use_vector_fe)
|
||||
if (use_vector_fe)
|
||||
{
|
||||
assembler.AddMortarIntegrator(make_shared<VectorL2MortarIntegrator>());
|
||||
}
|
||||
@@ -281,8 +243,8 @@ int main(int argc, char *argv[])
|
||||
<< std::endl;
|
||||
}
|
||||
|
||||
plot(*p_src_mesh, src_fun, "source", 0);
|
||||
plot(*p_dest_mesh, dest_fun, "destination", 1);
|
||||
plot(*p_src_mesh, src_fun, "source");
|
||||
plot(*p_dest_mesh, dest_fun, "destination");
|
||||
}
|
||||
}
|
||||
else
|
||||
|
||||
@@ -20,10 +20,6 @@
|
||||
#include "example_utils.hpp"
|
||||
#include "mfem.hpp"
|
||||
|
||||
#ifndef MFEM_USE_MOONOLITH
|
||||
#error This example requires that MFEM is built with MFEM_USE_MOONOLITH=YES
|
||||
#endif
|
||||
|
||||
using namespace mfem;
|
||||
using namespace std;
|
||||
|
||||
@@ -190,8 +186,8 @@ int main(int argc, char *argv[])
|
||||
<< std::endl;
|
||||
}
|
||||
|
||||
plot(*p_src_mesh, src_fun, "source", 0);
|
||||
plot(*p_dest_mesh, dest_fun, "destination", 1);
|
||||
plot(*p_src_mesh, src_fun, "source");
|
||||
plot(*p_dest_mesh, dest_fun, "destination");
|
||||
}
|
||||
}
|
||||
else
|
||||
|
||||
@@ -84,8 +84,7 @@ void vector_fun(const mfem::Vector &x, mfem::Vector &f)
|
||||
f = n;
|
||||
}
|
||||
|
||||
inline void plot(mfem::Mesh &mesh, mfem::GridFunction &x, std::string title,
|
||||
const int plot_number = 0)
|
||||
inline void plot(mfem::Mesh &mesh, mfem::GridFunction &x, std::string title)
|
||||
{
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
@@ -104,18 +103,5 @@ inline void plot(mfem::Mesh &mesh, mfem::GridFunction &x, std::string title,
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << mesh << x
|
||||
<< "window_title '"<< title << "'\n" << flush;
|
||||
|
||||
sol_sock << "window_geometry ";
|
||||
sol_sock << (plot_number * 600) << " " << 0 << " " << 600 << " " << 600 <<
|
||||
"\n";
|
||||
|
||||
if (mesh.Dimension() == 2)
|
||||
{
|
||||
sol_sock << "keys jRmclA\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
sol_sock << "keys rmclAa\n";
|
||||
}
|
||||
sol_sock << flush;
|
||||
}
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
// mpirun -np 4 ex1p -pa -d cuda --petscopts rc_ex1p_device
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
// Specifically, we discretize using a FE space of the specified
|
||||
// order, or if order < 1 using an isoparametric/isogeometric
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
// mpirun -np 4 ex6p -m ../../data/amr-quad.mesh -nonoverlapping
|
||||
//
|
||||
// Description: This is a version of Example 1 with a simple adaptive mesh
|
||||
// refinement loop. The problem being solved is again the Poisson
|
||||
// refinement loop. The problem being solved is again the Laplace
|
||||
// equation -Delta u = 1 with homogeneous Dirichlet boundary
|
||||
// conditions. The problem is solved on a sequence of meshes which
|
||||
// are locally refined in a conforming (triangles, tetrahedrons)
|
||||
@@ -131,7 +131,7 @@ int main(int argc, char *argv[])
|
||||
ParFiniteElementSpace fespace(&pmesh, &fec);
|
||||
|
||||
// 7. As in Example 1p, we set up bilinear and linear forms corresponding to
|
||||
// the Poisson problem -\Delta u = 1. We don't assemble the discrete
|
||||
// the Laplace problem -\Delta u = 1. We don't assemble the discrete
|
||||
// problem yet, this will be done in the main loop.
|
||||
ParBilinearForm a(&fespace);
|
||||
ParLinearForm b(&fespace);
|
||||
|
||||
@@ -11,7 +11,7 @@
|
||||
// creating a symbolic link to the above directory in ../../data.
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
// Specifically, we discretize using a FE space of the specified
|
||||
// order, or if order < 1 using an isoparametric/isogeometric
|
||||
|
||||
@@ -12,7 +12,7 @@
|
||||
// creating a symbolic link to the above directory in ../../data.
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
// Specifically, we discretize using a FE space of the specified
|
||||
// order, or if order < 1 using an isoparametric/isogeometric
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
// Sample runs: mpirun -np 8 ex6p
|
||||
//
|
||||
// Description: This is a version of Example 1 with a simple adaptive mesh
|
||||
// refinement loop. The problem being solved is again the Poisson
|
||||
// refinement loop. The problem being solved is again the Laplace
|
||||
// equation -Delta u = 1 with homogeneous Dirichlet boundary
|
||||
// conditions. The problem is solved on a sequence of meshes which
|
||||
// are adapted in a conforming (tetrahedrons) manner according
|
||||
|
||||
@@ -26,7 +26,7 @@
|
||||
// mpirun -np 4 ex1p -m ../../data/mobius-strip.mesh
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// simple finite element discretization of the Laplace problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
// Specifically, we discretize using a FE space of the specified
|
||||
// order, or if order < 1 using an isoparametric/isogeometric
|
||||
|
||||
@@ -35,7 +35,6 @@ set(SRCS
|
||||
integ/bilininteg_mass_ea.cpp
|
||||
integ/bilininteg_mixedcurl_pa.cpp
|
||||
integ/bilininteg_mixedvecgrad_pa.cpp
|
||||
integ/bilininteg_trace_jump_ea.cpp
|
||||
integ/bilininteg_transpose_ea.cpp
|
||||
integ/bilininteg_vecdiffusion_mf.cpp
|
||||
integ/bilininteg_vecdiffusion_pa.cpp
|
||||
@@ -47,7 +46,6 @@ set(SRCS
|
||||
integ/bilininteg_diffusion_kernels.cpp
|
||||
integ/bilininteg_elasticity_kernels.cpp
|
||||
integ/bilininteg_hcurl_kernels.cpp
|
||||
integ/bilininteg_hdiv_ea.cpp
|
||||
integ/bilininteg_hdiv_kernels.cpp
|
||||
integ/bilininteg_hcurlhdiv_kernels.cpp
|
||||
integ/bilininteg_mass_kernels.cpp
|
||||
@@ -64,9 +62,7 @@ set(SRCS
|
||||
datacollection.cpp
|
||||
dgmassinv.cpp
|
||||
doftrans.cpp
|
||||
dfem/doperator.cpp
|
||||
eltrans.cpp
|
||||
batchitrans.cpp
|
||||
estimators.cpp
|
||||
fe.cpp
|
||||
fe/face_map_utils.cpp
|
||||
@@ -77,7 +73,6 @@ set(SRCS
|
||||
fe/fe_nd.cpp
|
||||
fe/fe_nurbs.cpp
|
||||
fe/fe_pos.cpp
|
||||
fe/fe_pyramid.cpp
|
||||
fe/fe_rt.cpp
|
||||
fe/fe_ser.cpp
|
||||
fe_coll.cpp
|
||||
@@ -85,7 +80,6 @@ set(SRCS
|
||||
geom.cpp
|
||||
gridfunc.cpp
|
||||
hybridization.cpp
|
||||
hybridization_ext.cpp
|
||||
intrules.cpp
|
||||
intrules_cut.cpp
|
||||
ceed/interface/basis.cpp
|
||||
@@ -181,14 +175,6 @@ set(HDRS
|
||||
dgmassinv.hpp
|
||||
dgmassinv_kernels.hpp
|
||||
doftrans.hpp
|
||||
dfem/doperator.hpp
|
||||
dfem/fieldoperator.hpp
|
||||
dfem/integrate.hpp
|
||||
dfem/parameterspace.hpp
|
||||
dfem/qfunction_apply.hpp
|
||||
dfem/qfunction_transform.hpp
|
||||
dfem/tuple.hpp
|
||||
dfem/util.hpp
|
||||
eltrans.hpp
|
||||
estimators.hpp
|
||||
fe.hpp
|
||||
@@ -200,7 +186,6 @@ set(HDRS
|
||||
fe/fe_nd.hpp
|
||||
fe/fe_nurbs.hpp
|
||||
fe/fe_pos.hpp
|
||||
fe/fe_pyramid.hpp
|
||||
fe/fe_rt.hpp
|
||||
fe/fe_ser.hpp
|
||||
fe_coll.hpp
|
||||
@@ -209,7 +194,6 @@ set(HDRS
|
||||
geom.hpp
|
||||
gridfunc.hpp
|
||||
hybridization.hpp
|
||||
hybridization_ext.hpp
|
||||
intrules.hpp
|
||||
intrules_cut.hpp
|
||||
kernel_dispatch.hpp
|
||||
|
||||
-2035
File diff suppressed because it is too large
Load Diff
+83
-94
@@ -71,11 +71,15 @@ BilinearForm::BilinearForm(FiniteElementSpace * f)
|
||||
sequence = f->GetSequence();
|
||||
mat = mat_e = NULL;
|
||||
extern_bfs = 0;
|
||||
element_matrices = NULL;
|
||||
static_cond = NULL;
|
||||
hybridization = NULL;
|
||||
precompute_sparsity = 0;
|
||||
diag_policy = DIAG_KEEP;
|
||||
|
||||
assembly = AssemblyLevel::LEGACY;
|
||||
batch = 1;
|
||||
ext = NULL;
|
||||
}
|
||||
|
||||
BilinearForm::BilinearForm (FiniteElementSpace * f, BilinearForm * bf, int ps)
|
||||
@@ -85,11 +89,15 @@ BilinearForm::BilinearForm (FiniteElementSpace * f, BilinearForm * bf, int ps)
|
||||
sequence = f->GetSequence();
|
||||
mat_e = NULL;
|
||||
extern_bfs = 1;
|
||||
element_matrices = NULL;
|
||||
static_cond = NULL;
|
||||
hybridization = NULL;
|
||||
precompute_sparsity = ps;
|
||||
diag_policy = DIAG_KEEP;
|
||||
|
||||
assembly = AssemblyLevel::LEGACY;
|
||||
batch = 1;
|
||||
ext = NULL;
|
||||
|
||||
// Copy the pointers to the integrators
|
||||
domain_integs = bf->domain_integs;
|
||||
@@ -119,16 +127,16 @@ void BilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
break;
|
||||
case AssemblyLevel::FULL:
|
||||
SetDiagonalPolicy( DIAG_ONE ); // Only diagonal policy supported on device
|
||||
ext.reset(new FABilinearFormExtension(this));
|
||||
ext = new FABilinearFormExtension(this);
|
||||
break;
|
||||
case AssemblyLevel::ELEMENT:
|
||||
ext.reset(new EABilinearFormExtension(this));
|
||||
ext = new EABilinearFormExtension(this);
|
||||
break;
|
||||
case AssemblyLevel::PARTIAL:
|
||||
ext.reset(new PABilinearFormExtension(this));
|
||||
ext = new PABilinearFormExtension(this);
|
||||
break;
|
||||
case AssemblyLevel::NONE:
|
||||
ext.reset(new MFBilinearFormExtension(this));
|
||||
ext = new MFBilinearFormExtension(this);
|
||||
break;
|
||||
default:
|
||||
MFEM_ABORT("BilinearForm: unknown assembly level");
|
||||
@@ -137,13 +145,14 @@ void BilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
|
||||
void BilinearForm::EnableStaticCondensation()
|
||||
{
|
||||
delete static_cond;
|
||||
if (assembly != AssemblyLevel::LEGACY)
|
||||
{
|
||||
static_cond.reset();
|
||||
static_cond = NULL;
|
||||
MFEM_WARNING("Static condensation not supported for this assembly level");
|
||||
return;
|
||||
}
|
||||
static_cond.reset(new StaticCondensation(fes));
|
||||
static_cond = new StaticCondensation(fes);
|
||||
if (static_cond->ReducesTrueVSize())
|
||||
{
|
||||
bool symmetric = false; // TODO
|
||||
@@ -152,7 +161,8 @@ void BilinearForm::EnableStaticCondensation()
|
||||
}
|
||||
else
|
||||
{
|
||||
static_cond.reset();
|
||||
delete static_cond;
|
||||
static_cond = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -160,18 +170,15 @@ void BilinearForm::EnableHybridization(FiniteElementSpace *constr_space,
|
||||
BilinearFormIntegrator *constr_integ,
|
||||
const Array<int> &ess_tdof_list)
|
||||
{
|
||||
if (assembly != AssemblyLevel::LEGACY && assembly != AssemblyLevel::ELEMENT)
|
||||
delete hybridization;
|
||||
if (assembly != AssemblyLevel::LEGACY)
|
||||
{
|
||||
delete constr_integ;
|
||||
hybridization.reset();
|
||||
hybridization = NULL;
|
||||
MFEM_WARNING("Hybridization not supported for this assembly level");
|
||||
return;
|
||||
}
|
||||
hybridization.reset(new Hybridization(fes, constr_space));
|
||||
if (assembly == AssemblyLevel::ELEMENT)
|
||||
{
|
||||
hybridization->EnableDeviceExecution();
|
||||
}
|
||||
hybridization = new Hybridization(fes, constr_space);
|
||||
hybridization->SetConstraintIntegrator(constr_integ);
|
||||
hybridization->Init(ess_tdof_list);
|
||||
}
|
||||
@@ -224,8 +231,8 @@ void BilinearForm::Finalize (int skip_zeros)
|
||||
if (!static_cond) { mat->Finalize(skip_zeros); }
|
||||
if (mat_e) { mat_e->Finalize(skip_zeros); }
|
||||
if (static_cond) { static_cond->Finalize(); }
|
||||
if (hybridization) { hybridization->Finalize(); }
|
||||
}
|
||||
if (hybridization) { hybridization->Finalize(); }
|
||||
}
|
||||
|
||||
void BilinearForm::AddDomainIntegrator(BilinearFormIntegrator *bfi)
|
||||
@@ -458,14 +465,11 @@ void BilinearForm::Assemble(int skip_zeros)
|
||||
if (ext)
|
||||
{
|
||||
ext->Assemble();
|
||||
if (hybridization)
|
||||
{
|
||||
hybridization->AssembleElementMatrices(GetElementMatrices());
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
ElementTransformation *eltrans;
|
||||
DofTransformation * doftrans;
|
||||
Mesh *mesh = fes -> GetMesh();
|
||||
DenseMatrix elmat, *elmat_p;
|
||||
|
||||
@@ -502,14 +506,13 @@ void BilinearForm::Assemble(int skip_zeros)
|
||||
}
|
||||
}
|
||||
|
||||
DofTransformation doftrans;
|
||||
// Element-wise integration
|
||||
for (int i = 0; i < fes -> GetNE(); i++)
|
||||
{
|
||||
// Set both doftrans (potentially needed to assemble the element
|
||||
// matrix) and vdofs, which is also needed when the element matrices
|
||||
// are pre-assembled.
|
||||
fes->GetElementVDofs(i, vdofs, doftrans);
|
||||
doftrans = fes->GetElementVDofs(i, vdofs);
|
||||
if (element_matrices)
|
||||
{
|
||||
elmat_p = &(*element_matrices)(i);
|
||||
@@ -547,7 +550,10 @@ void BilinearForm::Assemble(int skip_zeros)
|
||||
{
|
||||
elmat_p = &elmat;
|
||||
}
|
||||
doftrans.TransformDual(elmat);
|
||||
if (doftrans)
|
||||
{
|
||||
doftrans->TransformDual(elmat);
|
||||
}
|
||||
elmat_p = &elmat;
|
||||
}
|
||||
if (static_cond)
|
||||
@@ -625,14 +631,13 @@ void BilinearForm::Assemble(int skip_zeros)
|
||||
}
|
||||
}
|
||||
|
||||
DofTransformation doftrans;
|
||||
for (int i = 0; i < fes -> GetNBE(); i++)
|
||||
{
|
||||
const int bdr_attr = mesh->GetBdrAttribute(i);
|
||||
if (bdr_attr_marker[bdr_attr-1] == 0) { continue; }
|
||||
|
||||
const FiniteElement &be = *fes->GetBE(i);
|
||||
fes -> GetBdrElementVDofs (i, vdofs, doftrans);
|
||||
doftrans = fes -> GetBdrElementVDofs (i, vdofs);
|
||||
eltrans = fes -> GetBdrElementTransformation (i);
|
||||
int k = 0;
|
||||
for (; k < boundary_integs.Size(); k++)
|
||||
@@ -652,7 +657,10 @@ void BilinearForm::Assemble(int skip_zeros)
|
||||
boundary_integs[k]->AssembleElementMatrix(be, *eltrans, elemmat);
|
||||
elmat += elemmat;
|
||||
}
|
||||
doftrans.TransformDual(elmat);
|
||||
if (doftrans)
|
||||
{
|
||||
doftrans->TransformDual(elmat);
|
||||
}
|
||||
elmat_p = &elmat;
|
||||
if (!static_cond)
|
||||
{
|
||||
@@ -827,19 +835,7 @@ void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
if (hybridization)
|
||||
{
|
||||
FormSystemMatrix(ess_tdof_list, A);
|
||||
ConstrainedOperator A_constrained(this, ess_tdof_list);
|
||||
A_constrained.EliminateRHS(x, b);
|
||||
hybridization->ReduceRHS(b, B);
|
||||
X.SetSize(B.Size());
|
||||
X = 0.0;
|
||||
}
|
||||
else
|
||||
{
|
||||
ext->FormLinearSystem(ess_tdof_list, x, b, A, X, B, copy_interior);
|
||||
}
|
||||
ext->FormLinearSystem(ess_tdof_list, x, b, A, X, B, copy_interior);
|
||||
return;
|
||||
}
|
||||
const SparseMatrix *P = fes->GetConformingProlongation();
|
||||
@@ -907,16 +903,7 @@ void BilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
if (hybridization)
|
||||
{
|
||||
const int remove_zeros = 0;
|
||||
Finalize(remove_zeros);
|
||||
A.Reset(&hybridization->GetMatrix(), false);
|
||||
}
|
||||
else
|
||||
{
|
||||
ext->FormSystemMatrix(ess_tdof_list, A);
|
||||
}
|
||||
ext->FormSystemMatrix(ess_tdof_list, A);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -957,7 +944,7 @@ void BilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
void BilinearForm::RecoverFEMSolution(const Vector &X,
|
||||
const Vector &b, Vector &x)
|
||||
{
|
||||
if (ext && !hybridization)
|
||||
if (ext)
|
||||
{
|
||||
ext->RecoverFEMSolution(X, b, x);
|
||||
return;
|
||||
@@ -1014,26 +1001,16 @@ void BilinearForm::RecoverFEMSolution(const Vector &X,
|
||||
|
||||
void BilinearForm::ComputeElementMatrices()
|
||||
{
|
||||
if (element_matrices) { return; }
|
||||
|
||||
if (auto *ea_ext = dynamic_cast<EABilinearFormExtension*>(ext.get()))
|
||||
if (element_matrices || domain_integs.Size() == 0 || fes->GetNE() == 0)
|
||||
{
|
||||
element_matrices.reset(new DenseTensor);
|
||||
ea_ext->GetElementMatrices(*element_matrices, ElementDofOrdering::NATIVE, true);
|
||||
return;
|
||||
}
|
||||
|
||||
if (domain_integs.Size() == 0 || fes->GetNE() == 0)
|
||||
{
|
||||
element_matrices.reset(new DenseTensor);
|
||||
return;
|
||||
}
|
||||
|
||||
int num_elements = fes->GetNE();
|
||||
int num_dofs_per_el = fes->GetTypicalFE()->GetDof() * fes->GetVDim();
|
||||
|
||||
element_matrices.reset(new DenseTensor(num_dofs_per_el, num_dofs_per_el,
|
||||
num_elements));
|
||||
element_matrices = new DenseTensor(num_dofs_per_el, num_dofs_per_el,
|
||||
num_elements);
|
||||
|
||||
DenseMatrix tmp;
|
||||
IsoparametricTransformation eltrans;
|
||||
@@ -1064,12 +1041,6 @@ void BilinearForm::ComputeElementMatrices()
|
||||
}
|
||||
}
|
||||
|
||||
const DenseTensor &BilinearForm::GetElementMatrices()
|
||||
{
|
||||
ComputeElementMatrices(); // Won't recompute if element_matrices exists
|
||||
return *element_matrices;
|
||||
}
|
||||
|
||||
void BilinearForm::EliminateEssentialBC(const Array<int> &bdr_attr_is_ess,
|
||||
const Vector &sol, Vector &rhs,
|
||||
DiagonalPolicy dpolicy)
|
||||
@@ -1257,13 +1228,15 @@ void BilinearForm::Update(FiniteElementSpace *nfes)
|
||||
delete mat_e;
|
||||
mat_e = NULL;
|
||||
FreeElementMatrices();
|
||||
static_cond.reset();
|
||||
delete static_cond;
|
||||
static_cond = NULL;
|
||||
|
||||
if (full_update)
|
||||
{
|
||||
delete mat;
|
||||
mat = NULL;
|
||||
hybridization.reset();
|
||||
delete hybridization;
|
||||
hybridization = NULL;
|
||||
sequence = fes->GetSequence();
|
||||
}
|
||||
else
|
||||
@@ -1286,6 +1259,9 @@ BilinearForm::~BilinearForm()
|
||||
{
|
||||
delete mat_e;
|
||||
delete mat;
|
||||
delete element_matrices;
|
||||
delete static_cond;
|
||||
delete hybridization;
|
||||
|
||||
if (!extern_bfs)
|
||||
{
|
||||
@@ -1297,6 +1273,8 @@ BilinearForm::~BilinearForm()
|
||||
for (k=0; k < boundary_face_integs.Size(); k++)
|
||||
{ delete boundary_face_integs[k]; }
|
||||
}
|
||||
|
||||
delete ext;
|
||||
}
|
||||
|
||||
|
||||
@@ -1323,6 +1301,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
|
||||
mat = NULL;
|
||||
mat_e = NULL;
|
||||
extern_bfs = 1;
|
||||
ext = NULL;
|
||||
|
||||
// Copy the pointers to the integrators
|
||||
domain_integs = mbf->domain_integs;
|
||||
@@ -1352,22 +1331,22 @@ void MixedBilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
case AssemblyLevel::LEGACY:
|
||||
break;
|
||||
case AssemblyLevel::FULL:
|
||||
// ext.reset(new FAMixedBilinearFormExtension(this));
|
||||
// ext = new FAMixedBilinearFormExtension(this);
|
||||
// Use the original BilinearForm implementation for now
|
||||
break;
|
||||
case AssemblyLevel::ELEMENT:
|
||||
MFEM_ABORT("Element assembly not supported yet... stay tuned!");
|
||||
// ext.reset(new EAMixedBilinearFormExtension(this));
|
||||
mfem_error("Element assembly not supported yet... stay tuned!");
|
||||
// ext = new EAMixedBilinearFormExtension(this);
|
||||
break;
|
||||
case AssemblyLevel::PARTIAL:
|
||||
ext.reset(new PAMixedBilinearFormExtension(this));
|
||||
ext = new PAMixedBilinearFormExtension(this);
|
||||
break;
|
||||
case AssemblyLevel::NONE:
|
||||
MFEM_ABORT("Matrix-free action not supported yet... stay tuned!");
|
||||
// ext.reset(new MFMixedBilinearFormExtension(this));
|
||||
mfem_error("Matrix-free action not supported yet... stay tuned!");
|
||||
// ext = new MFMixedBilinearFormExtension(this);
|
||||
break;
|
||||
default:
|
||||
MFEM_ABORT("Unknown assembly level");
|
||||
mfem_error("Unknown assembly level");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1525,6 +1504,8 @@ void MixedBilinearForm::Assemble(int skip_zeros)
|
||||
}
|
||||
|
||||
ElementTransformation *eltrans;
|
||||
DofTransformation * dom_dof_trans;
|
||||
DofTransformation * ran_dof_trans;
|
||||
DenseMatrix elmat;
|
||||
|
||||
Mesh *mesh = test_fes -> GetMesh();
|
||||
@@ -1547,12 +1528,11 @@ void MixedBilinearForm::Assemble(int skip_zeros)
|
||||
}
|
||||
}
|
||||
|
||||
DofTransformation dom_dof_trans, ran_dof_trans;
|
||||
for (int i = 0; i < test_fes -> GetNE(); i++)
|
||||
{
|
||||
const int elem_attr = mesh->GetAttribute(i);
|
||||
trial_fes->GetElementVDofs (i, trial_vdofs, dom_dof_trans);
|
||||
test_fes->GetElementVDofs (i, test_vdofs, ran_dof_trans);
|
||||
dom_dof_trans = trial_fes -> GetElementVDofs (i, trial_vdofs);
|
||||
ran_dof_trans = test_fes -> GetElementVDofs (i, test_vdofs);
|
||||
eltrans = test_fes -> GetElementTransformation (i);
|
||||
|
||||
elmat.SetSize(test_vdofs.Size(), trial_vdofs.Size());
|
||||
@@ -1568,7 +1548,10 @@ void MixedBilinearForm::Assemble(int skip_zeros)
|
||||
elmat += elemmat;
|
||||
}
|
||||
}
|
||||
TransformDual(ran_dof_trans, dom_dof_trans, elmat);
|
||||
if (ran_dof_trans || dom_dof_trans)
|
||||
{
|
||||
TransformDual(ran_dof_trans, dom_dof_trans, elmat);
|
||||
}
|
||||
mat -> AddSubMatrix (test_vdofs, trial_vdofs, elmat, skip_zeros);
|
||||
}
|
||||
}
|
||||
@@ -1596,14 +1579,13 @@ void MixedBilinearForm::Assemble(int skip_zeros)
|
||||
}
|
||||
}
|
||||
|
||||
DofTransformation dom_dof_trans, ran_dof_trans;
|
||||
for (int i = 0; i < test_fes -> GetNBE(); i++)
|
||||
{
|
||||
const int bdr_attr = mesh->GetBdrAttribute(i);
|
||||
if (bdr_attr_marker[bdr_attr-1] == 0) { continue; }
|
||||
|
||||
trial_fes->GetBdrElementVDofs (i, trial_vdofs, dom_dof_trans);
|
||||
test_fes->GetBdrElementVDofs (i, test_vdofs, ran_dof_trans);
|
||||
dom_dof_trans = trial_fes -> GetBdrElementVDofs (i, trial_vdofs);
|
||||
ran_dof_trans = test_fes -> GetBdrElementVDofs (i, test_vdofs);
|
||||
eltrans = test_fes -> GetBdrElementTransformation (i);
|
||||
|
||||
elmat.SetSize(test_vdofs.Size(), trial_vdofs.Size());
|
||||
@@ -1618,7 +1600,10 @@ void MixedBilinearForm::Assemble(int skip_zeros)
|
||||
*eltrans, elemmat);
|
||||
elmat += elemmat;
|
||||
}
|
||||
TransformDual(ran_dof_trans, dom_dof_trans, elmat);
|
||||
if (ran_dof_trans || dom_dof_trans)
|
||||
{
|
||||
TransformDual(ran_dof_trans, dom_dof_trans, elmat);
|
||||
}
|
||||
mat -> AddSubMatrix (test_vdofs, trial_vdofs, elmat, skip_zeros);
|
||||
}
|
||||
}
|
||||
@@ -2358,6 +2343,7 @@ MixedBilinearForm::~MixedBilinearForm()
|
||||
for (i = 0; i < boundary_trace_face_integs.Size(); i++)
|
||||
{ delete boundary_trace_face_integs[i]; }
|
||||
}
|
||||
delete ext;
|
||||
}
|
||||
|
||||
void DiscreteLinearOperator::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
@@ -2374,16 +2360,16 @@ void DiscreteLinearOperator::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
// Use the original implementation for now
|
||||
break;
|
||||
case AssemblyLevel::ELEMENT:
|
||||
MFEM_ABORT("Element assembly not supported yet... stay tuned!");
|
||||
mfem_error("Element assembly not supported yet... stay tuned!");
|
||||
break;
|
||||
case AssemblyLevel::PARTIAL:
|
||||
ext.reset(new PADiscreteLinearOperatorExtension(this));
|
||||
ext = new PADiscreteLinearOperatorExtension(this);
|
||||
break;
|
||||
case AssemblyLevel::NONE:
|
||||
MFEM_ABORT("Matrix-free action not supported yet... stay tuned!");
|
||||
mfem_error("Matrix-free action not supported yet... stay tuned!");
|
||||
break;
|
||||
default:
|
||||
MFEM_ABORT("Unknown assembly level");
|
||||
mfem_error("Unknown assembly level");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2396,6 +2382,8 @@ void DiscreteLinearOperator::Assemble(int skip_zeros)
|
||||
}
|
||||
|
||||
ElementTransformation *eltrans;
|
||||
DofTransformation * dom_dof_trans;
|
||||
DofTransformation * ran_dof_trans;
|
||||
DenseMatrix elmat;
|
||||
|
||||
Mesh *mesh = test_fes->GetMesh();
|
||||
@@ -2418,13 +2406,11 @@ void DiscreteLinearOperator::Assemble(int skip_zeros)
|
||||
}
|
||||
}
|
||||
|
||||
DofTransformation dom_dof_trans;
|
||||
DofTransformation ran_dof_trans;
|
||||
for (int i = 0; i < test_fes->GetNE(); i++)
|
||||
{
|
||||
const int elem_attr = mesh->GetAttribute(i);
|
||||
trial_fes->GetElementVDofs(i, trial_vdofs, dom_dof_trans);
|
||||
test_fes->GetElementVDofs(i, test_vdofs, ran_dof_trans);
|
||||
dom_dof_trans = trial_fes->GetElementVDofs(i, trial_vdofs);
|
||||
ran_dof_trans = test_fes->GetElementVDofs(i, test_vdofs);
|
||||
eltrans = test_fes->GetElementTransformation(i);
|
||||
|
||||
elmat.SetSize(test_vdofs.Size(), trial_vdofs.Size());
|
||||
@@ -2440,7 +2426,10 @@ void DiscreteLinearOperator::Assemble(int skip_zeros)
|
||||
elmat += elemmat;
|
||||
}
|
||||
}
|
||||
TransformPrimal(ran_dof_trans, dom_dof_trans, elemmat);
|
||||
if (ran_dof_trans || dom_dof_trans)
|
||||
{
|
||||
TransformPrimal(ran_dof_trans, dom_dof_trans, elemmat);
|
||||
}
|
||||
mat->SetSubMatrix(test_vdofs, trial_vdofs, elemmat, skip_zeros);
|
||||
}
|
||||
}
|
||||
|
||||
+13
-19
@@ -83,7 +83,7 @@ protected:
|
||||
/** @brief Extension for supporting Full Assembly (FA),
|
||||
Element Assembly (EA),Partial Assembly (PA),
|
||||
or Matrix Free assembly (MF). */
|
||||
std::unique_ptr<BilinearFormExtension> ext;
|
||||
BilinearFormExtension *ext;
|
||||
|
||||
/** Indicates if the sparse matrix is sorted after assembly when using
|
||||
Full Assembly (FA). */
|
||||
@@ -122,10 +122,10 @@ protected:
|
||||
mutable DenseMatrix elemmat;
|
||||
mutable Array<int> vdofs;
|
||||
|
||||
std::unique_ptr<DenseTensor> element_matrices;
|
||||
DenseTensor *element_matrices; ///< Owned.
|
||||
|
||||
std::unique_ptr<StaticCondensation> static_cond;
|
||||
std::unique_ptr<Hybridization> hybridization;
|
||||
StaticCondensation *static_cond; ///< Owned.
|
||||
Hybridization *hybridization; ///< Owned.
|
||||
|
||||
/** @brief This data member allows one to specify what should be done to the
|
||||
diagonal matrix entries and corresponding RHS values upon elimination of
|
||||
@@ -148,11 +148,13 @@ protected:
|
||||
BilinearForm() : Matrix (0)
|
||||
{
|
||||
fes = NULL; sequence = -1;
|
||||
mat = mat_e = NULL; extern_bfs = 0;
|
||||
mat = mat_e = NULL; extern_bfs = 0; element_matrices = NULL;
|
||||
static_cond = NULL; hybridization = NULL;
|
||||
precompute_sparsity = 0;
|
||||
diag_policy = DIAG_KEEP;
|
||||
assembly = AssemblyLevel::LEGACY;
|
||||
batch = 1;
|
||||
ext = NULL;
|
||||
}
|
||||
|
||||
private:
|
||||
@@ -212,7 +214,7 @@ public:
|
||||
/// Returns the assembly level
|
||||
AssemblyLevel GetAssemblyLevel() const { return assembly; }
|
||||
|
||||
Hybridization *GetHybridization() const { return hybridization.get(); }
|
||||
Hybridization *GetHybridization() const { return hybridization; }
|
||||
|
||||
/** @brief Enable the use of static condensation. For details see the
|
||||
description for class StaticCondensation in fem/staticcond.hpp This
|
||||
@@ -222,7 +224,7 @@ public:
|
||||
|
||||
/** @brief Check if static condensation was actually enabled by a previous
|
||||
call to EnableStaticCondensation(). */
|
||||
bool StaticCondensationIsEnabled() const { return static_cond != nullptr; }
|
||||
bool StaticCondensationIsEnabled() const { return static_cond; }
|
||||
|
||||
/// Return the trace FE space associated with static condensation.
|
||||
FiniteElementSpace *SCFESpace() const
|
||||
@@ -567,20 +569,12 @@ public:
|
||||
void RecoverFEMSolution(const Vector &X, const Vector &b,
|
||||
Vector &x) override;
|
||||
|
||||
/// @brief Compute and store internally all element matrices.
|
||||
///
|
||||
/// If AssemblyLevel::ELEMENT is selected with SetAssemblyLevel(), this will
|
||||
/// use efficient (device-accelerated) assembly of the element matrices.
|
||||
/// Compute and store internally all element matrices.
|
||||
void ComputeElementMatrices();
|
||||
|
||||
/// Free the memory used by the element matrices.
|
||||
void FreeElementMatrices() { element_matrices.reset(); }
|
||||
|
||||
/// @brief Return a DenseTensor containing the assembled element matrices.
|
||||
///
|
||||
/// If AssemblyLevel::ELEMENT is selected with SetAssemblyLevel(), this will
|
||||
/// use efficient (device-accelerated) assembly of the element matrices.
|
||||
const DenseTensor &GetElementMatrices();
|
||||
void FreeElementMatrices()
|
||||
{ delete element_matrices; element_matrices = NULL; }
|
||||
|
||||
/// Compute the element matrix of the given element
|
||||
/** The element matrix is computed by calling the domain integrators
|
||||
@@ -766,7 +760,7 @@ protected:
|
||||
|
||||
/** Extension for supporting Full Assembly (FA), Element Assembly (EA),
|
||||
Partial Assembly (PA), or Matrix Free assembly (MF). */
|
||||
std::unique_ptr<MixedBilinearFormExtension> ext;
|
||||
MixedBilinearFormExtension *ext;
|
||||
|
||||
/** @brief Indicates the BilinearFormIntegrator%s stored in
|
||||
MixedBilinearForm#domain_integs, MixedBilinearForm#boundary_integs,
|
||||
|
||||
+77
-287
@@ -16,7 +16,6 @@
|
||||
#include "bilinearform.hpp"
|
||||
#include "pbilinearform.hpp"
|
||||
#include "pgridfunc.hpp"
|
||||
#include "fe/face_map_utils.hpp"
|
||||
#include "ceed/interface/util.hpp"
|
||||
|
||||
namespace mfem
|
||||
@@ -865,137 +864,52 @@ void EABilinearFormExtension::Assemble()
|
||||
ne = trial_fes->GetMesh()->GetNE();
|
||||
elemDofs = trial_fes->GetTypicalFE()->GetDof();
|
||||
|
||||
Vector ea_data_tmp;
|
||||
ea_data.SetSize(ne*elemDofs*elemDofs, Device::GetMemoryType());
|
||||
ea_data.UseDevice(true);
|
||||
|
||||
auto add_with_markers = [&](const Vector &ea_1, Vector &ea_2, const int ne_,
|
||||
const Array<int> &markers, const Array<int> &attrs,
|
||||
const bool add)
|
||||
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
|
||||
const int integratorCount = integrators.Size();
|
||||
if ( integratorCount == 0 )
|
||||
{
|
||||
if (ne_ == 0) { return; }
|
||||
const int sz = ea_1.Size() / ne_;
|
||||
const int *d_m = markers.Read();
|
||||
const int *d_a = attrs.Read();
|
||||
const auto d_ea_1 = Reshape(ea_1.Read(), sz, ne_);
|
||||
auto d_ea_2 = Reshape(add ? ea_2.ReadWrite() : ea_2.Write(), sz, ne_);
|
||||
|
||||
mfem::forall(sz*ne_, [=] MFEM_HOST_DEVICE (int idx)
|
||||
{
|
||||
const int i = idx % sz;
|
||||
const int e = idx / sz;
|
||||
const real_t val = d_m[d_a[e] - 1] ? d_ea_1(i, e) : 0.0;
|
||||
if (add)
|
||||
{
|
||||
d_ea_2(i, e) += val;
|
||||
}
|
||||
else
|
||||
{
|
||||
d_ea_2(i, e) = val;
|
||||
}
|
||||
});
|
||||
};
|
||||
|
||||
ea_data = 0.0;
|
||||
}
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
ea_data.SetSize(ne*elemDofs*elemDofs);
|
||||
ea_data.UseDevice(true);
|
||||
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
|
||||
Array<Array<int>*> &markers_array = *a->GetDBFI_Marker();
|
||||
|
||||
if (integrators.Size() == 0) { ea_data = 0.0; }
|
||||
|
||||
for (int i = 0; i < integrators.Size(); ++i)
|
||||
{
|
||||
const bool add = (i > 0);
|
||||
const Array<int> *markers = markers_array[i];
|
||||
if (markers == nullptr)
|
||||
{
|
||||
integrators[i]->AssembleEA(*a->FESpace(), ea_data, add);
|
||||
}
|
||||
else
|
||||
{
|
||||
ea_data_tmp.SetSize(ea_data.Size());
|
||||
integrators[i]->AssembleEA(*a->FESpace(), ea_data_tmp, false);
|
||||
add_with_markers(ea_data_tmp, ea_data, ne, *markers,
|
||||
elem_attributes, add);
|
||||
}
|
||||
}
|
||||
integrators[i]->AssembleEA(*a->FESpace(), ea_data, i);
|
||||
}
|
||||
|
||||
faceDofs = trial_fes->GetTypicalTraceElement()->GetDof();
|
||||
|
||||
MFEM_VERIFY(a->GetBBFI()->Size() == 0,
|
||||
"Element assembly does not support AddBoundaryIntegrator yet.");
|
||||
|
||||
Array<BilinearFormIntegrator*> &intFaceIntegrators = *a->GetFBFI();
|
||||
const int intFaceIntegratorCount = intFaceIntegrators.Size();
|
||||
if (intFaceIntegratorCount>0)
|
||||
{
|
||||
Array<BilinearFormIntegrator*> &bdr_integs = *a->GetBBFI();
|
||||
Array<Array<int>*> &markers_array = *a->GetBBFI_Marker();
|
||||
const int n_bdr_integs = bdr_integs.Size();
|
||||
if (n_bdr_integs > 0)
|
||||
{
|
||||
nf_bdr = trial_fes->GetNFbyType(FaceType::Boundary);
|
||||
ea_data_bdr.SetSize(nf_bdr*faceDofs*faceDofs);
|
||||
}
|
||||
for (int i = 0; i < n_bdr_integs; ++i)
|
||||
{
|
||||
const bool add = (i > 0);
|
||||
const Array<int> *markers = markers_array[i];
|
||||
if (markers == nullptr)
|
||||
{
|
||||
bdr_integs[i]->AssembleEABoundary(*a->FESpace(), ea_data_bdr, add);
|
||||
}
|
||||
else
|
||||
{
|
||||
ea_data_tmp.SetSize(ea_data_bdr.Size());
|
||||
bdr_integs[i]->AssembleEABoundary(*a->FESpace(), ea_data_tmp, add);
|
||||
add_with_markers(ea_data_tmp, ea_data_bdr, nf_bdr, *markers,
|
||||
bdr_attributes, add);
|
||||
}
|
||||
}
|
||||
nf_int = trial_fes->GetNFbyType(FaceType::Interior);
|
||||
ea_data_int.SetSize(2*nf_int*faceDofs*faceDofs, Device::GetMemoryType());
|
||||
ea_data_ext.SetSize(2*nf_int*faceDofs*faceDofs, Device::GetMemoryType());
|
||||
}
|
||||
for (int i = 0; i < intFaceIntegratorCount; ++i)
|
||||
{
|
||||
intFaceIntegrators[i]->AssembleEAInteriorFaces(*a->FESpace(),
|
||||
ea_data_int,
|
||||
ea_data_ext,
|
||||
i);
|
||||
}
|
||||
|
||||
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
|
||||
const int boundFaceIntegratorCount = bdrFaceIntegrators.Size();
|
||||
if (boundFaceIntegratorCount>0)
|
||||
{
|
||||
Array<BilinearFormIntegrator*> &intFaceIntegrators = *a->GetFBFI();
|
||||
const int intFaceIntegratorCount = intFaceIntegrators.Size();
|
||||
if (intFaceIntegratorCount>0)
|
||||
{
|
||||
nf_int = trial_fes->GetNFbyType(FaceType::Interior);
|
||||
ea_data_int.SetSize(2*nf_int*faceDofs*faceDofs);
|
||||
ea_data_ext.SetSize(2*nf_int*faceDofs*faceDofs);
|
||||
}
|
||||
for (int i = 0; i < intFaceIntegratorCount; ++i)
|
||||
{
|
||||
const bool add = (i > 0);
|
||||
intFaceIntegrators[i]->AssembleEAInteriorFaces(*a->FESpace(),
|
||||
ea_data_int,
|
||||
ea_data_ext,
|
||||
add);
|
||||
}
|
||||
nf_bdr = trial_fes->GetNFbyType(FaceType::Boundary);
|
||||
ea_data_bdr.SetSize(nf_bdr*faceDofs*faceDofs, Device::GetMemoryType());
|
||||
ea_data_bdr = 0.0;
|
||||
}
|
||||
|
||||
for (int i = 0; i < boundFaceIntegratorCount; ++i)
|
||||
{
|
||||
Array<BilinearFormIntegrator*> &bdr_face_integs = *a->GetBFBFI();
|
||||
Array<Array<int>*> &markers_array = *a->GetBFBFI_Marker();
|
||||
const int n_bdr_face_integs = bdr_face_integs.Size();
|
||||
if (n_bdr_face_integs > 0)
|
||||
{
|
||||
nf_bdr = trial_fes->GetNFbyType(FaceType::Boundary);
|
||||
ea_data_bdr.SetSize(nf_bdr*faceDofs*faceDofs);
|
||||
}
|
||||
for (int i = 0; i < n_bdr_face_integs; ++i)
|
||||
{
|
||||
const bool add = (i > 0);
|
||||
const Array<int> *markers = markers_array[i];
|
||||
if (markers == nullptr)
|
||||
{
|
||||
bdr_face_integs[i]->AssembleEABoundaryFaces(
|
||||
*a->FESpace(), ea_data_bdr, add);
|
||||
}
|
||||
else
|
||||
{
|
||||
ea_data_tmp.SetSize(ea_data_bdr.Size());
|
||||
bdr_face_integs[i]->AssembleEABoundaryFaces(*a->FESpace(),
|
||||
ea_data_tmp,
|
||||
add);
|
||||
add_with_markers(ea_data_tmp, ea_data_bdr, nf_bdr, *markers,
|
||||
bdr_attributes, add);
|
||||
}
|
||||
}
|
||||
bdrFaceIntegrators[i]->AssembleEABoundaryFaces(*a->FESpace(),ea_data_bdr,i);
|
||||
}
|
||||
|
||||
if (factorize_face_terms && int_face_restrict_lex)
|
||||
@@ -1107,29 +1021,34 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
}
|
||||
|
||||
// Treatment of boundary faces
|
||||
if (!factorize_face_terms && bdr_face_restrict_lex && ea_data_bdr.Size() > 0)
|
||||
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
|
||||
const int bFISz = bdrFaceIntegrators.Size();
|
||||
if (!factorize_face_terms && bdr_face_restrict_lex && bFISz>0)
|
||||
{
|
||||
// Apply the Boundary Face Restriction
|
||||
bdr_face_restrict_lex->Mult(x, bdr_face_X);
|
||||
bdr_face_Y = 0.0;
|
||||
// Apply the boundary face matrices
|
||||
const int NDOFS = faceDofs;
|
||||
auto X = Reshape(bdr_face_X.Read(), NDOFS, nf_bdr);
|
||||
auto Y = Reshape(bdr_face_Y.ReadWrite(), NDOFS, nf_bdr);
|
||||
auto A = Reshape(ea_data_bdr.Read(), NDOFS, NDOFS, nf_bdr);
|
||||
mfem::forall(nf_bdr*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
if (bdr_face_X.Size()>0)
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
real_t res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
bdr_face_Y = 0.0;
|
||||
// Apply the boundary face matrices
|
||||
const int NDOFS = faceDofs;
|
||||
auto X = Reshape(bdr_face_X.Read(), NDOFS, nf_bdr);
|
||||
auto Y = Reshape(bdr_face_Y.ReadWrite(), NDOFS, nf_bdr);
|
||||
auto A = Reshape(ea_data_bdr.Read(), NDOFS, NDOFS, nf_bdr);
|
||||
mfem::forall(nf_bdr*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
{
|
||||
res += A(i, j, f)*X(i, f);
|
||||
}
|
||||
Y(j, f) += res;
|
||||
});
|
||||
// Apply the Boundary Face Restriction transposed
|
||||
bdr_face_restrict_lex->AddMultTransposeInPlace(bdr_face_Y, y);
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
real_t res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A(i, j, f)*X(i, f);
|
||||
}
|
||||
Y(j, f) += res;
|
||||
});
|
||||
// Apply the Boundary Face Restriction transposed
|
||||
bdr_face_restrict_lex->AddMultTransposeInPlace(bdr_face_Y, y);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1230,163 +1149,34 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
}
|
||||
|
||||
// Treatment of boundary faces
|
||||
if (!factorize_face_terms && bdr_face_restrict_lex && ea_data_bdr.Size() > 0)
|
||||
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
|
||||
const int bFISz = bdrFaceIntegrators.Size();
|
||||
if (!factorize_face_terms && bdr_face_restrict_lex && bFISz>0)
|
||||
{
|
||||
// Apply the Boundary Face Restriction
|
||||
bdr_face_restrict_lex->Mult(x, bdr_face_X);
|
||||
bdr_face_Y = 0.0;
|
||||
// Apply the boundary face matrices transposed
|
||||
const int NDOFS = faceDofs;
|
||||
auto X = Reshape(bdr_face_X.Read(), NDOFS, nf_bdr);
|
||||
auto Y = Reshape(bdr_face_Y.ReadWrite(), NDOFS, nf_bdr);
|
||||
auto A = Reshape(ea_data_bdr.Read(), NDOFS, NDOFS, nf_bdr);
|
||||
mfem::forall(nf_bdr*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
if (bdr_face_X.Size()>0)
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
real_t res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
bdr_face_Y = 0.0;
|
||||
// Apply the boundary face matrices transposed
|
||||
const int NDOFS = faceDofs;
|
||||
auto X = Reshape(bdr_face_X.Read(), NDOFS, nf_bdr);
|
||||
auto Y = Reshape(bdr_face_Y.ReadWrite(), NDOFS, nf_bdr);
|
||||
auto A = Reshape(ea_data_bdr.Read(), NDOFS, NDOFS, nf_bdr);
|
||||
mfem::forall(nf_bdr*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
{
|
||||
res += A(j, i, f)*X(i, f);
|
||||
}
|
||||
Y(j, f) += res;
|
||||
});
|
||||
// Apply the Boundary Face Restriction transposed
|
||||
bdr_face_restrict_lex->AddMultTransposeInPlace(bdr_face_Y, y);
|
||||
}
|
||||
}
|
||||
|
||||
void EABilinearFormExtension::GetElementMatrices(
|
||||
DenseTensor &element_matrices, ElementDofOrdering ordering, bool add_bdr)
|
||||
{
|
||||
// Ensure the EA data is assembled
|
||||
if (ea_data.Size() == 0) { Assemble(); }
|
||||
|
||||
const int ndofs = elemDofs;
|
||||
element_matrices.SetSize(ndofs, ndofs, ne);
|
||||
const int N = element_matrices.TotalSize();
|
||||
|
||||
const auto d_ea_data = Reshape(ea_data.Read(), ndofs, ndofs, ne);
|
||||
auto d_element_matrices = Reshape(element_matrices.Write(),
|
||||
ndofs, ndofs,
|
||||
ne);
|
||||
|
||||
const int *d_dof_map = nullptr;
|
||||
Array<int> dof_map;
|
||||
if (ordering == ElementDofOrdering::NATIVE)
|
||||
{
|
||||
const TensorBasisElement* tbe =
|
||||
dynamic_cast<const TensorBasisElement*>(trial_fes->GetFE(0));
|
||||
if (tbe)
|
||||
{
|
||||
// Deep copy to avoid issues with host device (see similar comment in
|
||||
// HybridizationExtension::ConstructC).
|
||||
dof_map = tbe->GetDofMap();
|
||||
d_dof_map = dof_map.Read();
|
||||
}
|
||||
}
|
||||
|
||||
if (d_dof_map)
|
||||
{
|
||||
// Reordering required
|
||||
mfem::forall(N, [=] MFEM_HOST_DEVICE (int idx)
|
||||
{
|
||||
const int e = idx / ndofs / ndofs;
|
||||
const int i = idx % ndofs;
|
||||
const int j = (idx / ndofs) % ndofs;
|
||||
const int ii_s = d_dof_map[i];
|
||||
const int ii = (ii_s >= 0) ? ii_s : -1 - ii_s;
|
||||
const int s_i = (ii_s >= 0) ? 1 : -1;
|
||||
const int jj_s = d_dof_map[j];
|
||||
const int jj = (jj_s >= 0) ? jj_s : -1 - jj_s;
|
||||
const int s_j = (jj_s >= 0) ? 1 : -1;
|
||||
d_element_matrices(ii, jj, e) = s_i*s_j*d_ea_data(j, i, e);
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
// No reordering required
|
||||
mfem::forall(N, [=] MFEM_HOST_DEVICE (int idx)
|
||||
{
|
||||
const int e = idx / ndofs / ndofs;
|
||||
const int i = idx % ndofs;
|
||||
const int j = (idx / ndofs) % ndofs;
|
||||
d_element_matrices(i, j, e) = d_ea_data(j, i, e);
|
||||
});
|
||||
}
|
||||
|
||||
if (add_bdr && ea_data_bdr.Size() > 0)
|
||||
{
|
||||
const int ndof_face = faceDofs;
|
||||
const auto d_ea_bdr = Reshape(ea_data_bdr.Read(),
|
||||
ndof_face, ndof_face, nf_bdr);
|
||||
|
||||
// Get all the local face maps (mapping from lexicographic face index to
|
||||
// lexicographic volume index, depending on the local face index).
|
||||
const Mesh &mesh = *trial_fes->GetMesh();
|
||||
const int dim = mesh.Dimension();
|
||||
const int n_faces_per_el = 2*dim; // assuming tensor product
|
||||
Array<int> face_maps(ndof_face * n_faces_per_el);
|
||||
for (int lf_i = 0; lf_i < n_faces_per_el; ++lf_i)
|
||||
{
|
||||
Array<int> face_map(ndof_face);
|
||||
trial_fes->GetFE(0)->GetFaceMap(lf_i, face_map);
|
||||
for (int i = 0; i < ndof_face; ++i)
|
||||
{
|
||||
face_maps[i + lf_i*ndof_face] = face_map[i];
|
||||
}
|
||||
}
|
||||
|
||||
Array<int> face_info(nf_bdr * 2);
|
||||
{
|
||||
int fidx = 0;
|
||||
for (int f = 0; f < mesh.GetNumFaces(); ++f)
|
||||
{
|
||||
Mesh::FaceInformation finfo = mesh.GetFaceInformation(f);
|
||||
if (!finfo.IsBoundary()) { continue; }
|
||||
face_info[0 + fidx*2] = finfo.element[0].local_face_id;
|
||||
face_info[1 + fidx*2] = finfo.element[0].index;
|
||||
fidx++;
|
||||
}
|
||||
}
|
||||
|
||||
const auto d_face_maps = Reshape(face_maps.Read(), ndof_face, n_faces_per_el);
|
||||
const auto d_face_info = Reshape(face_info.Read(), 2, nf_bdr);
|
||||
|
||||
const bool reorder = (ordering == ElementDofOrdering::NATIVE);
|
||||
|
||||
mfem::forall_2D(nf_bdr, ndof_face, ndof_face, [=] MFEM_HOST_DEVICE (int f)
|
||||
{
|
||||
const int lf_i = d_face_info(0, f);
|
||||
const int e = d_face_info(1, f);
|
||||
// Loop over face indices in "native ordering"
|
||||
MFEM_FOREACH_THREAD(i_lex_face, x, ndof_face)
|
||||
{
|
||||
// Convert from lexicographic face DOF to volume DOF
|
||||
const int i_lex = d_face_maps(i_lex_face, lf_i);
|
||||
|
||||
const int ii_s = d_dof_map[i_lex];
|
||||
const int ii = (ii_s >= 0) ? ii_s : -1 - ii_s;
|
||||
|
||||
const int i = reorder ? ii : i_lex;
|
||||
const int s_i = (ii_s < 0 && reorder) ? -1 : 1;
|
||||
|
||||
MFEM_FOREACH_THREAD(j_lex_face, y, ndof_face)
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
real_t res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
// Convert from lexicographic face DOF to volume DOF
|
||||
const int j_lex = d_face_maps(j_lex_face, lf_i);
|
||||
|
||||
const int jj_s = d_dof_map[j_lex];
|
||||
const int jj = (jj_s >= 0) ? jj_s : -1 - jj_s;
|
||||
|
||||
const int j = reorder ? jj : j_lex;
|
||||
const int s_j = (jj_s < 0 && reorder) ? -1 : 1;
|
||||
|
||||
AtomicAdd(d_element_matrices(i, j, e),
|
||||
s_i*s_j*d_ea_bdr(i_lex_face, j_lex_face, f));
|
||||
res += A(j, i, f)*X(i, f);
|
||||
}
|
||||
}
|
||||
});
|
||||
Y(j, f) += res;
|
||||
});
|
||||
// Apply the Boundary Face Restriction transposed
|
||||
bdr_face_restrict_lex->AddMultTransposeInPlace(bdr_face_Y, y);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -154,17 +154,6 @@ public:
|
||||
void Assemble() override;
|
||||
void Mult(const Vector &x, Vector &y) const override;
|
||||
void MultTranspose(const Vector &x, Vector &y) const override;
|
||||
|
||||
/// @brief Populates @a element_matrices with the element matrices.
|
||||
///
|
||||
/// The element matrices are converted from row-major (how they are stored in
|
||||
/// @a ea_data) to column-major format.
|
||||
///
|
||||
/// If @a ordering is ElementDofOrdering::NATIVE, then the matrices are
|
||||
/// reordered from the lexicographic ordering used internally.
|
||||
void GetElementMatrices(DenseTensor &element_matrices,
|
||||
ElementDofOrdering ordering,
|
||||
bool add_bdr);
|
||||
};
|
||||
|
||||
/// Data and methods for fully-assembled bilinear forms
|
||||
|
||||
+5
-31
@@ -72,14 +72,6 @@ void BilinearFormIntegrator::AssembleEA(const FiniteElementSpace &fes,
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleEABoundary(const FiniteElementSpace &fes,
|
||||
Vector &emat,
|
||||
const bool add)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleEABoundary(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace
|
||||
&fes,
|
||||
Vector &ea_data_int,
|
||||
@@ -90,16 +82,6 @@ void BilinearFormIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleEAInteriorFaces(
|
||||
const FiniteElementSpace &trial_fes,
|
||||
const FiniteElementSpace &test_fes,
|
||||
Vector &emat,
|
||||
const bool add)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleEAInteriorFaces(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleEABoundaryFaces(const FiniteElementSpace
|
||||
&fes,
|
||||
Vector &ea_data_bdr,
|
||||
@@ -2847,18 +2829,17 @@ void VectorDivergenceIntegrator::AssembleElementMatrix2(
|
||||
DenseMatrix &elmat)
|
||||
{
|
||||
dim = trial_fe.GetDim();
|
||||
sdim = Trans.GetSpaceDim();
|
||||
int trial_dof = trial_fe.GetDof();
|
||||
int test_dof = test_fe.GetDof();
|
||||
real_t c;
|
||||
|
||||
dshape.SetSize (trial_dof, dim);
|
||||
gshape.SetSize (trial_dof, sdim);
|
||||
Jadj.SetSize (dim, sdim);
|
||||
divshape.SetSize (sdim*trial_dof);
|
||||
gshape.SetSize (trial_dof, dim);
|
||||
Jadj.SetSize (dim);
|
||||
divshape.SetSize (dim*trial_dof);
|
||||
shape.SetSize (test_dof);
|
||||
|
||||
elmat.SetSize (test_dof, sdim*trial_dof);
|
||||
elmat.SetSize (test_dof, dim*trial_dof);
|
||||
|
||||
const IntegrationRule *ir = GetIntegrationRule(trial_fe, test_fe, Trans);
|
||||
|
||||
@@ -2872,15 +2853,13 @@ void VectorDivergenceIntegrator::AssembleElementMatrix2(
|
||||
trial_fe.CalcDShape (ip, dshape);
|
||||
test_fe.CalcPhysShape (Trans, shape);
|
||||
|
||||
// AdjugateJacobian = / adj(J), if J is square
|
||||
// \ adj(J^t.J).J^t, otherwise
|
||||
CalcAdjugate(Trans.Jacobian(), Jadj);
|
||||
|
||||
Mult (dshape, Jadj, gshape);
|
||||
|
||||
gshape.GradToDiv (divshape);
|
||||
|
||||
c = ip.weight;
|
||||
if (dim != sdim) { c /= Trans.Weight(); }
|
||||
if (Q)
|
||||
{
|
||||
c *= Q -> Eval (Trans, ip);
|
||||
@@ -2921,11 +2900,6 @@ void DivDivIntegrator::AssembleElementMatrix(
|
||||
if (ir == NULL)
|
||||
{
|
||||
int order = 2 * el.GetOrder() - 2; // <--- OK for RTk
|
||||
if (el.Space() == FunctionSpace::Uk)
|
||||
{
|
||||
order += 2;
|
||||
}
|
||||
|
||||
ir = &IntRules.Get(el.GetGeomType(), order);
|
||||
}
|
||||
|
||||
|
||||
+7
-39
@@ -124,25 +124,11 @@ public:
|
||||
/// Assemble diagonal and add it to Vector @a diag.
|
||||
virtual void AssembleDiagonalMF(Vector &diag);
|
||||
|
||||
virtual void AssembleEABoundary(const FiniteElementSpace &fes,
|
||||
Vector &ea_data_bdr,
|
||||
const bool add = true);
|
||||
|
||||
virtual void AssembleEAInteriorFaces(const FiniteElementSpace &fes,
|
||||
Vector &ea_data_int,
|
||||
Vector &ea_data_ext,
|
||||
const bool add = true);
|
||||
|
||||
/// @brief Method defining element assembly for mixed trace integrators.
|
||||
///
|
||||
/// This is the element assembly analogue of AssembleFaceMatrix(const
|
||||
/// FiniteElement&, const FiniteElement&, const FiniteElement&,
|
||||
/// FaceElementTransformations&, DenseMatrix&).
|
||||
virtual void AssembleEAInteriorFaces(const FiniteElementSpace &trial_fes,
|
||||
const FiniteElementSpace &test_fes,
|
||||
Vector &emat,
|
||||
const bool add = true);
|
||||
|
||||
virtual void AssembleEABoundaryFaces(const FiniteElementSpace &fes,
|
||||
Vector &ea_data_bdr,
|
||||
const bool add = true);
|
||||
@@ -397,7 +383,6 @@ public:
|
||||
void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
|
||||
const bool add) override;
|
||||
|
||||
using BilinearFormIntegrator::AssembleEAInteriorFaces;
|
||||
void AssembleEAInteriorFaces(const FiniteElementSpace &fes,
|
||||
Vector &ea_data_int,
|
||||
Vector &ea_data_ext,
|
||||
@@ -509,7 +494,6 @@ public:
|
||||
void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
|
||||
const bool add) override;
|
||||
|
||||
using BilinearFormIntegrator::AssembleEAInteriorFaces;
|
||||
void AssembleEAInteriorFaces(const FiniteElementSpace &fes,
|
||||
Vector &ea_data_int,
|
||||
Vector &ea_data_ext,
|
||||
@@ -2366,8 +2350,6 @@ protected:
|
||||
const FaceGeometricFactors *face_geom; ///< Not owned
|
||||
int dim, ne, nq, dofs1D, quad1D;
|
||||
|
||||
void AssembleEA_(Vector &ea, const bool add);
|
||||
|
||||
public:
|
||||
|
||||
using ApplyKernelType = void(*)(const int, const Array<real_t>&,
|
||||
@@ -2408,10 +2390,7 @@ public:
|
||||
void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
|
||||
const bool add) override;
|
||||
|
||||
virtual void AssembleEABoundary(const FiniteElementSpace &fes, Vector &emat,
|
||||
const bool add) override;
|
||||
|
||||
virtual void AssembleDiagonalPA(Vector &diag) override;
|
||||
void AssembleDiagonalPA(Vector &diag) override;
|
||||
|
||||
void AssembleDiagonalMF(Vector &diag) override;
|
||||
|
||||
@@ -2935,8 +2914,6 @@ public:
|
||||
void AddMultPA(const Vector &x, Vector &y) const override;
|
||||
void AddMultTransposePA(const Vector &x, Vector &y) const override;
|
||||
void AssembleDiagonalPA(Vector& diag) override;
|
||||
void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
|
||||
const bool add) override;
|
||||
|
||||
const Coefficient *GetCoefficient() const { return Q; }
|
||||
};
|
||||
@@ -2958,7 +2935,7 @@ private:
|
||||
Vector pa_data;
|
||||
const DofToQuad *trial_maps, *test_maps; ///< Not owned
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
int dim, sdim, ne, nq;
|
||||
int dim, ne, nq;
|
||||
int trial_dofs1D, test_dofs1D, quad1D;
|
||||
|
||||
public:
|
||||
@@ -3004,6 +2981,11 @@ class DivDivIntegrator: public BilinearFormIntegrator
|
||||
protected:
|
||||
Coefficient *Q;
|
||||
|
||||
using BilinearFormIntegrator::AssemblePA;
|
||||
void AssemblePA(const FiniteElementSpace &fes) override;
|
||||
void AddMultPA(const Vector &x, Vector &y) const override;
|
||||
void AssembleDiagonalPA(Vector& diag) override;
|
||||
|
||||
private:
|
||||
#ifndef MFEM_THREAD_SAFE
|
||||
Vector divshape, te_divshape;
|
||||
@@ -3030,13 +3012,6 @@ public:
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &elmat) override;
|
||||
|
||||
using BilinearFormIntegrator::AssemblePA;
|
||||
void AssemblePA(const FiniteElementSpace &fes) override;
|
||||
void AddMultPA(const Vector &x, Vector &y) const override;
|
||||
void AssembleDiagonalPA(Vector& diag) override;
|
||||
void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
|
||||
const bool add) override;
|
||||
|
||||
const Coefficient *GetCoefficient() const { return Q; }
|
||||
};
|
||||
|
||||
@@ -3336,7 +3311,6 @@ public:
|
||||
|
||||
void AddMultPA(const Vector&, Vector&) const override;
|
||||
|
||||
using BilinearFormIntegrator::AssembleEAInteriorFaces;
|
||||
void AssembleEAInteriorFaces(const FiniteElementSpace& fes,
|
||||
Vector &ea_data_int,
|
||||
Vector &ea_data_ext,
|
||||
@@ -3650,12 +3624,6 @@ public:
|
||||
const FiniteElement &test_fe2,
|
||||
FaceElementTransformations &Trans,
|
||||
DenseMatrix &elmat) override;
|
||||
|
||||
using BilinearFormIntegrator::AssembleEAInteriorFaces;
|
||||
void AssembleEAInteriorFaces(const FiniteElementSpace &trial_fes,
|
||||
const FiniteElementSpace &test_fes,
|
||||
Vector &emat,
|
||||
const bool add = true) override;
|
||||
};
|
||||
|
||||
/** Integrator for the DPG form:$ \langle v, w \rangle $ over a face (the interface) where
|
||||
|
||||
+2
-7
@@ -240,9 +240,7 @@ public:
|
||||
Vector argument instead of Vector. */
|
||||
MFEM_DEPRECATED FunctionCoefficient(real_t (*f)(Vector &))
|
||||
{
|
||||
// Cast first to (void*) to suppress a warning from newer version of
|
||||
// Clang when using -Wextra.
|
||||
Function = reinterpret_cast<real_t(*)(const Vector&)>((void*)f);
|
||||
Function = reinterpret_cast<real_t(*)(const Vector&)>(f);
|
||||
TDFunction = NULL;
|
||||
}
|
||||
|
||||
@@ -252,10 +250,7 @@ public:
|
||||
MFEM_DEPRECATED FunctionCoefficient(real_t (*tdf)(Vector &, real_t))
|
||||
{
|
||||
Function = NULL;
|
||||
// Cast first to (void*) to suppress a warning from newer version of
|
||||
// Clang when using -Wextra.
|
||||
TDFunction =
|
||||
reinterpret_cast<real_t(*)(const Vector&,real_t)>((void*)tdf);
|
||||
TDFunction = reinterpret_cast<real_t(*)(const Vector&,real_t)>(tdf);
|
||||
}
|
||||
|
||||
/// Evaluate the coefficient at @a ip.
|
||||
|
||||
+4
-4
@@ -487,7 +487,7 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
// Modify RHS and off-diagonal blocks (imaginary parts of the matrix) to
|
||||
// Modify RHS and offdiagonal blocks (imaginary parts of the matrix) to
|
||||
// conform with standard essential BC treatment
|
||||
if (A_i.Is<ConstrainedOperator>())
|
||||
{
|
||||
@@ -576,7 +576,7 @@ SesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
// Modify off-diagonal blocks (imaginary parts of the matrix) to conform
|
||||
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
|
||||
// with standard essential BC treatment
|
||||
if (A_i.Is<ConstrainedOperator>())
|
||||
{
|
||||
@@ -1236,7 +1236,7 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
d_B_r[j] = d_X_r[j];
|
||||
d_B_i[j] = d_X_i[j];
|
||||
});
|
||||
// Modify off-diagonal blocks (imaginary parts of the matrix) to conform
|
||||
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
|
||||
// with standard essential BC treatment
|
||||
if (A_i.Type() == Operator::Hypre_ParCSR)
|
||||
{
|
||||
@@ -1324,7 +1324,7 @@ ParSesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
// Modify off-diagonal blocks (imaginary parts of the matrix) to conform
|
||||
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
|
||||
// with standard essential BC treatment
|
||||
if ( A_i.Type() == Operator::Hypre_ParCSR )
|
||||
{
|
||||
|
||||
+52
-118
@@ -12,7 +12,6 @@
|
||||
#include "fem.hpp"
|
||||
#include "../mesh/nurbs.hpp"
|
||||
#include "../mesh/vtk.hpp"
|
||||
#include "../mesh/vtkhdf.hpp"
|
||||
#include "../general/binaryio.hpp"
|
||||
#include "../general/text.hpp"
|
||||
#include "picojson.h"
|
||||
@@ -759,10 +758,18 @@ void VisItDataCollection::ParseVisItRootString(const std::string& json)
|
||||
}
|
||||
}
|
||||
|
||||
ParaViewDataCollectionBase::ParaViewDataCollectionBase(
|
||||
const std::string &name, Mesh *mesh) : DataCollection(name, mesh)
|
||||
ParaViewDataCollection::ParaViewDataCollection(const std::string&
|
||||
collection_name,
|
||||
Mesh *mesh_)
|
||||
: DataCollection(collection_name, mesh_),
|
||||
levels_of_detail(1),
|
||||
pv_data_format(VTKFormat::BINARY),
|
||||
high_order_output(false),
|
||||
restart_mode(false)
|
||||
{
|
||||
cycle = 0;
|
||||
cycle = 0; // always include a valid cycle index in file names
|
||||
|
||||
compression_level = -1; // default zlib compression level, equivalent to 6
|
||||
#ifdef MFEM_USE_ZLIB
|
||||
compression = true; // if we have zlib, enable compression
|
||||
#else
|
||||
@@ -770,53 +777,16 @@ ParaViewDataCollectionBase::ParaViewDataCollectionBase(
|
||||
#endif
|
||||
}
|
||||
|
||||
void ParaViewDataCollectionBase::SetLevelsOfDetail(int levels_of_detail_)
|
||||
void ParaViewDataCollection::SetLevelsOfDetail(int levels_of_detail_)
|
||||
{
|
||||
levels_of_detail = levels_of_detail_;
|
||||
}
|
||||
|
||||
void ParaViewDataCollectionBase::SetHighOrderOutput(bool high_order_output_)
|
||||
void ParaViewDataCollection::Load(int )
|
||||
{
|
||||
high_order_output = high_order_output_;
|
||||
MFEM_WARNING("ParaViewDataCollection::Load() is not implemented!");
|
||||
}
|
||||
|
||||
void ParaViewDataCollectionBase::SetCompressionLevel(int compression_level_)
|
||||
{
|
||||
MFEM_ASSERT(compression_level_ >= -1 && compression_level_ <= 9,
|
||||
"Compression level must be between -1 and 9 (inclusive).");
|
||||
compression_level = compression_level_;
|
||||
compression = compression_level_ != 0;
|
||||
}
|
||||
|
||||
void ParaViewDataCollectionBase::SetCompression(bool compression_)
|
||||
{
|
||||
compression = compression_;
|
||||
}
|
||||
|
||||
int ParaViewDataCollectionBase::GetCompressionLevel() const
|
||||
{
|
||||
return compression ? compression_level : 0;
|
||||
}
|
||||
|
||||
void ParaViewDataCollectionBase::SetDataFormat(VTKFormat fmt)
|
||||
{
|
||||
pv_data_format = fmt;
|
||||
}
|
||||
|
||||
bool ParaViewDataCollectionBase::IsBinaryFormat() const
|
||||
{
|
||||
return pv_data_format != VTKFormat::ASCII;
|
||||
}
|
||||
|
||||
void ParaViewDataCollectionBase::UseRestartMode(bool restart_mode_)
|
||||
{
|
||||
restart_mode = restart_mode_;
|
||||
}
|
||||
|
||||
ParaViewDataCollection::ParaViewDataCollection(
|
||||
const std::string& collection_name, Mesh *mesh_)
|
||||
: ParaViewDataCollectionBase(collection_name, mesh_) { }
|
||||
|
||||
std::string ParaViewDataCollection::GenerateCollectionPath()
|
||||
{
|
||||
return prefix_path + DataCollection::GetCollectionName();
|
||||
@@ -931,7 +901,7 @@ void ParaViewDataCollection::Save()
|
||||
// Initialize new pvd file.
|
||||
pvd_stream.open(pvdname,std::ios::out|std::ios::trunc);
|
||||
pvd_stream << "<?xml version=\"1.0\"?>\n";
|
||||
pvd_stream << "<VTKFile type=\"Collection\" version=\"2.2\"";
|
||||
pvd_stream << "<VTKFile type=\"Collection\" version=\"0.1\"";
|
||||
pvd_stream << " byte_order=\"" << VTKByteOrder() << "\">\n";
|
||||
pvd_stream << "<Collection>" << std::endl;
|
||||
}
|
||||
@@ -1031,7 +1001,7 @@ void ParaViewDataCollection::WritePVTUHeader(std::ostream &os)
|
||||
{
|
||||
os << "<?xml version=\"1.0\"?>\n";
|
||||
os << "<VTKFile type=\"PUnstructuredGrid\"";
|
||||
os << " version =\"2.2\" byte_order=\"" << VTKByteOrder() << "\">\n";
|
||||
os << " version =\"0.1\" byte_order=\"" << VTKByteOrder() << "\">\n";
|
||||
os << "<PUnstructuredGrid GhostLevel=\"0\">\n";
|
||||
|
||||
os << "<PPoints>\n";
|
||||
@@ -1072,7 +1042,7 @@ void ParaViewDataCollection::SaveDataVTU(std::ostream &os, int ref)
|
||||
{
|
||||
os << " compressor=\"vtkZLibDataCompressor\"";
|
||||
}
|
||||
os << " version=\"2.2\" byte_order=\"" << VTKByteOrder() << "\">\n";
|
||||
os << " version=\"0.1\" byte_order=\"" << VTKByteOrder() << "\">\n";
|
||||
os << "<UnstructuredGrid>\n";
|
||||
mesh->PrintVTU(os,ref,pv_data_format,high_order_output,GetCompressionLevel());
|
||||
|
||||
@@ -1145,6 +1115,39 @@ void ParaViewDataCollection::SaveGFieldVTU(std::ostream &os, int ref_,
|
||||
os << "</DataArray>" << std::endl;
|
||||
}
|
||||
|
||||
void ParaViewDataCollection::SetDataFormat(VTKFormat fmt)
|
||||
{
|
||||
pv_data_format = fmt;
|
||||
}
|
||||
|
||||
bool ParaViewDataCollection::IsBinaryFormat() const
|
||||
{
|
||||
return pv_data_format != VTKFormat::ASCII;
|
||||
}
|
||||
|
||||
void ParaViewDataCollection::SetHighOrderOutput(bool high_order_output_)
|
||||
{
|
||||
high_order_output = high_order_output_;
|
||||
}
|
||||
|
||||
void ParaViewDataCollection::SetCompressionLevel(int compression_level_)
|
||||
{
|
||||
MFEM_ASSERT(compression_level_ >= -1 && compression_level_ <= 9,
|
||||
"Compression level must be between -1 and 9 (inclusive).");
|
||||
compression_level = compression_level_;
|
||||
compression = compression_level_ != 0;
|
||||
}
|
||||
|
||||
void ParaViewDataCollection::SetCompression(bool compression_)
|
||||
{
|
||||
compression = compression_;
|
||||
}
|
||||
|
||||
void ParaViewDataCollection::UseRestartMode(bool restart_mode_)
|
||||
{
|
||||
restart_mode = restart_mode_;
|
||||
}
|
||||
|
||||
const char *ParaViewDataCollection::GetDataFormatString() const
|
||||
{
|
||||
if (pv_data_format == VTKFormat::ASCII)
|
||||
@@ -1169,78 +1172,9 @@ const char *ParaViewDataCollection::GetDataTypeString() const
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_HDF5
|
||||
|
||||
ParaViewHDFDataCollection::ParaViewHDFDataCollection(
|
||||
const std::string &collection_name, Mesh *mesh)
|
||||
: ParaViewDataCollectionBase(collection_name, mesh)
|
||||
{ }
|
||||
|
||||
void ParaViewHDFDataCollection::EnsureVTKHDF()
|
||||
int ParaViewDataCollection::GetCompressionLevel() const
|
||||
{
|
||||
if (!vtkhdf)
|
||||
{
|
||||
if (!prefix_path.empty())
|
||||
{
|
||||
const int error_code = create_directory(prefix_path, mesh, myid);
|
||||
MFEM_VERIFY(error_code == 0, "Error creating directory " << prefix_path);
|
||||
}
|
||||
|
||||
std::string fname = prefix_path + name + ".vtkhdf";
|
||||
bool use_mpi = false;
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (ParMesh *pmesh = dynamic_cast<ParMesh*>(mesh))
|
||||
{
|
||||
use_mpi = true;
|
||||
#ifdef MFEM_PARALLEL_HDF5
|
||||
vtkhdf.reset(new VTKHDF(fname, pmesh->GetComm(), {restart_mode, time}));
|
||||
#else
|
||||
MFEM_ABORT("Requires HDF5 library with parallel support enabled");
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
if (!use_mpi)
|
||||
{
|
||||
vtkhdf.reset(new VTKHDF(fname, {restart_mode, time}));
|
||||
}
|
||||
}
|
||||
return compression ? compression_level : 0;
|
||||
}
|
||||
|
||||
template <typename FP_T>
|
||||
void ParaViewHDFDataCollection::TSave()
|
||||
{
|
||||
EnsureVTKHDF();
|
||||
|
||||
if (compression)
|
||||
{
|
||||
vtkhdf->EnableCompression(compression_level >= 0 ? compression_level : 6);
|
||||
}
|
||||
else
|
||||
{
|
||||
vtkhdf->DisableCompression();
|
||||
}
|
||||
|
||||
vtkhdf->SaveMesh<FP_T>(*mesh, high_order_output, levels_of_detail);
|
||||
for (const auto &field : field_map)
|
||||
{
|
||||
vtkhdf->SaveGridFunction<FP_T>(*field.second, field.first);
|
||||
}
|
||||
vtkhdf->UpdateSteps(time);
|
||||
vtkhdf->Flush();
|
||||
}
|
||||
|
||||
void ParaViewHDFDataCollection::Save()
|
||||
{
|
||||
switch (pv_data_format)
|
||||
{
|
||||
case VTKFormat::BINARY32: TSave<float>(); break;
|
||||
case VTKFormat::BINARY: TSave<double>(); break;
|
||||
default: MFEM_ABORT("Unsupported VTK format.");
|
||||
}
|
||||
}
|
||||
|
||||
ParaViewHDFDataCollection::~ParaViewHDFDataCollection() = default;
|
||||
|
||||
#endif
|
||||
|
||||
} // end namespace MFEM
|
||||
|
||||
+63
-112
@@ -502,27 +502,60 @@ public:
|
||||
};
|
||||
|
||||
|
||||
/// Abstract base class for ParaViewDataCollection and ParaViewHDFDataCollection
|
||||
class ParaViewDataCollectionBase : public DataCollection
|
||||
/// Helper class for ParaView visualization data
|
||||
class ParaViewDataCollection : public DataCollection
|
||||
{
|
||||
protected:
|
||||
int levels_of_detail = 1;
|
||||
int compression_level = -1;
|
||||
bool high_order_output = false;
|
||||
bool restart_mode = false;
|
||||
VTKFormat pv_data_format = VTKFormat::BINARY;
|
||||
public:
|
||||
ParaViewDataCollectionBase(const std::string &name, Mesh *mesh);
|
||||
private:
|
||||
int levels_of_detail;
|
||||
int compression_level;
|
||||
std::fstream pvd_stream;
|
||||
VTKFormat pv_data_format;
|
||||
bool high_order_output;
|
||||
bool restart_mode;
|
||||
|
||||
/// @brief Set the refinement level.
|
||||
///
|
||||
/// In "low-order mode", every element is uniformly split based on the levels
|
||||
/// of detail. In "high-order mode", this sets the polynomial degree for the
|
||||
/// element transformations.
|
||||
///
|
||||
/// The initial value is 1.
|
||||
protected:
|
||||
void WritePVTUHeader(std::ostream &out);
|
||||
void WritePVTUFooter(std::ostream &out, const std::string &vtu_prefix);
|
||||
void SaveDataVTU(std::ostream &out, int ref);
|
||||
void SaveGFieldVTU(std::ostream& out, int ref_, const FieldMapIterator& it);
|
||||
const char *GetDataFormatString() const;
|
||||
const char *GetDataTypeString() const;
|
||||
/// @brief If compression is enabled, return the compression level, otherwise
|
||||
/// return 0.
|
||||
int GetCompressionLevel() const;
|
||||
|
||||
std::string GenerateCollectionPath();
|
||||
std::string GenerateVTUFileName(const std::string &prefix, int rank);
|
||||
std::string GenerateVTUPath();
|
||||
std::string GeneratePVDFileName();
|
||||
std::string GeneratePVTUFileName(const std::string &prefix);
|
||||
std::string GeneratePVTUPath();
|
||||
|
||||
|
||||
public:
|
||||
/// Constructor. The collection name is used when saving the data.
|
||||
/** If @a mesh_ is NULL, then the mesh can be set later by calling SetMesh().
|
||||
Before saving the data collection, some parameters in the collection can
|
||||
be adjusted, e.g. SetPadDigits(), SetPrefixPath(), etc. */
|
||||
ParaViewDataCollection(const std::string& collection_name,
|
||||
mfem::Mesh *mesh_ = NULL);
|
||||
|
||||
/// Set refinement levels - every element is uniformly split based on
|
||||
/// levels_of_detail_. The initial value is 1.
|
||||
void SetLevelsOfDetail(int levels_of_detail_);
|
||||
|
||||
/// Save the collection - the directory name is constructed based on the
|
||||
/// cycle value
|
||||
void Save() override;
|
||||
|
||||
/// Set the data format for the ParaView output files. Possible options are
|
||||
/// VTKFormat::ASCII, VTKFormat::BINARY, and VTKFormat::BINARY32.
|
||||
/// The ASCII and BINARY options output double precision data, whereas the
|
||||
/// BINARY32 option outputs single precision data.
|
||||
///
|
||||
/// The initial format is VTKFormat::BINARY.
|
||||
void SetDataFormat(VTKFormat fmt);
|
||||
|
||||
/// @brief Set the zlib compression level.
|
||||
///
|
||||
/// 0 indicates no compression, -1 indicates the default compression level.
|
||||
@@ -537,110 +570,28 @@ public:
|
||||
/// Any nonzero compression level will enable compression.
|
||||
void SetCompressionLevel(int compression_level_);
|
||||
|
||||
/// @brief Enable or disable zlib compression.
|
||||
///
|
||||
/// If the input is true, use the default zlib compression level (unless the
|
||||
/// compression level has previously been set by calling
|
||||
/// SetCompressionLevel()).
|
||||
/// Enable or disable zlib compression. If the input is true, use the default
|
||||
/// zlib compression level (unless the compression level has previously been
|
||||
/// set by calling SetCompressionLevel()).
|
||||
void SetCompression(bool compression_) override;
|
||||
|
||||
/// @brief Sets whether or not to output the data as high-order elements
|
||||
/// (false by default).
|
||||
///
|
||||
/// Reading high-order data requires ParaView 5.5 or later.
|
||||
void SetHighOrderOutput(bool high_order_output_);
|
||||
|
||||
/// If compression is enabled, return the compression level, else return 0.
|
||||
int GetCompressionLevel() const;
|
||||
|
||||
/// @brief Set the data format for the ParaView output files.
|
||||
///
|
||||
/// Possible options are VTKFormat::ASCII, VTKFormat::BINARY, and
|
||||
/// VTKFormat::BINARY32. The ASCII and BINARY options output double precision
|
||||
/// data, whereas the BINARY32 option outputs single precision data.
|
||||
///
|
||||
/// The initial format is VTKFormat::BINARY.
|
||||
///
|
||||
/// VTKFormat::ASCII is not supported by ParaViewHDFDataCollection.
|
||||
void SetDataFormat(VTKFormat fmt);
|
||||
|
||||
/// Returns true if the output format is BINARY or BINARY32, false if ASCII.
|
||||
bool IsBinaryFormat() const;
|
||||
|
||||
/// @brief Enable or disable restart mode.
|
||||
///
|
||||
/// If restart is enabled, new writes will preserve timestep metadata for any
|
||||
/// solutions prior to the currently defined time.
|
||||
/// Sets whether or not to output the data as high-order elements (false
|
||||
/// by default). Reading high-order data requires ParaView 5.5 or later.
|
||||
void SetHighOrderOutput(bool high_order_output_);
|
||||
|
||||
/// Enable or disable restart mode. If restart is enabled, new writes will
|
||||
/// preserve timestep metadata for any solutions prior to the currently
|
||||
/// defined time.
|
||||
///
|
||||
/// Initially, restart mode is disabled.
|
||||
void UseRestartMode(bool restart_mode_);
|
||||
|
||||
/// Load the collection - not implemented in the ParaView writer
|
||||
void Load(int cycle_ = 0) override;
|
||||
};
|
||||
|
||||
/// Writer for ParaView visualization (PVD and VTU format)
|
||||
class ParaViewDataCollection : public ParaViewDataCollectionBase
|
||||
{
|
||||
private:
|
||||
std::fstream pvd_stream;
|
||||
|
||||
protected:
|
||||
void WritePVTUHeader(std::ostream &out);
|
||||
void WritePVTUFooter(std::ostream &out, const std::string &vtu_prefix);
|
||||
void SaveDataVTU(std::ostream &out, int ref);
|
||||
void SaveGFieldVTU(std::ostream& out, int ref_, const FieldMapIterator& it);
|
||||
const char *GetDataFormatString() const;
|
||||
const char *GetDataTypeString() const;
|
||||
|
||||
std::string GenerateCollectionPath();
|
||||
std::string GenerateVTUFileName(const std::string &prefix, int rank);
|
||||
std::string GenerateVTUPath();
|
||||
std::string GeneratePVDFileName();
|
||||
std::string GeneratePVTUFileName(const std::string &prefix);
|
||||
std::string GeneratePVTUPath();
|
||||
|
||||
public:
|
||||
/// Constructor. The collection name is used when saving the data.
|
||||
/** If @a mesh_ is NULL, then the mesh can be set later by calling SetMesh().
|
||||
Before saving the data collection, some parameters in the collection can
|
||||
be adjusted, e.g. SetPadDigits(), SetPrefixPath(), etc. */
|
||||
ParaViewDataCollection(const std::string& collection_name,
|
||||
Mesh *mesh_ = nullptr);
|
||||
|
||||
/// Save the collection - the directory name is constructed based on the
|
||||
/// cycle value
|
||||
void Save() override;
|
||||
};
|
||||
|
||||
#ifdef MFEM_USE_HDF5
|
||||
|
||||
/// Writer for ParaView visualization (%VTKHDF format)
|
||||
class ParaViewHDFDataCollection : public ParaViewDataCollectionBase
|
||||
{
|
||||
/// The low-level VTKHDF object for I/O (pointer to implementation idiom).
|
||||
std::unique_ptr<class VTKHDF> vtkhdf;
|
||||
|
||||
/// Create the VTKHDF object if it doesn't exist already.
|
||||
void EnsureVTKHDF();
|
||||
|
||||
/// Save the collection (templated on floating point type).
|
||||
template <typename FP_T> void TSave();
|
||||
|
||||
public:
|
||||
/// @brief Constructor. The collection name is used when saving the data.
|
||||
///
|
||||
/// If @a mesh_ is NULL, then the mesh can be set later by calling SetMesh().
|
||||
/// Before saving the data collection, some parameters in the collection can
|
||||
/// be adjusted, e.g. SetPadDigits(), SetPrefixPath(), etc.
|
||||
ParaViewHDFDataCollection(const std::string& collection_name,
|
||||
Mesh *mesh_ = nullptr);
|
||||
|
||||
/// Save the collection.
|
||||
void Save() override;
|
||||
|
||||
/// Destructor.
|
||||
~ParaViewHDFDataCollection();
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1,54 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "doperator.hpp"
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
using namespace mfem;
|
||||
using namespace mfem::future;
|
||||
|
||||
void DifferentiableOperator::SetParameters(std::vector<Vector *> p) const
|
||||
{
|
||||
MFEM_ASSERT(parameters.size() == p.size(),
|
||||
"number of parameters doesn't match descriptors");
|
||||
for (size_t i = 0; i < parameters.size(); i++)
|
||||
{
|
||||
p[i]->Read();
|
||||
parameters_l[i] = *p[i];
|
||||
}
|
||||
}
|
||||
|
||||
DifferentiableOperator::DifferentiableOperator(
|
||||
const std::vector<FieldDescriptor> &solutions,
|
||||
const std::vector<FieldDescriptor> ¶meters,
|
||||
const ParMesh &mesh) :
|
||||
mesh(mesh),
|
||||
solutions(solutions),
|
||||
parameters(parameters)
|
||||
{
|
||||
fields.resize(solutions.size() + parameters.size());
|
||||
fields_e.resize(fields.size());
|
||||
solutions_l.resize(solutions.size());
|
||||
parameters_l.resize(parameters.size());
|
||||
|
||||
for (size_t i = 0; i < solutions.size(); i++)
|
||||
{
|
||||
fields[i] = solutions[i];
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < parameters.size(); i++)
|
||||
{
|
||||
fields[i + solutions.size()] = parameters[i];
|
||||
}
|
||||
}
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
@@ -1,819 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include "../fespace.hpp"
|
||||
|
||||
#include "util.hpp"
|
||||
#include "interpolate.hpp"
|
||||
#include "integrate.hpp"
|
||||
#include "qfunction_apply.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
/// @brief Type alias for a function that computes the action of an operator
|
||||
using action_t =
|
||||
std::function<void(std::vector<Vector> &, const std::vector<Vector> &, Vector &)>;
|
||||
|
||||
/// @brief Type alias for a function that computes the action of a derivative
|
||||
using derivative_action_t =
|
||||
std::function<void(std::vector<Vector> &, const Vector &, Vector &)>;
|
||||
|
||||
/// @brief Type alias for a function that assembles the sparse matrix of a
|
||||
/// derivative operator
|
||||
using assemble_derivative_hypreparmatrix_callback_t =
|
||||
std::function<void(std::vector<Vector> &, HypreParMatrix &)>;
|
||||
|
||||
/// @brief Type alias for a function that applies the appropriate restriction to
|
||||
/// the solution and parameters
|
||||
using restriction_callback_t =
|
||||
std::function<void(std::vector<Vector> &,
|
||||
const std::vector<Vector> &,
|
||||
std::vector<Vector> &)>;
|
||||
|
||||
/// Class representing the derivative (Jacobian) operator of a
|
||||
/// DifferentiableOperator.
|
||||
///
|
||||
/// This class implements a derivative operator that computes directional
|
||||
/// derivatives for a given set of solution and parameter fields. It supports
|
||||
/// both forward and transpose operations, as well as assembly into sparse
|
||||
/// matrices.
|
||||
///
|
||||
/// @note The derivative operator uses only forward mode differentiation in Mult
|
||||
/// and MultTranspose. It does not support reverse mode differentiation. The
|
||||
/// MultTranspose operation is achieved by using the transpose of the derivative
|
||||
/// actions on each quadrature point.
|
||||
///
|
||||
/// @see DifferentiableOperator
|
||||
class DerivativeOperator : public Operator
|
||||
{
|
||||
public:
|
||||
/// Constructor for the DerivativeOperator class.
|
||||
///
|
||||
/// This is usually not called directly from a user. A DifferentiableOperator
|
||||
/// calls this constructor when using
|
||||
/// DifferentiableOperator::GetDerivative().
|
||||
DerivativeOperator(
|
||||
const int &height,
|
||||
const int &width,
|
||||
const std::vector<derivative_action_t> &derivative_actions,
|
||||
const FieldDescriptor &direction,
|
||||
const int &daction_l_size,
|
||||
const std::vector<derivative_action_t> &derivative_actions_transpose,
|
||||
const FieldDescriptor &transpose_direction,
|
||||
const int &daction_transpose_l_size,
|
||||
const std::vector<Vector *> &solutions_l,
|
||||
const std::vector<Vector *> ¶meters_l,
|
||||
const restriction_callback_t &restriction_callback,
|
||||
const std::function<void(Vector &, Vector &)> &prolongation_transpose,
|
||||
const std::vector<assemble_derivative_hypreparmatrix_callback_t>
|
||||
&assemble_derivative_hypreparmatrix_callbacks) :
|
||||
Operator(height, width),
|
||||
derivative_actions(derivative_actions),
|
||||
direction(direction),
|
||||
daction_l(daction_l_size),
|
||||
daction_l_size(daction_l_size),
|
||||
derivative_actions_transpose(derivative_actions_transpose),
|
||||
transpose_direction(transpose_direction),
|
||||
prolongation_transpose(prolongation_transpose),
|
||||
assemble_derivative_hypreparmatrix_callbacks(
|
||||
assemble_derivative_hypreparmatrix_callbacks)
|
||||
{
|
||||
std::vector<Vector> s_l(solutions_l.size());
|
||||
for (size_t i = 0; i < s_l.size(); i++)
|
||||
{
|
||||
s_l[i] = *solutions_l[i];
|
||||
}
|
||||
|
||||
std::vector<Vector> p_l(parameters_l.size());
|
||||
for (size_t i = 0; i < p_l.size(); i++)
|
||||
{
|
||||
p_l[i] = *parameters_l[i];
|
||||
}
|
||||
|
||||
fields_e.resize(solutions_l.size() + parameters_l.size());
|
||||
restriction_callback(s_l, p_l, fields_e);
|
||||
}
|
||||
|
||||
/// @brief Compute the action of the derivative operator on a given vector.
|
||||
///
|
||||
/// @param direction_t The direction vector in which to compute the
|
||||
/// derivative. This has to be a T-dof vector.
|
||||
/// @param result_t Result vector of the action of the derivative on
|
||||
/// direction_t on T-dofs.
|
||||
void Mult(const Vector &direction_t, Vector &result_t) const override
|
||||
{
|
||||
daction_l.SetSize(daction_l_size);
|
||||
daction_l = 0.0;
|
||||
|
||||
prolongation(direction, direction_t, direction_l);
|
||||
for (const auto &f : derivative_actions)
|
||||
{
|
||||
f(fields_e, direction_l, daction_l);
|
||||
}
|
||||
prolongation_transpose(daction_l, result_t);
|
||||
};
|
||||
|
||||
/// @brief Compute the transpose of the derivative operator on a given
|
||||
/// vector.
|
||||
///
|
||||
/// This function computes the transpose of the derivative operator on a
|
||||
/// given vector by transposing the quadrature point local forward derivative
|
||||
/// action. It does not use reverse mode automatic differentiation.
|
||||
///
|
||||
/// @param direction_t The direction vector in which to compute the
|
||||
/// derivative. This has to be a T-dof vector.
|
||||
/// @param result_t Result vector of the transpose action of the derivative on
|
||||
/// direction_t on T-dofs.
|
||||
void MultTranspose(const Vector &direction_t, Vector &result_t) const override
|
||||
{
|
||||
MFEM_ASSERT(!derivative_actions_transpose.empty(),
|
||||
"derivative can't be used to be multiplied in transpose mode");
|
||||
|
||||
daction_l.SetSize(width);
|
||||
daction_l = 0.0;
|
||||
|
||||
prolongation(transpose_direction, direction_t, direction_l);
|
||||
for (const auto &f : derivative_actions_transpose)
|
||||
{
|
||||
f(fields_e, direction_l, daction_l);
|
||||
}
|
||||
prolongation_transpose(daction_l, result_t);
|
||||
};
|
||||
|
||||
/// @brief Assemble the derivative operator into a HypreParMatrix.
|
||||
///
|
||||
/// @param A The HypreParMatrix to assemble the derivative operator into. Can
|
||||
/// be an uninitialized object.
|
||||
void Assemble(HypreParMatrix &A)
|
||||
{
|
||||
MFEM_ASSERT(!assemble_derivative_hypreparmatrix_callbacks.empty(),
|
||||
"derivative can't be assembled into a matrix");
|
||||
|
||||
for (const auto &f : assemble_derivative_hypreparmatrix_callbacks)
|
||||
{
|
||||
f(fields_e, A);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
/// Derivative action callbacks. Depending on the requested derivatives in
|
||||
/// DifferentiableOperator the callbacks represent certain combinations of
|
||||
/// actions of derivatives of the forward operator.
|
||||
std::vector<derivative_action_t> derivative_actions;
|
||||
|
||||
FieldDescriptor direction;
|
||||
|
||||
mutable Vector daction_l;
|
||||
|
||||
const int daction_l_size;
|
||||
|
||||
/// Transpose Derivative action callbacks. Depending on the requested
|
||||
/// derivatives in DifferentiableOperator the callbacks represent certain
|
||||
/// combinations of actions of derivatives of the forward operator.
|
||||
std::vector<derivative_action_t> derivative_actions_transpose;
|
||||
|
||||
FieldDescriptor transpose_direction;
|
||||
|
||||
mutable std::vector<Vector> fields_e;
|
||||
|
||||
mutable Vector direction_l;
|
||||
|
||||
std::function<void(Vector &, Vector &)> prolongation_transpose;
|
||||
|
||||
/// Callbacks that assemble derivatives into a HypreParMatrix.
|
||||
std::vector<assemble_derivative_hypreparmatrix_callback_t>
|
||||
assemble_derivative_hypreparmatrix_callbacks;
|
||||
};
|
||||
|
||||
/// Class representing a differentiable operator which acts on solution and
|
||||
/// parameter fields to compute residuals.
|
||||
///
|
||||
/// This class provides functionality to define differentiable operators by
|
||||
/// composing functions that compute values at quadrature points. It supports
|
||||
/// automatic differentiation to compute derivatives with respect to solutions
|
||||
/// (Jacobians) and parameter fields (general derivative operators).
|
||||
///
|
||||
/// The operator is constructed with solution fields that it will act on and
|
||||
/// parameter fields that define coefficients. Quadrature functions are added by
|
||||
/// e.g. using AddDomainIntegrator() which specify how the operator evaluates f
|
||||
/// those functionas and parameters at quadrature points.
|
||||
///
|
||||
/// Derivatives can be computed by obtaining a DerivativeOperator using
|
||||
/// GetDerivative().
|
||||
///
|
||||
/// @see DerivativeOperator
|
||||
class DifferentiableOperator : public Operator
|
||||
{
|
||||
public:
|
||||
/// Constructor for the DifferentiableOperator class.
|
||||
///
|
||||
/// @param solutions The solution fields that the operator will act on.
|
||||
/// @param parameters The parameter fields that define coefficients.
|
||||
/// @param mesh The mesh on which the operator is defined.
|
||||
DifferentiableOperator(
|
||||
const std::vector<FieldDescriptor> &solutions,
|
||||
const std::vector<FieldDescriptor> ¶meters,
|
||||
const ParMesh &mesh);
|
||||
|
||||
/// @brief Compute the action of the operator on a given vector.
|
||||
///
|
||||
/// @param solutions_t The solution vector in which to compute the action.
|
||||
/// This has to be a T-dof vector.
|
||||
/// @param result_t Result vector of the action of the operator on
|
||||
/// solutions_t. The result is a T-dof vector.
|
||||
void Mult(const Vector &solutions_t, Vector &result_t) const override
|
||||
{
|
||||
MFEM_ASSERT(!action_callbacks.empty(), "no integrators have been set");
|
||||
prolongation(solutions, solutions_t, solutions_l);
|
||||
for (auto &action : action_callbacks)
|
||||
{
|
||||
action(solutions_l, parameters_l, residual_l);
|
||||
}
|
||||
prolongation_transpose(residual_l, result_t);
|
||||
}
|
||||
|
||||
/// @brief Add a domain integrator to the operator.
|
||||
///
|
||||
/// @param qfunc The quadrature function to be added.
|
||||
/// @param inputs Tuple of FieldOperators for the inputs of the quadrature
|
||||
/// function.
|
||||
/// @param outputs Tuple of FieldOperators for the outputs of the quadrature
|
||||
/// function.
|
||||
/// @param integration_rule IntegrationRule to use with this integrator.
|
||||
/// @param domain_attributes Domain attributes marker array indicating over
|
||||
/// which attributes this integrator will integrate over.
|
||||
/// @param derivative_ids Derivatives to be made available for this
|
||||
/// integrator.
|
||||
template <
|
||||
typename qfunc_t,
|
||||
typename input_t,
|
||||
typename output_t,
|
||||
typename derivative_ids_t = decltype(std::make_index_sequence<0> {})>
|
||||
void AddDomainIntegrator(
|
||||
qfunc_t &qfunc,
|
||||
input_t inputs,
|
||||
output_t outputs,
|
||||
const IntegrationRule &integration_rule,
|
||||
const Array<int> &domain_attributes,
|
||||
derivative_ids_t derivative_ids = std::make_index_sequence<0> {});
|
||||
|
||||
/// @brief Set the parameters for the operator.
|
||||
///
|
||||
/// This has to be called before using Mult() or MultTranspose().
|
||||
///
|
||||
/// @param p The parameters to be set. This should be a vector of pointers to
|
||||
/// the parameter vectors. The vectors have to be L-vectors (e.g.
|
||||
/// GridFunctions).
|
||||
void SetParameters(std::vector<Vector *> p) const;
|
||||
|
||||
/// @brief Disable the use of tensor product structure.
|
||||
///
|
||||
/// This function disables the use of tensor product structure for the
|
||||
/// operator. Usually, DifferentiableOperator creates callbacks based on
|
||||
/// heuristics that achieve good performance for each element type. Some
|
||||
/// functionality is not implemented for these performant algorithms but only
|
||||
/// for generic assembly. Therefore the user can decide to use fallback
|
||||
/// methods.
|
||||
void DisableTensorProductStructure(bool disable = true)
|
||||
{
|
||||
use_tensor_product_structure = !disable;
|
||||
}
|
||||
|
||||
/// @brief Get the derivative operator for a given derivative ID.
|
||||
///
|
||||
/// This function returns a shared pointer to a DerivativeOperator that
|
||||
/// computes the derivative of the operator with respect to the given
|
||||
/// derivative ID. The derivative ID is used to identify the specific
|
||||
/// derivative action to be performed.
|
||||
///
|
||||
/// @param derivative_id The ID of the derivative to be computed.
|
||||
/// @param sol_l The solution vectors to be used for the derivative
|
||||
/// computation. This should be a vector of pointers to the solution
|
||||
/// vectors. The vectors have to be L-vectors (e.g. GridFunctions).
|
||||
/// @param par_l The parameter vectors to be used for the derivative
|
||||
/// computation. This should be a vector of pointers to the parameter
|
||||
/// vectors. The vectors have to be L-vectors (e.g. GridFunctions).
|
||||
/// @return A shared pointer to the DerivativeOperator.
|
||||
std::shared_ptr<DerivativeOperator> GetDerivative(
|
||||
size_t derivative_id, std::vector<Vector *> sol_l, std::vector<Vector *> par_l)
|
||||
{
|
||||
MFEM_ASSERT(derivative_action_callbacks.find(derivative_id) !=
|
||||
derivative_action_callbacks.end(),
|
||||
"no derivative action has been found for ID " << derivative_id);
|
||||
|
||||
MFEM_ASSERT(sol_l.size() == solutions.size(),
|
||||
"wrong number of solutions");
|
||||
|
||||
MFEM_ASSERT(par_l.size() == parameters.size(),
|
||||
"wrong number of parameters");
|
||||
|
||||
const size_t derivative_idx = FindIdx(derivative_id, fields);
|
||||
|
||||
return std::make_shared<DerivativeOperator>(
|
||||
height,
|
||||
GetTrueVSize(fields[derivative_idx]),
|
||||
derivative_action_callbacks[derivative_id],
|
||||
fields[derivative_idx],
|
||||
residual_l.Size(),
|
||||
daction_transpose_callbacks[derivative_id],
|
||||
fields[test_space_field_idx],
|
||||
GetVSize(fields[test_space_field_idx]),
|
||||
sol_l,
|
||||
par_l,
|
||||
restriction_callback,
|
||||
prolongation_transpose,
|
||||
assemble_derivative_hypreparmatrix_callbacks[derivative_id]);
|
||||
}
|
||||
|
||||
private:
|
||||
const ParMesh &mesh;
|
||||
|
||||
std::vector<action_t> action_callbacks;
|
||||
std::map<size_t,
|
||||
std::vector<derivative_action_t>> derivative_action_callbacks;
|
||||
std::map<size_t,
|
||||
std::vector<derivative_action_t>> daction_transpose_callbacks;
|
||||
std::map<size_t,
|
||||
std::vector<assemble_derivative_hypreparmatrix_callback_t>>
|
||||
assemble_derivative_hypreparmatrix_callbacks;
|
||||
|
||||
|
||||
std::vector<FieldDescriptor> solutions;
|
||||
std::vector<FieldDescriptor> parameters;
|
||||
// solutions and parameters
|
||||
std::vector<FieldDescriptor> fields;
|
||||
|
||||
mutable std::vector<Vector> solutions_l;
|
||||
mutable std::vector<Vector> parameters_l;
|
||||
mutable Vector residual_l;
|
||||
|
||||
mutable std::vector<Vector> fields_e;
|
||||
mutable Vector residual_e;
|
||||
|
||||
std::function<void(Vector &, Vector &)> prolongation_transpose;
|
||||
|
||||
// TODO: This can probably be removed, as it is only used in
|
||||
// the callbacks so they can copy it during the capture.
|
||||
std::function<void(Vector &, Vector &)> output_restriction_transpose;
|
||||
|
||||
restriction_callback_t restriction_callback;
|
||||
|
||||
std::map<size_t, std::function<void(const Vector &, Vector &)>>
|
||||
daction_prolongation_transpose;
|
||||
|
||||
std::map<size_t, size_t> assembled_vector_sizes;
|
||||
|
||||
bool use_tensor_product_structure = true;
|
||||
|
||||
size_t test_space_field_idx = SIZE_MAX;
|
||||
};
|
||||
|
||||
template <
|
||||
typename qfunc_t,
|
||||
typename input_t,
|
||||
typename output_t,
|
||||
typename derivative_ids_t>
|
||||
void DifferentiableOperator::AddDomainIntegrator(
|
||||
qfunc_t &qfunc,
|
||||
input_t inputs,
|
||||
output_t outputs,
|
||||
const IntegrationRule &integration_rule,
|
||||
const Array<int> &domain_attributes,
|
||||
derivative_ids_t derivative_ids)
|
||||
{
|
||||
using entity_t = Entity::Element;
|
||||
|
||||
static constexpr size_t num_inputs =
|
||||
tuple_size<decltype(inputs)>::value;
|
||||
|
||||
static constexpr size_t num_outputs =
|
||||
tuple_size<decltype(outputs)>::value;
|
||||
|
||||
using qf_signature =
|
||||
typename create_function_signature<decltype(&qfunc_t::operator())>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
using qf_output_t = typename qf_signature::return_t;
|
||||
|
||||
// Consistency checks
|
||||
if constexpr (num_outputs > 1)
|
||||
{
|
||||
static_assert(dfem::always_false<qfunc_t>,
|
||||
"more than one output per quadrature functions is not supported right now");
|
||||
}
|
||||
|
||||
if constexpr (std::is_same_v<qf_output_t, void>)
|
||||
{
|
||||
static_assert(dfem::always_false<qfunc_t>,
|
||||
"quadrature function has no return value");
|
||||
}
|
||||
|
||||
constexpr size_t num_qfinputs = tuple_size<qf_param_ts>::value;
|
||||
static_assert(num_qfinputs == num_inputs,
|
||||
"quadrature function inputs and descriptor inputs have to match");
|
||||
|
||||
constexpr size_t num_qf_outputs = tuple_size<qf_output_t>::value;
|
||||
static_assert(num_qf_outputs == num_outputs,
|
||||
"quadrature function outputs and descriptor outputs have to match");
|
||||
|
||||
constexpr auto inout_tuple =
|
||||
merge_mfem_tuples_as_empty_std_tuple(inputs, outputs);
|
||||
constexpr auto filtered_inout_tuple = filter_fields(inout_tuple);
|
||||
static constexpr size_t num_fields =
|
||||
count_unique_field_ids(filtered_inout_tuple);
|
||||
|
||||
MFEM_ASSERT(num_fields == solutions.size() + parameters.size(),
|
||||
"Total number of fields doesn't match sum of solutions and parameters."
|
||||
" This indicates that some fields are not used in the integrator,"
|
||||
" which currently is not supported.");
|
||||
|
||||
auto dependency_map = make_dependency_map(inputs);
|
||||
|
||||
// pretty_print(dependency_map);
|
||||
|
||||
auto input_to_field =
|
||||
create_descriptors_to_fields_map<entity_t>(fields, inputs);
|
||||
auto output_to_field =
|
||||
create_descriptors_to_fields_map<entity_t>(fields, outputs);
|
||||
|
||||
// TODO: factor out
|
||||
std::vector<int> inputs_vdim(num_inputs);
|
||||
for_constexpr<num_inputs>([&](auto i)
|
||||
{
|
||||
inputs_vdim[i] = get<i>(inputs).vdim;
|
||||
});
|
||||
|
||||
|
||||
Array<int> elem_attributes;
|
||||
elem_attributes.SetSize(mesh.GetNE());
|
||||
for (int i = 0; i < mesh.GetNE(); ++i)
|
||||
{
|
||||
elem_attributes[i] = mesh.GetAttribute(i);
|
||||
}
|
||||
|
||||
const auto output_fop = get<0>(outputs);
|
||||
test_space_field_idx = FindIdx(output_fop.GetFieldId(), fields);
|
||||
|
||||
bool use_sum_factorization = false;
|
||||
auto entity_element_type =
|
||||
Element::TypeFromGeometry(mesh.GetTypicalElementGeometry());
|
||||
if ((entity_element_type == Element::QUADRILATERAL ||
|
||||
entity_element_type == Element::HEXAHEDRON) &&
|
||||
use_tensor_product_structure == true)
|
||||
{
|
||||
use_sum_factorization = true;
|
||||
}
|
||||
|
||||
ElementDofOrdering element_dof_ordering = ElementDofOrdering::NATIVE;
|
||||
DofToQuad::Mode doftoquad_mode = DofToQuad::Mode::FULL;
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
element_dof_ordering = ElementDofOrdering::LEXICOGRAPHIC;
|
||||
doftoquad_mode = DofToQuad::Mode::TENSOR;
|
||||
}
|
||||
|
||||
auto [output_rt,
|
||||
output_e_sz] = get_restriction_transpose<entity_t>
|
||||
(fields[test_space_field_idx],
|
||||
element_dof_ordering, output_fop);
|
||||
auto &output_e_size = output_e_sz;
|
||||
|
||||
output_restriction_transpose = output_rt;
|
||||
residual_e.SetSize(output_e_size);
|
||||
|
||||
// The explicit captures are necessary to avoid dependency on
|
||||
// the specific instance of this class (this pointer).
|
||||
restriction_callback =
|
||||
[=, solutions = this->solutions, parameters = this->parameters]
|
||||
(std::vector<Vector> &sol,
|
||||
const std::vector<Vector> &par,
|
||||
std::vector<Vector> &f)
|
||||
{
|
||||
restriction<entity_t>(solutions, sol, f,
|
||||
element_dof_ordering);
|
||||
restriction<entity_t>(parameters, par, f,
|
||||
element_dof_ordering,
|
||||
solutions.size());
|
||||
};
|
||||
|
||||
prolongation_transpose = get_prolongation_transpose(
|
||||
fields[test_space_field_idx], output_fop, mesh.GetComm());
|
||||
|
||||
const int dimension = mesh.Dimension();
|
||||
[[maybe_unused]] const int num_elements = GetNumEntities<Entity::Element>(mesh);
|
||||
const int num_entities = GetNumEntities<entity_t>(mesh);
|
||||
const int num_qp = integration_rule.GetNPoints();
|
||||
|
||||
if constexpr (is_sum_fop<std::remove_cv_t<decltype(output_fop)>>::value)
|
||||
{
|
||||
residual_l.SetSize(1);
|
||||
height = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
const int residual_lsize = GetVSize(fields[test_space_field_idx]);
|
||||
residual_l.SetSize(residual_lsize);
|
||||
height = GetTrueVSize(fields[test_space_field_idx]);
|
||||
}
|
||||
|
||||
// TODO: Is this a hack?
|
||||
width = GetTrueVSize(fields[0]);
|
||||
|
||||
std::vector<const DofToQuad*> dtq;
|
||||
for (const auto &field : fields)
|
||||
{
|
||||
dtq.emplace_back(GetDofToQuad<entity_t>(
|
||||
field,
|
||||
integration_rule,
|
||||
doftoquad_mode));
|
||||
}
|
||||
const int q1d = (int)floor(std::pow(num_qp, 1.0/dimension) + 0.5);
|
||||
|
||||
const int residual_size_on_qp =
|
||||
GetSizeOnQP<entity_t>(output_fop,
|
||||
fields[test_space_field_idx]);
|
||||
|
||||
auto input_dtq_maps = create_dtq_maps<entity_t>(inputs, dtq, input_to_field);
|
||||
auto output_dtq_maps = create_dtq_maps<entity_t>(outputs, dtq, output_to_field);
|
||||
|
||||
const int test_vdim = output_fop.vdim;
|
||||
const int test_op_dim = output_fop.size_on_qp / output_fop.vdim;
|
||||
|
||||
int num_test_dof = 0;
|
||||
if (num_entities)
|
||||
{
|
||||
if constexpr (is_sum_fop<std::decay_t<decltype(output_fop)>>::value)
|
||||
{
|
||||
num_test_dof = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
num_test_dof = output_e_size / output_fop.vdim / num_entities;
|
||||
}
|
||||
}
|
||||
|
||||
auto ir_weights = Reshape(integration_rule.GetWeights().Read(), num_qp);
|
||||
|
||||
auto input_size_on_qp =
|
||||
get_input_size_on_qp(inputs, std::make_index_sequence<num_inputs> {});
|
||||
|
||||
auto action_shmem_info =
|
||||
get_shmem_info<entity_t, num_fields, num_inputs, num_outputs>
|
||||
(input_dtq_maps, output_dtq_maps, fields, num_entities, inputs, num_qp,
|
||||
input_size_on_qp, residual_size_on_qp, element_dof_ordering);
|
||||
|
||||
Vector shmem_cache(action_shmem_info.total_size);
|
||||
|
||||
// print_shared_memory_info(action_shmem_info);
|
||||
|
||||
ThreadBlocks thread_blocks;
|
||||
if (dimension == 3)
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
thread_blocks.x = q1d;
|
||||
thread_blocks.y = q1d;
|
||||
thread_blocks.z = q1d;
|
||||
}
|
||||
}
|
||||
else if (dimension == 2)
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
thread_blocks.x = q1d;
|
||||
thread_blocks.y = q1d;
|
||||
thread_blocks.z = 1;
|
||||
}
|
||||
}
|
||||
|
||||
action_callbacks.push_back(
|
||||
// Explicitly capture everything we need, so we can make explicit choice
|
||||
// how to capture every variable, by copy or by ref.
|
||||
[
|
||||
// capture by copy:
|
||||
dimension, // int
|
||||
num_entities, // int
|
||||
num_test_dof, // int
|
||||
num_qp, // int
|
||||
q1d, // int
|
||||
residual_size_on_qp, // int
|
||||
test_vdim, // int (= output_fop.vdim)
|
||||
test_op_dim, // int (derived from output_fop)
|
||||
inputs, // mfem::future::tuple
|
||||
domain_attributes, // Array<int>
|
||||
ir_weights, // DeviceTensor
|
||||
use_sum_factorization, // bool
|
||||
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
|
||||
output_dtq_maps, // std::array<DofToQuadMap, num_fields>
|
||||
input_to_field, // std::array<int, s>
|
||||
output_fop, // class derived from FieldOperator
|
||||
qfunc, // qfunc_t
|
||||
thread_blocks, // ThreadBlocks
|
||||
shmem_cache, // Vector (local)
|
||||
action_shmem_info, // SharedMemoryInfo
|
||||
// TODO: make this Array<int> a member of the DifferentiableOperator
|
||||
// and capture it by ref.
|
||||
elem_attributes, // Array<int>
|
||||
|
||||
// capture by ref:
|
||||
&restriction_cb = this->restriction_callback,
|
||||
&fields_e = this->fields_e,
|
||||
&residual_e = this->residual_e,
|
||||
&output_restriction_transpose = this->output_restriction_transpose
|
||||
]
|
||||
(std::vector<Vector> &sol, const std::vector<Vector> &par, Vector &res)
|
||||
mutable // mutable: needed to modify 'shmem_cache'
|
||||
{
|
||||
restriction_cb(sol, par, fields_e);
|
||||
|
||||
residual_e = 0.0;
|
||||
auto ye = Reshape(residual_e.ReadWrite(), test_vdim, num_test_dof, num_entities);
|
||||
|
||||
auto wrapped_fields_e = wrap_fields(fields_e,
|
||||
action_shmem_info.field_sizes,
|
||||
num_entities);
|
||||
|
||||
const bool has_attr = domain_attributes.Size() > 0;
|
||||
const auto d_domain_attr = domain_attributes.Read();
|
||||
const auto d_elem_attr = elem_attributes.Read();
|
||||
|
||||
forall([=] MFEM_HOST_DEVICE (int e, void *shmem)
|
||||
{
|
||||
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
|
||||
|
||||
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem, input_shmem,
|
||||
residual_shmem, scratch_shmem] =
|
||||
unpack_shmem(shmem, action_shmem_info, input_dtq_maps, output_dtq_maps,
|
||||
wrapped_fields_e, num_qp, e);
|
||||
|
||||
map_fields_to_quadrature_data(
|
||||
input_shmem, fields_shmem, input_dtq_shmem, input_to_field, inputs, ir_weights,
|
||||
scratch_shmem, dimension, use_sum_factorization);
|
||||
|
||||
call_qfunction<qf_param_ts>(
|
||||
qfunc, input_shmem, residual_shmem,
|
||||
residual_size_on_qp, num_qp, q1d, dimension, use_sum_factorization);
|
||||
|
||||
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
|
||||
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
|
||||
map_quadrature_data_to_fields(
|
||||
y, fhat, output_fop, output_dtq_shmem[0],
|
||||
scratch_shmem, dimension, use_sum_factorization);
|
||||
}, num_entities, thread_blocks, action_shmem_info.total_size, shmem_cache.ReadWrite());
|
||||
output_restriction_transpose(residual_e, res);
|
||||
});
|
||||
|
||||
// Without this compile-time check, some valid instantiations of this method
|
||||
// will fail.
|
||||
if constexpr (derivative_ids_t::size() != 0)
|
||||
{
|
||||
// Create the action of the derivatives
|
||||
for_constexpr([&,
|
||||
&or_transpose = this->output_restriction_transpose,
|
||||
&dapr_transpose = this->daction_prolongation_transpose]
|
||||
(const std::size_t derivative_id)
|
||||
{
|
||||
const size_t d_field_idx = FindIdx(derivative_id, fields);
|
||||
const auto direction = fields[d_field_idx];
|
||||
|
||||
dapr_transpose[derivative_id] = get_generic_prolongation_transpose(direction);
|
||||
|
||||
const int da_size_on_qp =
|
||||
GetSizeOnQP<entity_t>(output_fop, fields[test_space_field_idx]);
|
||||
|
||||
auto shmem_info =
|
||||
get_shmem_info<entity_t, num_fields, num_inputs, num_outputs>(
|
||||
input_dtq_maps, output_dtq_maps, fields, num_entities, inputs,
|
||||
num_qp, input_size_on_qp, residual_size_on_qp,
|
||||
element_dof_ordering, d_field_idx);
|
||||
|
||||
Vector shmem_cache(shmem_info.total_size);
|
||||
|
||||
// print_shared_memory_info(shmem_info);
|
||||
|
||||
Vector direction_e;
|
||||
Vector derivative_action_e(output_e_size);
|
||||
derivative_action_e = 0.0;
|
||||
|
||||
// Lookup the derivative_id key in the dependency map
|
||||
auto it = dependency_map.find(derivative_id);
|
||||
if (it == dependency_map.end())
|
||||
{
|
||||
MFEM_ABORT("Derivative ID not found in dependency map");
|
||||
}
|
||||
const auto input_is_dependent = it->second;
|
||||
|
||||
derivative_action_callbacks[derivative_id].push_back(
|
||||
[
|
||||
// capture by copy:
|
||||
dimension, // int
|
||||
num_entities, // int
|
||||
num_test_dof, // int
|
||||
num_qp, // int
|
||||
q1d, // int
|
||||
test_vdim, // int (= output_fop.vdim)
|
||||
test_op_dim, // int (derived from output_fop)
|
||||
inputs, // mfem::future::tuple
|
||||
domain_attributes, // Array<int>
|
||||
ir_weights, // DeviceTensor
|
||||
use_sum_factorization, // bool
|
||||
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
|
||||
output_dtq_maps, // std::array<DofToQuadMap, num_fields>
|
||||
input_to_field, // std::array<int, s>
|
||||
output_fop, // class derived from FieldOperator
|
||||
qfunc, // qfunc_t
|
||||
thread_blocks, // ThreadBlocks
|
||||
shmem_cache, // Vector (local)
|
||||
shmem_info, // SharedMemoryInfo
|
||||
// TODO: make this Array<int> a member of the DifferentiableOperator
|
||||
// and capture it by ref.
|
||||
elem_attributes, // Array<int>
|
||||
|
||||
input_is_dependent, // std::array<bool, num_inputs>
|
||||
direction, // FieldDescriptor
|
||||
direction_e, // Vector
|
||||
derivative_action_e, // Vector
|
||||
element_dof_ordering, // ElementDofOrdering
|
||||
da_size_on_qp, // int
|
||||
|
||||
// capture by ref:
|
||||
&or_transpose
|
||||
](
|
||||
std::vector<Vector> &f_e, const Vector &dir_l,
|
||||
Vector &der_action_l) mutable
|
||||
{
|
||||
restriction<entity_t>(direction, dir_l, direction_e,
|
||||
element_dof_ordering);
|
||||
auto ye = Reshape(derivative_action_e.ReadWrite(), num_test_dof,
|
||||
test_vdim, num_entities);
|
||||
auto wrapped_fields_e = wrap_fields(f_e, shmem_info.field_sizes,
|
||||
num_entities);
|
||||
auto wrapped_direction_e = Reshape(direction_e.ReadWrite(),
|
||||
shmem_info.direction_size,
|
||||
num_entities);
|
||||
|
||||
const auto d_elem_attr = elem_attributes.Read();
|
||||
const bool has_attr = domain_attributes.Size() > 0;
|
||||
const auto d_domain_attr = domain_attributes.Read();
|
||||
|
||||
derivative_action_e = 0.0;
|
||||
forall([=] MFEM_HOST_DEVICE (int e, real_t *shmem)
|
||||
{
|
||||
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
|
||||
|
||||
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem,
|
||||
direction_shmem, input_shmem,
|
||||
shadow_shmem_, residual_shmem,
|
||||
scratch_shmem] =
|
||||
unpack_shmem(shmem, shmem_info, input_dtq_maps, output_dtq_maps,
|
||||
wrapped_fields_e, wrapped_direction_e, num_qp, e);
|
||||
auto &shadow_shmem = shadow_shmem_;
|
||||
|
||||
map_fields_to_quadrature_data(
|
||||
input_shmem, fields_shmem, input_dtq_shmem, input_to_field,
|
||||
inputs, ir_weights, scratch_shmem, dimension,
|
||||
use_sum_factorization);
|
||||
|
||||
// TODO: Probably redundant
|
||||
set_zero(shadow_shmem);
|
||||
|
||||
map_direction_to_quadrature_data_conditional(
|
||||
shadow_shmem, direction_shmem, input_dtq_shmem, inputs,
|
||||
ir_weights, scratch_shmem, input_is_dependent, dimension,
|
||||
use_sum_factorization);
|
||||
|
||||
call_qfunction_derivative_action<qf_param_ts>(
|
||||
qfunc, input_shmem, shadow_shmem, residual_shmem,
|
||||
da_size_on_qp, num_qp, q1d, dimension, use_sum_factorization);
|
||||
|
||||
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim,
|
||||
test_op_dim, num_qp);
|
||||
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
|
||||
map_quadrature_data_to_fields(
|
||||
y, fhat, output_fop, output_dtq_shmem[0],
|
||||
scratch_shmem, dimension, use_sum_factorization);
|
||||
}, num_entities, thread_blocks, shmem_info.total_size,
|
||||
shmem_cache.ReadWrite());
|
||||
or_transpose(derivative_action_e, der_action_l);
|
||||
});
|
||||
}, derivative_ids);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::future
|
||||
#endif
|
||||
@@ -1,144 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
/// @brief Base class for FieldOperators.
|
||||
///
|
||||
/// This class serves as a base for different FieldOperator types which can be
|
||||
/// applied to fields that are used with inputs to a quadrature point function.
|
||||
/// See DifferentialOperator.
|
||||
template <int FIELD_ID = -1>
|
||||
class FieldOperator
|
||||
{
|
||||
public:
|
||||
/// @brief Constructor for the FieldOperator.
|
||||
///
|
||||
/// This constructor initializes the FieldOperator with it's size on
|
||||
/// quadrature points. The size on quadrature points has to be determined by
|
||||
/// the FieldOperator type, the dimension and the vector dimension (number
|
||||
/// of components). See the following examples
|
||||
///
|
||||
/// Scalar FiniteElementSpace with Value FieldOperator:
|
||||
/// size = vdim x dim x 1 = 1 x dim x 1 = dim
|
||||
///
|
||||
/// Vector FiniteElementSpace with Gradient FieldOperator:
|
||||
/// size = vdim x dim x dim = vdim x dim x dim = vdim * dim^2
|
||||
///
|
||||
/// ParameterSpace with Identity FieldOperator:
|
||||
/// size = vdim = vdim
|
||||
constexpr FieldOperator(int size_on_qp = 0) :
|
||||
size_on_qp(size_on_qp) {};
|
||||
|
||||
/// @brief Get the field id this FieldOperator is attached to.
|
||||
static constexpr int GetFieldId() { return FIELD_ID; }
|
||||
|
||||
/// @brief Get the size on quadrature point for this FieldOperator.
|
||||
int size_on_qp = -1;
|
||||
|
||||
/// @brief Get the dimension of the FieldOperator.
|
||||
int dim = -1;
|
||||
|
||||
/// @brief Get the vector dimension (number of components)
|
||||
/// of the FieldOperator.
|
||||
int vdim = -1;
|
||||
};
|
||||
|
||||
/// @brief Identity FieldOperator.
|
||||
///
|
||||
/// This FieldOperator does nothing to the field. The field (usually a
|
||||
/// ParametricFunction) transfers the values to the quadrature point data and
|
||||
/// Identity can be viewed as an identity operation.
|
||||
template <int FIELD_ID = -1>
|
||||
class Identity : public FieldOperator<FIELD_ID>
|
||||
{
|
||||
public:
|
||||
constexpr Identity() : FieldOperator<FIELD_ID>() {}
|
||||
};
|
||||
|
||||
template< typename T >
|
||||
struct is_identity_fop : std::false_type {};
|
||||
|
||||
template <int FIELD_ID>
|
||||
struct is_identity_fop<Identity<FIELD_ID>> : std::true_type {};
|
||||
|
||||
/// @brief Weight FieldOperator.
|
||||
///
|
||||
/// This FieldOperator is used to signal that this field contains the quadrature
|
||||
/// point weights.
|
||||
class Weight : public FieldOperator<-1>
|
||||
{
|
||||
public:
|
||||
constexpr Weight() : FieldOperator<-1>() {};
|
||||
};
|
||||
|
||||
template< typename T >
|
||||
struct is_weight_fop : std::false_type {};
|
||||
|
||||
template <>
|
||||
struct is_weight_fop<Weight> : std::true_type {};
|
||||
|
||||
/// @brief Value FieldOperator.
|
||||
///
|
||||
/// This FieldOperator is used to signal that the field contains the
|
||||
/// interpolated values of the degrees of freedom at the quadrature points.
|
||||
template <int FIELD_ID = -1>
|
||||
class Value : public FieldOperator<FIELD_ID>
|
||||
{
|
||||
public:
|
||||
constexpr Value() : FieldOperator<FIELD_ID>() {};
|
||||
};
|
||||
|
||||
template< typename T >
|
||||
struct is_value_fop : std::false_type {};
|
||||
|
||||
template <int FIELD_ID>
|
||||
struct is_value_fop<Value<FIELD_ID>> : std::true_type {};
|
||||
|
||||
/// @brief Gradient FieldOperator.
|
||||
///
|
||||
/// This FieldOperator is used to signal that the field contains the
|
||||
/// interpolated gradients of the degrees of freedom at the quadrature points.
|
||||
template <int FIELD_ID = -1>
|
||||
class Gradient : public FieldOperator<FIELD_ID>
|
||||
{
|
||||
public:
|
||||
constexpr Gradient() : FieldOperator<FIELD_ID>() {};
|
||||
};
|
||||
|
||||
template< typename T >
|
||||
struct is_gradient_fop : std::false_type {};
|
||||
|
||||
template <int FIELD_ID>
|
||||
struct is_gradient_fop<Gradient<FIELD_ID>> : std::true_type {};
|
||||
|
||||
/// @brief Sum FieldOperator.
|
||||
///
|
||||
/// This FieldOperator is commonly used to signal that an output of a quadrature
|
||||
/// function should be summed.
|
||||
template <int FIELD_ID = -1>
|
||||
class Sum : public FieldOperator<FIELD_ID>
|
||||
{
|
||||
public:
|
||||
constexpr Sum() : FieldOperator<FIELD_ID>() {};
|
||||
};
|
||||
|
||||
template< typename T >
|
||||
struct is_sum_fop : std::false_type {};
|
||||
|
||||
template <int FIELD_ID>
|
||||
struct is_sum_fop<Sum<FIELD_ID>> : std::true_type {};
|
||||
|
||||
} // namespace mfem::future
|
||||
@@ -1,472 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "util.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
template <typename output_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_quadrature_data_to_fields_impl(
|
||||
DeviceTensor<2, real_t> &y,
|
||||
const DeviceTensor<3, real_t> &f,
|
||||
const output_t &output,
|
||||
const DofToQuadMap &dtq)
|
||||
{
|
||||
[[maybe_unused]] auto B = dtq.B;
|
||||
[[maybe_unused]] auto G = dtq.G;
|
||||
|
||||
// assuming the quadrature point residual has to "play nice with
|
||||
// the test function"
|
||||
if constexpr (is_value_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [num_qp, cdim, num_dof] = B.GetShape();
|
||||
const int vdim = output.vdim > 0 ? output.vdim : cdim ;
|
||||
for (int dof = 0; dof < num_dof; dof++)
|
||||
{
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int qp = 0; qp < num_qp; qp++)
|
||||
{
|
||||
acc += B(qp, 0, dof) * f(vd, 0, qp);
|
||||
}
|
||||
y(dof, vd) += acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if constexpr (
|
||||
is_gradient_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [num_qp, dim, num_dof] = G.GetShape();
|
||||
const int vdim = output.vdim;
|
||||
for (int dof = 0; dof < num_dof; dof++)
|
||||
{
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
for (int qp = 0; qp < num_qp; qp++)
|
||||
{
|
||||
acc += G(qp, d, dof) * f(vd, d, qp);
|
||||
}
|
||||
}
|
||||
y(dof, vd) += acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if constexpr (is_sum_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
// This is the "integral over all quadrature points type" applying
|
||||
// B = 1 s.t. B^T * C \in R^1.
|
||||
const auto [num_qp, unused, unused1] = B.GetShape();
|
||||
auto cc = Reshape(&f(0, 0, 0), num_qp);
|
||||
for (int i = 0; i < num_qp; i++)
|
||||
{
|
||||
y(0, 0) += cc(i);
|
||||
}
|
||||
}
|
||||
else if constexpr (is_identity_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [num_qp, unused, num_dof] = B.GetShape();
|
||||
const auto vdim = output.vdim;
|
||||
auto cc = Reshape(&f(0, 0, 0), num_qp * vdim);
|
||||
auto yy = Reshape(&y(0, 0), num_qp * vdim);
|
||||
for (int i = 0; i < num_qp * vdim; i++)
|
||||
{
|
||||
yy(i) = cc(i);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("quadrature data mapping to field is not implemented for"
|
||||
" this field descriptor");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename output_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_quadrature_data_to_fields_tensor_impl_2d(
|
||||
DeviceTensor<2, real_t> &y,
|
||||
const DeviceTensor<3, real_t> &f,
|
||||
const output_t &output,
|
||||
const DofToQuadMap &dtq,
|
||||
std::array<DeviceTensor<1>, 6> &scratch_mem)
|
||||
{
|
||||
[[maybe_unused]] auto B = dtq.B;
|
||||
[[maybe_unused]] auto G = dtq.G;
|
||||
|
||||
if constexpr (is_value_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = output.vdim;
|
||||
const int test_dim = output.size_on_qp / vdim;
|
||||
|
||||
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d, q1d);
|
||||
auto yd = Reshape(&y(0, 0), d1d, d1d, vdim);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), q1d, d1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
acc += fqp(vd, 0, qx, qy) * B(qx, 0, dx);
|
||||
}
|
||||
s0(qy, dx) = acc;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int qy = 0; qy < q1d; qy++)
|
||||
{
|
||||
acc += s0(qy, dx) * B(qy, 0, dy);
|
||||
}
|
||||
yd(dx, dy, vd) += acc;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (is_gradient_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = G.GetShape();
|
||||
const int vdim = output.vdim;
|
||||
const int test_dim = output.size_on_qp / vdim;
|
||||
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d, q1d);
|
||||
auto yd = Reshape(&y(0, 0), d1d, d1d, vdim);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), q1d, d1d);
|
||||
auto s1 = Reshape(&scratch_mem[1](0), q1d, d1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t uv[2] = {0.0, 0.0};
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
uv[0] += fqp(vd, 0, qx, qy) * G(qx, 0, dx);
|
||||
uv[1] += fqp(vd, 1, qx, qy) * B(qx, 0, dx);
|
||||
}
|
||||
s0(qy, dx) = uv[0];
|
||||
s1(qy, dx) = uv[1];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t uv[2] = {0.0, 0.0};
|
||||
for (int qy = 0; qy < q1d; qy++)
|
||||
{
|
||||
uv[0] += s0(qy, dx) * B(qy, 0, dy);
|
||||
uv[1] += s1(qy, dx) * G(qy, 0, dy);
|
||||
}
|
||||
yd(dx, dy, vd) += uv[0] + uv[1];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (is_identity_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
|
||||
// // TODO: Check if this is the right fix for all cases
|
||||
// auto fqp = Reshape(&f(0, 0, 0), output.size_on_qp, q1d);
|
||||
// auto yqp = Reshape(&y(0, 0), output.size_on_qp, q1d);
|
||||
// for (int sq = 0; sq < output.size_on_qp; sq++)
|
||||
// {
|
||||
// MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
// {
|
||||
// yqp(sq, qx) = fqp(sq, qx);
|
||||
// }
|
||||
// MFEM_SYNC_THREAD;
|
||||
// }
|
||||
|
||||
auto fqp = Reshape(&f(0, 0, 0), output.size_on_qp, q1d, q1d);
|
||||
auto yqp = Reshape(&y(0, 0), output.size_on_qp, q1d, q1d);
|
||||
|
||||
for (int sq = 0; sq < output.size_on_qp; sq++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
yqp(sq, qx, qy) = fqp(sq, qx, qy);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (is_sum_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
auto fqp = Reshape(&f(0, 0, 0), q1d * q1d);
|
||||
auto yqp = Reshape(&y(0, 0), output.size_on_qp);
|
||||
for (int i = 0; i < q1d * q1d; i++)
|
||||
{
|
||||
yqp(0) += fqp(i);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("quadrature data mapping to field is not implemented for"
|
||||
" this field descriptor with sum factorization on tensor product elements");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename output_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_quadrature_data_to_fields_tensor_impl_3d(
|
||||
DeviceTensor<2, real_t> &y,
|
||||
const DeviceTensor<3, real_t> &f,
|
||||
const output_t &output,
|
||||
const DofToQuadMap &dtq,
|
||||
std::array<DeviceTensor<1>, 6> &scratch_mem)
|
||||
{
|
||||
[[maybe_unused]] auto B = dtq.B;
|
||||
[[maybe_unused]] auto G = dtq.G;
|
||||
|
||||
if constexpr (is_value_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = output.vdim;
|
||||
const int test_dim = output.size_on_qp / vdim;
|
||||
|
||||
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d, q1d, q1d);
|
||||
auto yd = Reshape(&y(0, 0), d1d, d1d, d1d, vdim);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), q1d, q1d, d1d);
|
||||
auto s1 = Reshape(&scratch_mem[1](0), q1d, d1d, d1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
acc += fqp(vd, 0, qx, qy, qz) * B(qx, 0, dx);
|
||||
}
|
||||
s0(qz, qy, dx) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int qy = 0; qy < q1d; qy++)
|
||||
{
|
||||
acc += s0(qz, qy, dx) * B(qy, 0, dy);
|
||||
}
|
||||
s1(qz, dy, dx) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int qz = 0; qz < q1d; qz++)
|
||||
{
|
||||
acc += s1(qz, dy, dx) * B(qz, 0, dz);
|
||||
}
|
||||
yd(dx, dy, dz, vd) += acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (is_gradient_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = G.GetShape();
|
||||
const int vdim = output.vdim;
|
||||
const int test_dim = output.size_on_qp / vdim;
|
||||
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d, q1d, q1d);
|
||||
auto yd = Reshape(&y(0, 0), d1d, d1d, d1d, vdim);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), q1d, q1d, d1d);
|
||||
auto s1 = Reshape(&scratch_mem[1](0), q1d, q1d, d1d);
|
||||
auto s2 = Reshape(&scratch_mem[2](0), q1d, q1d, d1d);
|
||||
auto s3 = Reshape(&scratch_mem[3](0), q1d, d1d, d1d);
|
||||
auto s4 = Reshape(&scratch_mem[4](0), q1d, d1d, d1d);
|
||||
auto s5 = Reshape(&scratch_mem[5](0), q1d, d1d, d1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t uvw[3] = {0.0, 0.0, 0.0};
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
uvw[0] += fqp(vd, 0, qx, qy, qz) * G(qx, 0, dx);
|
||||
uvw[1] += fqp(vd, 1, qx, qy, qz) * B(qx, 0, dx);
|
||||
uvw[2] += fqp(vd, 2, qx, qy, qz) * B(qx, 0, dx);
|
||||
}
|
||||
s0(qz, qy, dx) = uvw[0];
|
||||
s1(qz, qy, dx) = uvw[1];
|
||||
s2(qz, qy, dx) = uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t uvw[3] = {0.0, 0.0, 0.0};
|
||||
for (int qy = 0; qy < q1d; qy++)
|
||||
{
|
||||
uvw[0] += s0(qz, qy, dx) * B(qy, 0, dy);
|
||||
uvw[1] += s1(qz, qy, dx) * G(qy, 0, dy);
|
||||
uvw[2] += s2(qz, qy, dx) * B(qy, 0, dy);
|
||||
}
|
||||
s3(qz, dy, dx) = uvw[0];
|
||||
s4(qz, dy, dx) = uvw[1];
|
||||
s5(qz, dy, dx) = uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t uvw[3] = {0.0, 0.0, 0.0};
|
||||
for (int qz = 0; qz < q1d; qz++)
|
||||
{
|
||||
uvw[0] += s3(qz, dy, dx) * B(qz, 0, dz);
|
||||
uvw[1] += s4(qz, dy, dx) * B(qz, 0, dz);
|
||||
uvw[2] += s5(qz, dy, dx) * G(qz, 0, dz);
|
||||
}
|
||||
yd(dx, dy, dz, vd) += uvw[0] + uvw[1] + uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (is_identity_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
auto fqp = Reshape(&f(0, 0, 0), output.size_on_qp, q1d, q1d, q1d);
|
||||
auto yqp = Reshape(&y(0, 0), output.size_on_qp, q1d, q1d, q1d);
|
||||
|
||||
for (int sq = 0; sq < output.size_on_qp; sq++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
yqp(sq, qx, qy, qz) = fqp(sq, qx, qy, qz);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (is_sum_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
auto fqp = Reshape(&f(0, 0, 0), q1d * q1d * q1d);
|
||||
auto yqp = Reshape(&y(0, 0), output.size_on_qp);
|
||||
for (int i = 0; i < q1d * q1d * q1d; i++)
|
||||
{
|
||||
yqp(0) += fqp(i);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("quadrature data mapping to field is not implemented for"
|
||||
" this field descriptor with sum factorization on tensor product elements");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename output_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_quadrature_data_to_fields(
|
||||
DeviceTensor<2, real_t> &y,
|
||||
const DeviceTensor<3, real_t> &f,
|
||||
const output_t &output,
|
||||
const DofToQuadMap &dtq,
|
||||
std::array<DeviceTensor<1>, 6> &scratch_mem,
|
||||
const int &dimension,
|
||||
const bool &use_sum_factorization)
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
{
|
||||
map_quadrature_data_to_fields_tensor_impl_2d(y, f, output, dtq, scratch_mem);
|
||||
}
|
||||
else if (dimension == 3)
|
||||
{
|
||||
map_quadrature_data_to_fields_tensor_impl_3d(y, f, output, dtq, scratch_mem);
|
||||
}
|
||||
else { MFEM_ABORT_KERNEL("dimension not supported"); }
|
||||
}
|
||||
else
|
||||
{
|
||||
map_quadrature_data_to_fields_impl(y, f, output, dtq);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::future
|
||||
@@ -1,573 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "util.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
template <typename field_operator_t>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void map_field_to_quadrature_data_tensor_product_3d(
|
||||
DeviceTensor<2> &field_qp,
|
||||
const DofToQuadMap &dtq,
|
||||
const DeviceTensor<1> &field_e,
|
||||
const field_operator_t &input,
|
||||
const DeviceTensor<1, const real_t> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem)
|
||||
{
|
||||
[[maybe_unused]] auto B = dtq.B;
|
||||
[[maybe_unused]] auto G = dtq.G;
|
||||
|
||||
if constexpr (is_value_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
|
||||
auto fqp = Reshape(&field_qp[0], vdim, q1d, q1d, q1d);
|
||||
auto s0 = Reshape(&scratch_mem[0](0), d1d, d1d, q1d);
|
||||
auto s1 = Reshape(&scratch_mem[1](0), d1d, q1d, q1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
acc += B(qx, 0, dx) * field(dx, dy, dz, vd);
|
||||
}
|
||||
s0(dz, dy, qx) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int dy = 0; dy < d1d; dy++)
|
||||
{
|
||||
acc += s0(dz, dy, qx) * B(qy, 0, dy);
|
||||
}
|
||||
s1(dz, qy, qx) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int dz = 0; dz < d1d; dz++)
|
||||
{
|
||||
acc += s1(dz, qy, qx) * B(qz, 0, dz);
|
||||
}
|
||||
fqp(vd, qx, qy, qz) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (
|
||||
is_gradient_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const int dim = input.dim;
|
||||
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
|
||||
auto fqp = Reshape(&field_qp[0], vdim, dim, q1d, q1d, q1d);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), d1d, d1d, q1d);
|
||||
auto s1 = Reshape(&scratch_mem[1](0), d1d, d1d, q1d);
|
||||
auto s2 = Reshape(&scratch_mem[2](0), d1d, q1d, q1d);
|
||||
auto s3 = Reshape(&scratch_mem[3](0), d1d, q1d, q1d);
|
||||
auto s4 = Reshape(&scratch_mem[4](0), d1d, q1d, q1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t uv[2] = {0.0, 0.0};
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
const real_t f = field(dx, dy, dz, vd);
|
||||
uv[0] += f * B(qx, 0, dx);
|
||||
uv[1] += f * G(qx, 0, dx);
|
||||
}
|
||||
s0(dz, dy, qx) = uv[0];
|
||||
s1(dz, dy, qx) = uv[1];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t uvw[3] = {0.0, 0.0, 0.0};
|
||||
for (int dy = 0; dy < d1d; dy++)
|
||||
{
|
||||
const real_t s0i = s0(dz, dy, qx);
|
||||
uvw[0] += s1(dz, dy, qx) * B(qy, 0, dy);
|
||||
uvw[1] += s0i * G(qy, 0, dy);
|
||||
uvw[2] += s0i * B(qy, 0, dy);
|
||||
}
|
||||
s2(dz, qy, qx) = uvw[0];
|
||||
s3(dz, qy, qx) = uvw[1];
|
||||
s4(dz, qy, qx) = uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t uvw[3] = {0.0, 0.0, 0.0};
|
||||
for (int dz = 0; dz < d1d; dz++)
|
||||
{
|
||||
uvw[0] += s2(dz, qy, qx) * B(qz, 0, dz);
|
||||
uvw[1] += s3(dz, qy, qx) * B(qz, 0, dz);
|
||||
uvw[2] += s4(dz, qy, qx) * G(qz, 0, dz);
|
||||
}
|
||||
fqp(vd, 0, qx, qy, qz) = uvw[0];
|
||||
fqp(vd, 1, qx, qy, qz) = uvw[1];
|
||||
fqp(vd, 2, qx, qy, qz) = uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
// TODO: Create separate function for clarity
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<field_operator_t>, Weight>)
|
||||
{
|
||||
const int num_qp = integration_weights.GetShape()[0];
|
||||
// TODO: eeek
|
||||
const int q1d = (int)floor(std::pow(num_qp, 1.0/input.dim) + 0.5);
|
||||
auto w = Reshape(&integration_weights[0], q1d, q1d, q1d);
|
||||
auto f = Reshape(&field_qp[0], q1d, q1d, q1d);
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
f(qx, qy, qz) = w(qx, qy, qz);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else if constexpr (is_identity_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
const int q1d = B.GetShape()[0];
|
||||
auto field = Reshape(&field_e[0], input.size_on_qp, q1d * q1d * q1d);
|
||||
field_qp = field;
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(dfem::always_false<std::decay_t<field_operator_t>>,
|
||||
"can't map field to quadrature data");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename field_operator_t>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void map_field_to_quadrature_data_tensor_product_2d(
|
||||
DeviceTensor<2> &field_qp,
|
||||
const DofToQuadMap &dtq,
|
||||
const DeviceTensor<1> &field_e,
|
||||
const field_operator_t &input,
|
||||
const DeviceTensor<1, const real_t> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem)
|
||||
{
|
||||
[[maybe_unused]] auto B = dtq.B;
|
||||
[[maybe_unused]] auto G = dtq.G;
|
||||
|
||||
if constexpr (is_value_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const auto field = Reshape(&field_e[0], d1d, d1d, vdim);
|
||||
auto fqp = Reshape(&field_qp[0], vdim, q1d, q1d);
|
||||
auto s0 = Reshape(&scratch_mem[0](0), d1d, q1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
acc += B(qx, 0, dx) * field(dx, dy, vd);
|
||||
}
|
||||
s0(dy, qx) = acc;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int dy = 0; dy < d1d; dy++)
|
||||
{
|
||||
acc += s0(dy, qx) * B(qy, 0, dy);
|
||||
}
|
||||
fqp(vd, qx, qy) = acc;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (
|
||||
is_gradient_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const int dim = input.dim;
|
||||
const auto field = Reshape(&field_e[0], d1d, d1d, vdim);
|
||||
auto fqp = Reshape(&field_qp[0], vdim, dim, q1d, q1d);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), d1d, q1d);
|
||||
auto s1 = Reshape(&scratch_mem[1](0), d1d, q1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t uv[2] = {0.0, 0.0};
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
const real_t f = field(dx, dy, vd);
|
||||
uv[0] += f * B(qx, 0, dx);
|
||||
uv[1] += f * G(qx, 0, dx);
|
||||
}
|
||||
s0(dy, qx) = uv[0];
|
||||
s1(dy, qx) = uv[1];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t uv[2] = {0.0, 0.0};
|
||||
for (int dy = 0; dy < d1d; dy++)
|
||||
{
|
||||
const real_t s0i = s0(dy, qx);
|
||||
uv[0] += s1(dy, qx) * B(qy, 0, dy);
|
||||
uv[1] += s0i * G(qy, 0, dy);
|
||||
}
|
||||
fqp(vd, 0, qx, qy) = uv[0];
|
||||
fqp(vd, 1, qx, qy) = uv[1];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
// TODO: Create separate function for clarity
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<field_operator_t>, Weight>)
|
||||
{
|
||||
const int num_qp = integration_weights.GetShape()[0];
|
||||
// TODO: eeek
|
||||
const int q1d = (int)floor(std::pow(num_qp, 1.0/input.dim) + 0.5);
|
||||
auto w = Reshape(&integration_weights[0], q1d, q1d);
|
||||
auto f = Reshape(&field_qp[0], q1d, q1d);
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
f(qx, qy) = w(qx, qy);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else if constexpr (is_identity_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
const int q1d = B.GetShape()[0];
|
||||
auto field = Reshape(&field_e[0], input.size_on_qp, q1d * q1d);
|
||||
field_qp = field;
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(dfem::always_false<std::decay_t<field_operator_t>>,
|
||||
"can't map field to quadrature data");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename field_operator_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_field_to_quadrature_data(
|
||||
DeviceTensor<2> field_qp,
|
||||
const DofToQuadMap &dtq,
|
||||
const DeviceTensor<1> &field_e,
|
||||
const field_operator_t &input,
|
||||
const DeviceTensor<1, const real_t> &integration_weights)
|
||||
{
|
||||
[[maybe_unused]] auto B = dtq.B;
|
||||
[[maybe_unused]] auto G = dtq.G;
|
||||
if constexpr (is_value_fop<field_operator_t>::value)
|
||||
{
|
||||
auto [num_qp, dim, num_dof] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const auto field = Reshape(&field_e(0), num_dof, vdim);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
for (int qp = 0; qp < num_qp; qp++)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int dof = 0; dof < num_dof; dof++)
|
||||
{
|
||||
acc += B(qp, 0, dof) * field(dof, vd);
|
||||
}
|
||||
field_qp(vd, qp) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if constexpr (is_gradient_fop<field_operator_t>::value)
|
||||
{
|
||||
const auto [num_qp, dim, num_dof] = G.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const auto field = Reshape(&field_e(0), num_dof, vdim);
|
||||
|
||||
auto f = Reshape(&field_qp[0], vdim, dim, num_qp);
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
for (int qp = 0; qp < num_qp; qp++)
|
||||
{
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int dof = 0; dof < num_dof; dof++)
|
||||
{
|
||||
acc += G(qp, d, dof) * field(dof, vd);
|
||||
}
|
||||
f(vd, d, qp) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else if constexpr (std::is_same_v<field_operator_t, Weight>)
|
||||
{
|
||||
const int num_qp = integration_weights.GetShape()[0];
|
||||
auto f = Reshape(&field_qp[0], num_qp);
|
||||
for (int qp = 0; qp < num_qp; qp++)
|
||||
{
|
||||
f(qp) = integration_weights(qp);
|
||||
}
|
||||
}
|
||||
else if constexpr (is_identity_fop<field_operator_t>::value)
|
||||
{
|
||||
auto [num_qp, unused, num_dof] = B.GetShape();
|
||||
const int size_on_qp = input.size_on_qp;
|
||||
const auto field = Reshape(&field_e[0], size_on_qp * num_qp);
|
||||
auto f = Reshape(&field_qp[0], size_on_qp * num_qp);
|
||||
for (int i = 0; i < size_on_qp * num_qp; i++)
|
||||
{
|
||||
f(i) = field(i);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(dfem::always_false<field_operator_t>,
|
||||
"can't map field to quadrature data");
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
template <typename field_operator_ts, size_t num_inputs, size_t num_fields>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void map_fields_to_quadrature_data(
|
||||
std::array<DeviceTensor<2>, num_inputs> &fields_qp,
|
||||
const std::array<DeviceTensor<1>, num_fields> &fields_e,
|
||||
const std::array<DofToQuadMap, num_inputs> &dtqmaps,
|
||||
const std::array<int, num_inputs> &input_to_field,
|
||||
const field_operator_ts &fops,
|
||||
const DeviceTensor<1, const real_t> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem,
|
||||
const int &dimension,
|
||||
const bool &use_sum_factorization = false)
|
||||
{
|
||||
// When the input_to_field map returns -1, this means the requested input
|
||||
// is the integration weight. Weights don't have a user defined field
|
||||
// attached to them and we create a dummy field which is not accessed
|
||||
// inside the functions it is passed to.
|
||||
const auto dummy_field_weight = DeviceTensor<1>(nullptr, 0);
|
||||
for_constexpr<num_inputs>([&](auto i)
|
||||
{
|
||||
const DeviceTensor<1> &field_e =
|
||||
(input_to_field[i] == -1) ? dummy_field_weight : fields_e[input_to_field[i]];
|
||||
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_2d(
|
||||
fields_qp[i], dtqmaps[i], field_e, get<i>(fops),
|
||||
integration_weights, scratch_mem);
|
||||
}
|
||||
else if (dimension == 3)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_3d(
|
||||
fields_qp[i], dtqmaps[i], field_e, get<i>(fops),
|
||||
integration_weights, scratch_mem);
|
||||
}
|
||||
else
|
||||
{
|
||||
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
|
||||
MFEM_ABORT("unsupported dimension");
|
||||
#endif
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
map_field_to_quadrature_data(
|
||||
fields_qp[i], dtqmaps[i], field_e, get<i>(fops),
|
||||
integration_weights);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <typename field_operator_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_field_to_quadrature_data_conditional(
|
||||
DeviceTensor<2> &field_qp,
|
||||
const DeviceTensor<1> &field_e,
|
||||
const DofToQuadMap &dtqmap,
|
||||
field_operator_t &fop,
|
||||
const DeviceTensor<1, const real_t> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem,
|
||||
const bool &condition,
|
||||
const int &dimension,
|
||||
const bool &use_sum_factorization = false)
|
||||
{
|
||||
if (condition)
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_3d(
|
||||
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
|
||||
}
|
||||
else if (dimension == 3)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_2d(
|
||||
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
map_field_to_quadrature_data(
|
||||
field_qp, dtqmap, field_e, fop, integration_weights);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <size_t num_fields, size_t num_inputs, typename field_operator_ts>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_fields_to_quadrature_data_conditional(
|
||||
std::array<DeviceTensor<2>, num_inputs> &fields_qp,
|
||||
const std::array<DeviceTensor<1, const real_t>, num_fields> &fields_e,
|
||||
const std::array<DofToQuadMap, num_inputs> &dtqmaps,
|
||||
field_operator_ts fops,
|
||||
const DeviceTensor<1, const real_t> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem,
|
||||
const std::array<bool, num_inputs> &conditions,
|
||||
const bool &use_sum_factorization = false)
|
||||
{
|
||||
for_constexpr<num_inputs>([&](auto i)
|
||||
{
|
||||
map_field_to_quadrature_data_conditional(
|
||||
fields_qp[i], fields_e[i], dtqmaps[i], get<i>(fops), integration_weights,
|
||||
scratch_mem, conditions[i], use_sum_factorization);
|
||||
});
|
||||
}
|
||||
|
||||
template <size_t num_inputs, typename field_operator_ts>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_direction_to_quadrature_data_conditional(
|
||||
std::array<DeviceTensor<2>, num_inputs> &directions_qp,
|
||||
const DeviceTensor<1> &direction_e,
|
||||
const std::array<DofToQuadMap, num_inputs> &dtqmaps,
|
||||
field_operator_ts fops,
|
||||
const DeviceTensor<1, const real_t> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem,
|
||||
const std::array<bool, num_inputs> &conditions,
|
||||
const int &dimension,
|
||||
const bool &use_sum_factorization = false)
|
||||
{
|
||||
for_constexpr<num_inputs>([&](auto i)
|
||||
{
|
||||
if (conditions[i])
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_2d(
|
||||
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
|
||||
integration_weights, scratch_mem);
|
||||
}
|
||||
else if (dimension == 3)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_3d(
|
||||
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
|
||||
integration_weights, scratch_mem);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
map_field_to_quadrature_data(
|
||||
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
|
||||
integration_weights);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,153 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "../fe/fe_base.hpp"
|
||||
#include "../../fem/fespace.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
/// Base class for parametric spaces
|
||||
class ParameterSpace
|
||||
{
|
||||
public:
|
||||
ParameterSpace(int vdim = 1) : vdim(vdim) {}
|
||||
|
||||
/// @brief Get vector dimension at each point
|
||||
///
|
||||
/// This is the number of components at each point in the parametric space.
|
||||
int GetVDim() const { return vdim; }
|
||||
|
||||
/// Get DofToQuad information
|
||||
const DofToQuad& GetDofToQuad() const { return dtq; }
|
||||
|
||||
/// Get total size of the space (T-vector size)
|
||||
///
|
||||
/// returns the true size vsize of the space
|
||||
virtual int GetTrueVSize() const = 0;
|
||||
|
||||
/// Get local vector size (L-vector size)
|
||||
///
|
||||
/// returns the local size of the space
|
||||
virtual int GetVSize() const = 0;
|
||||
|
||||
/// Get spatial dimension
|
||||
///
|
||||
/// returns always 1.
|
||||
int Dimension() const
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
|
||||
/// @brief Get T-vector to L-vector transformation
|
||||
///
|
||||
/// returns identity by default that is lazy evaluated.
|
||||
virtual const Operator* GetProlongationMatrix() const
|
||||
{
|
||||
if (!prolongation)
|
||||
{
|
||||
prolongation.reset(new IdentityOperator(GetTrueVSize()));
|
||||
}
|
||||
return prolongation.get();
|
||||
}
|
||||
|
||||
/// @brief Get L-vector to E-vector transformation
|
||||
/// @note This is a mock call to replicate interface of FiniteElementSpace.
|
||||
/// It should not be used by a user.
|
||||
///
|
||||
/// returns identity by default that is lazy evaluated.
|
||||
virtual const Operator* GetElementRestriction(ElementDofOrdering o) const
|
||||
{
|
||||
if (!elem_restr)
|
||||
{
|
||||
elem_restr.reset(new IdentityOperator(GetVSize()));
|
||||
}
|
||||
return elem_restr.get();
|
||||
}
|
||||
|
||||
protected:
|
||||
int vdim;
|
||||
DofToQuad dtq;
|
||||
mutable std::unique_ptr<Operator> prolongation;
|
||||
mutable std::unique_ptr<Operator> elem_restr;
|
||||
};
|
||||
|
||||
/// @brief Uniform parameter space
|
||||
class UniformParameterSpace : public ParameterSpace
|
||||
{
|
||||
public:
|
||||
/// @brief Constructor for a uniform parameter space
|
||||
///
|
||||
/// @param mesh The mesh to determine dimension and number of elements.
|
||||
/// @param ir The integration rule to determine the number of quadrature points.
|
||||
/// @param vdim The vector dimension at each point.
|
||||
/// @param used_in_tensor_product If true, the number of quadrature points is
|
||||
/// calculated as the nth root of the number of points in the integration rule,
|
||||
/// where n is the mesh dimension. If false, the number of quadrature points is
|
||||
/// taken directly from the integration rule.
|
||||
UniformParameterSpace(Mesh &mesh, const IntegrationRule &ir, int vdim,
|
||||
bool used_in_tensor_product = true) :
|
||||
ParameterSpace(vdim)
|
||||
{
|
||||
// Setup DofToQuad information
|
||||
dtq.nqpt = (int)floor(std::pow(ir.GetNPoints(), 1.0 / mesh.Dimension()) + 0.5);
|
||||
dtq.ndof = dtq.nqpt;
|
||||
|
||||
// Calculate sizes
|
||||
const int num_qp = used_in_tensor_product ?
|
||||
std::pow(dtq.nqpt, mesh.Dimension()) :
|
||||
ir.GetNPoints();
|
||||
|
||||
tsize = vdim * num_qp * mesh.GetNE();
|
||||
lsize = tsize;
|
||||
}
|
||||
|
||||
int GetTrueVSize() const override
|
||||
{
|
||||
return tsize;
|
||||
}
|
||||
|
||||
int GetVSize() const override
|
||||
{
|
||||
return lsize;
|
||||
}
|
||||
|
||||
private:
|
||||
/// T-vector size
|
||||
int tsize;
|
||||
|
||||
/// L-vector size
|
||||
int lsize;
|
||||
};
|
||||
|
||||
class ParameterFunction : public Vector
|
||||
{
|
||||
public:
|
||||
ParameterFunction(ParameterSpace &space) :
|
||||
Vector(space.GetTrueVSize()),
|
||||
space(space)
|
||||
{}
|
||||
|
||||
/// @brief Get the ParameterSpace
|
||||
const ParameterSpace& GetParameterSpace() const
|
||||
{
|
||||
return space;
|
||||
}
|
||||
|
||||
using Vector::operator=;
|
||||
|
||||
private:
|
||||
/// the parametric space
|
||||
ParameterSpace &space;
|
||||
};
|
||||
|
||||
} // namespace mfem::future
|
||||
@@ -1,298 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "util.hpp"
|
||||
#include "qfunction_transform.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
/// @brief Call a qfunction with the given parameters.
|
||||
///
|
||||
/// @param qfunc the qfunction to call.
|
||||
/// @param input_shmem the input shared memory.
|
||||
/// @param residual_shmem the residual shared memory.
|
||||
/// @param rs_qp the size of the residual.
|
||||
/// @param num_qp the number of quadrature points.
|
||||
/// @param q1d the number of quadrature points in 1D.
|
||||
/// @param dimension the spatial dimension.
|
||||
/// @param use_sum_factorization whether to use sum factorization.
|
||||
/// @tparam qf_param_ts the tuple type of the qfunction parameters.
|
||||
template <
|
||||
typename qf_param_ts,
|
||||
typename qfunc_t,
|
||||
std::size_t num_fields>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void call_qfunction(
|
||||
qfunc_t &qfunc,
|
||||
const std::array<DeviceTensor<2>, num_fields> &input_shmem,
|
||||
DeviceTensor<2> &residual_shmem,
|
||||
const int &rs_qp,
|
||||
const int &num_qp,
|
||||
const int &q1d,
|
||||
const int &dimension,
|
||||
const bool &use_sum_factorization)
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
const int q = qx + q1d * qy;
|
||||
auto qf_args = decay_tuple<qf_param_ts> {};
|
||||
auto r = Reshape(&residual_shmem(0, q), rs_qp);
|
||||
apply_kernel(r, qfunc, qf_args, input_shmem, q);
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (dimension == 3)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
const int q = qx + q1d * (qy + q1d * qz);
|
||||
auto qf_args = decay_tuple<qf_param_ts> {};
|
||||
auto r = Reshape(&residual_shmem(0, q), rs_qp);
|
||||
apply_kernel(r, qfunc, qf_args, input_shmem, q);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
|
||||
MFEM_ABORT("unsupported dimension for sum factorization");
|
||||
#endif
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q, x, num_qp)
|
||||
{
|
||||
auto qf_args = decay_tuple<qf_param_ts> {};
|
||||
auto r = Reshape(&residual_shmem(0, q), rs_qp);
|
||||
apply_kernel(r, qfunc, qf_args, input_shmem, q);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// @brief Call a qfunction with the given parameters and
|
||||
/// compute it's derivative action.
|
||||
///
|
||||
/// @param qfunc the qfunction to call.
|
||||
/// @param input_shmem the input shared memory.
|
||||
/// @param shadow_shmem the shadow shared memory.
|
||||
/// @param residual_shmem the residual shared memory.
|
||||
/// @param das_qp the size of the derivative action.
|
||||
/// @param num_qp the number of quadrature points.
|
||||
/// @param q1d the number of quadrature points in 1D.
|
||||
/// @param dimension the spatial dimension.
|
||||
/// @param use_sum_factorization whether to use sum factorization.
|
||||
/// @tparam qf_param_ts the tuple type of the qfunction parameters.
|
||||
template <
|
||||
typename qf_param_ts,
|
||||
typename qfunc_t,
|
||||
std::size_t num_fields>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void call_qfunction_derivative_action(
|
||||
qfunc_t &qfunc,
|
||||
const std::array<DeviceTensor<2>, num_fields> &input_shmem,
|
||||
const std::array<DeviceTensor<2>, num_fields> &shadow_shmem,
|
||||
DeviceTensor<2> &residual_shmem,
|
||||
const int &das_qp,
|
||||
const int &num_qp,
|
||||
const int &q1d,
|
||||
const int &dimension,
|
||||
const bool &use_sum_factorization)
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
const int q = qx + q1d * qy;
|
||||
auto r = Reshape(&residual_shmem(0, q), das_qp);
|
||||
auto qf_args = decay_tuple<qf_param_ts> {};
|
||||
#ifdef MFEM_USE_ENZYME
|
||||
auto qf_shadow_args = decay_tuple<qf_param_ts> {};
|
||||
apply_kernel_fwddiff_enzyme(r, qfunc, qf_args, qf_shadow_args, input_shmem,
|
||||
shadow_shmem, q);
|
||||
#else
|
||||
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (dimension == 3)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
const int q = qx + q1d * (qy + q1d * qz);
|
||||
auto r = Reshape(&residual_shmem(0, q), das_qp);
|
||||
auto qf_args = decay_tuple<qf_param_ts> {};
|
||||
#ifdef MFEM_USE_ENZYME
|
||||
auto qf_shadow_args = decay_tuple<qf_param_ts> {};
|
||||
apply_kernel_fwddiff_enzyme(r, qfunc, qf_args, qf_shadow_args, input_shmem,
|
||||
shadow_shmem, q);
|
||||
#else
|
||||
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q, x, num_qp)
|
||||
{
|
||||
auto r = Reshape(&residual_shmem(0, q), das_qp);
|
||||
auto qf_args = decay_tuple<qf_param_ts> {};
|
||||
#ifdef MFEM_USE_ENZYME
|
||||
auto qf_shadow_args = decay_tuple<qf_param_ts> {};
|
||||
apply_kernel_fwddiff_enzyme(r, qfunc, qf_args, qf_shadow_args, input_shmem,
|
||||
shadow_shmem, q);
|
||||
#else
|
||||
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
|
||||
#endif
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename qfunc_t, typename args_ts, size_t num_args>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void apply_kernel(
|
||||
DeviceTensor<1, real_t> &f_qp,
|
||||
const qfunc_t &qfunc,
|
||||
args_ts &args,
|
||||
const std::array<DeviceTensor<2>, num_args> &u,
|
||||
int qp)
|
||||
{
|
||||
process_qf_args(u, args, qp);
|
||||
process_qf_result(f_qp, get<0>(apply(qfunc, args)));
|
||||
}
|
||||
|
||||
template <typename qfunc_t, typename arg_ts, size_t num_args>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void apply_kernel_native_dual(
|
||||
DeviceTensor<1, real_t> &f_qp,
|
||||
const qfunc_t &qfunc,
|
||||
arg_ts &args,
|
||||
const std::array<DeviceTensor<2>, num_args> &u,
|
||||
const std::array<DeviceTensor<2>, num_args> &v,
|
||||
const int &qp_idx)
|
||||
{
|
||||
process_qf_args(u, v, args, qp_idx);
|
||||
auto r = get<0>(apply(qfunc, args));
|
||||
process_derivative_from_native_dual(f_qp, r);
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_ENZYME
|
||||
|
||||
template <typename func_t, typename... arg_ts>
|
||||
MFEM_HOST_DEVICE inline
|
||||
auto qfunction_wrapper(const func_t &f, arg_ts &&...args)
|
||||
{
|
||||
return f(args...);
|
||||
}
|
||||
|
||||
// Version for active function arguments only
|
||||
//
|
||||
// This is an Enzyme regression and can be removed in later versions.
|
||||
template <typename qfunc_t, typename arg_ts, std::size_t... Is,
|
||||
typename inactive_arg_ts>
|
||||
MFEM_HOST_DEVICE inline
|
||||
auto fwddiff_apply_enzyme_indexed(qfunc_t &qfunc, arg_ts &&args,
|
||||
arg_ts &&shadow_args,
|
||||
std::index_sequence<Is...>,
|
||||
inactive_arg_ts &&inactive_args,
|
||||
std::index_sequence<>)
|
||||
{
|
||||
using qf_return_t = typename create_function_signature<
|
||||
decltype(&qfunc_t::operator())>::type::return_t;
|
||||
return __enzyme_fwddiff<qf_return_t>(
|
||||
qfunction_wrapper<qfunc_t, decltype(get<Is>(args))...>, enzyme_const,
|
||||
(void *)&qfunc, enzyme_dup, &get<Is>(args)..., enzyme_interleave,
|
||||
&get<Is>(shadow_args)...);
|
||||
}
|
||||
|
||||
// Interleave function arguments for enzyme
|
||||
template <typename qfunc_t, typename arg_ts, std::size_t... Is,
|
||||
typename inactive_arg_ts, std::size_t... Js>
|
||||
MFEM_HOST_DEVICE inline
|
||||
auto fwddiff_apply_enzyme_indexed(qfunc_t &qfunc, arg_ts &&args,
|
||||
arg_ts &&shadow_args,
|
||||
std::index_sequence<Is...>,
|
||||
inactive_arg_ts &&inactive_args,
|
||||
std::index_sequence<Js...>)
|
||||
{
|
||||
using qf_return_t = typename create_function_signature<
|
||||
decltype(&qfunc_t::operator())>::type::return_t;
|
||||
return __enzyme_fwddiff<qf_return_t>(
|
||||
qfunction_wrapper<qfunc_t, decltype(get<Is>(args))...,
|
||||
decltype(get<Js>(inactive_args))...>,
|
||||
enzyme_const, (void *)&qfunc, enzyme_dup, &get<Is>(args)...,
|
||||
enzyme_const, &get<Js>(inactive_args)..., enzyme_interleave,
|
||||
&get<Is>(shadow_args)...);
|
||||
}
|
||||
|
||||
template <typename qfunc_t, typename arg_ts, typename inactive_arg_ts>
|
||||
MFEM_HOST_DEVICE inline
|
||||
auto fwddiff_apply_enzyme(qfunc_t &qfunc, arg_ts &&args,
|
||||
arg_ts &&shadow_args,
|
||||
inactive_arg_ts &&inactive_args)
|
||||
{
|
||||
auto arg_indices = std::make_index_sequence<
|
||||
tuple_size<std::remove_reference_t<arg_ts>>::value> {};
|
||||
|
||||
auto inactive_arg_indices = std::make_index_sequence<
|
||||
tuple_size<std::remove_reference_t<inactive_arg_ts>>::value> {};
|
||||
|
||||
return fwddiff_apply_enzyme_indexed(qfunc, args, shadow_args, arg_indices,
|
||||
inactive_args, inactive_arg_indices);
|
||||
}
|
||||
|
||||
template <typename qfunc_t, typename arg_ts, size_t num_args>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void apply_kernel_fwddiff_enzyme(
|
||||
DeviceTensor<1, real_t> &f_qp,
|
||||
qfunc_t &qfunc,
|
||||
arg_ts &args,
|
||||
arg_ts &shadow_args,
|
||||
const std::array<DeviceTensor<2>, num_args> &u,
|
||||
const std::array<DeviceTensor<2>, num_args> &v,
|
||||
int qp_idx)
|
||||
{
|
||||
process_qf_args(u, args, qp_idx);
|
||||
process_qf_args(v, shadow_args, qp_idx);
|
||||
process_qf_result(f_qp,
|
||||
get<0>(fwddiff_apply_enzyme(qfunc, args, shadow_args, tuple<> {})));
|
||||
}
|
||||
#endif // MFEM_USE_ENZYME
|
||||
|
||||
} // namespace mfem::future
|
||||
@@ -1,346 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "util.hpp"
|
||||
#include "../../linalg/tensor.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
template <typename T0, typename T1, typename T2>
|
||||
MFEM_HOST_DEVICE
|
||||
void process_qf_arg(const T0 &, const T1 &, T2 &)
|
||||
{
|
||||
static_assert(dfem::always_false<T0, T1, T2>,
|
||||
"process_qf_arg not implemented for arg type");
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1, T> &u,
|
||||
const DeviceTensor<1, T> &v,
|
||||
T &arg)
|
||||
{
|
||||
arg = u(0);
|
||||
}
|
||||
|
||||
template <typename T, int n, int m>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
tensor<dual<T, T>, n, m> &arg)
|
||||
{
|
||||
for (int i = 0; i < m; i++)
|
||||
{
|
||||
for (int j = 0; j < n; j++)
|
||||
{
|
||||
arg(j, i).value = u((i * m) + j);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
dual<T, T> &arg)
|
||||
{
|
||||
arg.value = u(0);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
const DeviceTensor<1> &v,
|
||||
dual<T, T> &arg)
|
||||
{
|
||||
arg.value = u(0);
|
||||
arg.gradient = v(0);
|
||||
}
|
||||
|
||||
template <typename T, int n>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
const DeviceTensor<1> &v,
|
||||
tensor<dual<T, T>, n> &arg)
|
||||
{
|
||||
for (int i = 0; i < n; i++)
|
||||
{
|
||||
arg(i).value = u(i);
|
||||
arg(i).gradient = v(i);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int n, int m>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
const DeviceTensor<1> &v,
|
||||
tensor<dual<T, T>, n, m> &arg)
|
||||
{
|
||||
for (int i = 0; i < m; i++)
|
||||
{
|
||||
for (int j = 0; j < n; j++)
|
||||
{
|
||||
arg(j, i).value = u((i * m) + j);
|
||||
arg(j, i).gradient = v((i * m) + j);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int n>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_result(
|
||||
DeviceTensor<1, T> &r,
|
||||
const tensor<dual<T, T>, n> &x)
|
||||
{
|
||||
for (size_t i = 0; i < n; i++)
|
||||
{
|
||||
r(i) = x(i).value;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int n, int m>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_result(
|
||||
DeviceTensor<1, T> &r,
|
||||
const tensor<dual<T, T>, n, m> &x)
|
||||
{
|
||||
for (size_t i = 0; i < n; i++)
|
||||
{
|
||||
for (size_t j = 0; j < m; j++)
|
||||
{
|
||||
r(i + n * j) = x(i, j).value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename arg_type>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<2> &u,
|
||||
const DeviceTensor<2> &v,
|
||||
arg_type &arg,
|
||||
const int &qp)
|
||||
{
|
||||
const auto u_qp = Reshape(&u(0, qp), u.GetShape()[0]);
|
||||
const auto v_qp = Reshape(&v(0, qp), v.GetShape()[0]);
|
||||
process_qf_arg(u_qp, v_qp, arg);
|
||||
}
|
||||
|
||||
template <size_t num_fields, typename qf_args>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_args(
|
||||
const std::array<DeviceTensor<2>, num_fields> &u,
|
||||
const std::array<DeviceTensor<2>, num_fields> &v,
|
||||
qf_args &args,
|
||||
const int &qp)
|
||||
{
|
||||
for_constexpr<tuple_size<qf_args>::value>([&](auto i)
|
||||
{
|
||||
process_qf_arg(u[i], v[i], get<i>(args), qp);
|
||||
});
|
||||
}
|
||||
|
||||
template <typename T, int n, int m>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_derivative_from_native_dual(
|
||||
DeviceTensor<1, T> &r,
|
||||
const tensor<dual<T, T>, n, m> &x)
|
||||
{
|
||||
for (size_t i = 0; i < n; i++)
|
||||
{
|
||||
for (size_t j = 0; j < m; j++)
|
||||
{
|
||||
r(i + n * j) = x(i, j).gradient;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int n>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_derivative_from_native_dual(
|
||||
DeviceTensor<1, T> &r,
|
||||
const tensor<dual<T, T>, n> &x)
|
||||
{
|
||||
for (size_t i = 0; i < n; i++)
|
||||
{
|
||||
r(i) = x(i).gradient;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_derivative_from_native_dual(
|
||||
DeviceTensor<1, T> &r,
|
||||
const dual<T, T> &x)
|
||||
{
|
||||
r(0) = x.gradient;
|
||||
}
|
||||
|
||||
template <typename T0, typename T1>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(const T0 &, T1 &)
|
||||
{
|
||||
static_assert(dfem::always_false<T0, T1>,
|
||||
"process_qf_arg not implemented for arg type");
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1, T> &u,
|
||||
T &arg)
|
||||
{
|
||||
arg = u(0);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1, T> &u,
|
||||
tensor<T> &arg)
|
||||
{
|
||||
arg(0) = u(0);
|
||||
}
|
||||
|
||||
template <typename T, int n>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
tensor<T, n> &arg)
|
||||
{
|
||||
for (int i = 0; i < n; i++)
|
||||
{
|
||||
arg(i) = u(i);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int n, int m>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
tensor<T, n, m> &arg)
|
||||
{
|
||||
for (int i = 0; i < m; i++)
|
||||
{
|
||||
for (int j = 0; j < n; j++)
|
||||
{
|
||||
arg(j, i) = u((i * m) + j);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename arg_type>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(const DeviceTensor<2> &u, arg_type &arg, int qp)
|
||||
{
|
||||
const auto u_qp = Reshape(&u(0, qp), u.GetShape()[0]);
|
||||
process_qf_arg(u_qp, arg);
|
||||
}
|
||||
|
||||
template <size_t num_fields, typename qf_args>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_args(
|
||||
const std::array<DeviceTensor<2>, num_fields> &u,
|
||||
qf_args &args,
|
||||
const int &qp)
|
||||
{
|
||||
for_constexpr<tuple_size<qf_args>::value>([&](auto i)
|
||||
{
|
||||
process_qf_arg(u[i], get<i>(args), qp);
|
||||
});
|
||||
}
|
||||
|
||||
template <typename T0, typename T1>
|
||||
MFEM_HOST_DEVICE inline
|
||||
Vector process_qf_result(T0, T1)
|
||||
{
|
||||
static_assert(dfem::always_false<T0, T1>,
|
||||
"process_qf_result not implemented for result type");
|
||||
return Vector{};
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_result(
|
||||
DeviceTensor<1, T> &r,
|
||||
const T &x)
|
||||
{
|
||||
r(0) = x;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_result(
|
||||
DeviceTensor<1> &r,
|
||||
const dual<T, T> &x)
|
||||
{
|
||||
r(0) = x.value;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_result(
|
||||
DeviceTensor<1, T> &r,
|
||||
const tensor<T> &x)
|
||||
{
|
||||
r(0) = x(0);
|
||||
}
|
||||
|
||||
template <typename T, int n>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_result(
|
||||
DeviceTensor<1, T> &r,
|
||||
const tensor<T, n> &x)
|
||||
{
|
||||
for (size_t i = 0; i < n; i++)
|
||||
{
|
||||
r(i) = x(i);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int n, int m>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_result(
|
||||
DeviceTensor<1, T> &r,
|
||||
const tensor<T, n, m> &x)
|
||||
{
|
||||
for (size_t i = 0; i < n; i++)
|
||||
{
|
||||
for (size_t j = 0; j < m; j++)
|
||||
{
|
||||
r(i + n * j) = x(i, j);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int n, int m>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(
|
||||
const DeviceTensor<1, T> &u,
|
||||
const DeviceTensor<1, T> &v,
|
||||
tensor<T, n, m> &arg)
|
||||
{
|
||||
for (int i = 0; i < m; i++)
|
||||
{
|
||||
for (int j = 0; j < n; j++)
|
||||
{
|
||||
arg(j, i) = u((i * m) + j);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::future
|
||||
@@ -1,885 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
// This is serac's tuple implementation
|
||||
|
||||
#include <ostream>
|
||||
#include "../../config/config.hpp"
|
||||
#include <utility>
|
||||
|
||||
// Define a portable unreachable macro
|
||||
#if defined(__GNUC__) || defined(__clang__)
|
||||
#if defined(__CUDACC_VER_MAJOR__)
|
||||
#if __CUDACC_VER_MAJOR__ <= 11 && __CUDACC_VER_MINOR__ < 3
|
||||
// nvcc didn't add __builtin_unreachable() until cuda 11.3
|
||||
#define MFEM_UNREACHABLE()
|
||||
#else
|
||||
// nvcc >= 11.3
|
||||
#define MFEM_UNREACHABLE() __builtin_unreachable()
|
||||
#endif
|
||||
#else
|
||||
// host-only version
|
||||
#define MFEM_UNREACHABLE() __builtin_unreachable()
|
||||
#endif
|
||||
#elif defined(_MSC_VER)
|
||||
#define MFEM_UNREACHABLE() __assume(0)
|
||||
#endif
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @brief This is a class that mimics most of std::tuple's interface,
|
||||
* except that it is usable in CUDA kernels and admits some arithmetic operator overloads.
|
||||
*
|
||||
* see https://en.cppreference.com/w/cpp/utility/tuple for more information about std::tuple
|
||||
*/
|
||||
template <typename... T>
|
||||
struct tuple
|
||||
{
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
*/
|
||||
template <typename T0>
|
||||
struct tuple<T0>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1>
|
||||
struct tuple<T0, T1>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2>
|
||||
struct tuple<T0, T1, T2>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
* @tparam T3 The fourth type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2, typename T3>
|
||||
struct tuple<T0, T1, T2, T3>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
* @tparam T3 The fourth type stored in the tuple
|
||||
* @tparam T4 The fifth type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2, typename T3, typename T4>
|
||||
struct tuple<T0, T1, T2, T3, T4>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
T4 v4; ///< The fifth member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
* @tparam T3 The fourth type stored in the tuple
|
||||
* @tparam T4 The fifth type stored in the tuple
|
||||
* @tparam T5 The sixth type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5>
|
||||
struct tuple<T0, T1, T2, T3, T4, T5>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
T4 v4; ///< The fifth member of the tuple
|
||||
T5 v5; ///< The sixth member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
* @tparam T3 The fourth type stored in the tuple
|
||||
* @tparam T4 The fifth type stored in the tuple
|
||||
* @tparam T5 The sixth type stored in the tuple
|
||||
* @tparam T6 The seventh type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5, typename T6>
|
||||
struct tuple<T0, T1, T2, T3, T4, T5, T6>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
T4 v4; ///< The fifth member of the tuple
|
||||
T5 v5; ///< The sixth member of the tuple
|
||||
T6 v6; ///< The seventh member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
* @tparam T3 The fourth type stored in the tuple
|
||||
* @tparam T4 The fifth type stored in the tuple
|
||||
* @tparam T5 The sixth type stored in the tuple
|
||||
* @tparam T6 The seventh type stored in the tuple
|
||||
* @tparam T7 The eighth type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5, typename T6, typename T7>
|
||||
struct tuple<T0, T1, T2, T3, T4, T5, T6, T7>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
T4 v4; ///< The fifth member of the tuple
|
||||
T5 v5; ///< The sixth member of the tuple
|
||||
T6 v6; ///< The seventh member of the tuple
|
||||
T7 v7; ///< The eighth member of the tuple
|
||||
};
|
||||
|
||||
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5, typename T6, typename T7, typename T8>
|
||||
struct tuple<T0, T1, T2, T3, T4, T5, T6, T7, T8>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
T4 v4; ///< The fifth member of the tuple
|
||||
T5 v5; ///< The sixth member of the tuple
|
||||
T6 v6; ///< The seventh member of the tuple
|
||||
T7 v7; ///< The eighth member of the tuple
|
||||
T8 v8;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Class template argument deduction rule for tuples
|
||||
* @tparam T The variadic template parameter for tuple types
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE
|
||||
tuple(T...) -> tuple<T...>;
|
||||
|
||||
/**
|
||||
* @brief helper function for combining a list of values into a tuple
|
||||
* @tparam T types of the values to be tuple-d
|
||||
* @param args the actual values to be put into a tuple
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE tuple<T...> make_tuple(const T&... args)
|
||||
{
|
||||
return tuple<T...> {args...};
|
||||
}
|
||||
|
||||
template <class... Types>
|
||||
struct tuple_size
|
||||
{
|
||||
};
|
||||
|
||||
template <class... Types>
|
||||
struct tuple_size<tuple<Types...>> :
|
||||
std::integral_constant<std::size_t, sizeof...(Types)>
|
||||
{
|
||||
};
|
||||
|
||||
/**
|
||||
* @tparam i the tuple index to access
|
||||
* @tparam T the types stored in the tuple
|
||||
* @brief return a reference to the ith tuple entry
|
||||
*/
|
||||
template <int i, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto& get(tuple<T...>& values)
|
||||
{
|
||||
static_assert(i < sizeof...(T));
|
||||
if constexpr (i == 0)
|
||||
{
|
||||
return values.v0;
|
||||
}
|
||||
if constexpr (i == 1)
|
||||
{
|
||||
return values.v1;
|
||||
}
|
||||
if constexpr (i == 2)
|
||||
{
|
||||
return values.v2;
|
||||
}
|
||||
if constexpr (i == 3)
|
||||
{
|
||||
return values.v3;
|
||||
}
|
||||
if constexpr (i == 4)
|
||||
{
|
||||
return values.v4;
|
||||
}
|
||||
if constexpr (i == 5)
|
||||
{
|
||||
return values.v5;
|
||||
}
|
||||
if constexpr (i == 6)
|
||||
{
|
||||
return values.v6;
|
||||
}
|
||||
if constexpr (i == 7)
|
||||
{
|
||||
return values.v7;
|
||||
}
|
||||
if constexpr (i == 8)
|
||||
{
|
||||
return values.v8;
|
||||
}
|
||||
MFEM_UNREACHABLE();
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam i the tuple index to access
|
||||
* @tparam T the types stored in the tuple
|
||||
* @brief return a copy of the ith tuple entry
|
||||
*/
|
||||
template <int i, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr const auto& get(const tuple<T...>& values)
|
||||
{
|
||||
static_assert(i < sizeof...(T));
|
||||
if constexpr (i == 0)
|
||||
{
|
||||
return values.v0;
|
||||
}
|
||||
if constexpr (i == 1)
|
||||
{
|
||||
return values.v1;
|
||||
}
|
||||
if constexpr (i == 2)
|
||||
{
|
||||
return values.v2;
|
||||
}
|
||||
if constexpr (i == 3)
|
||||
{
|
||||
return values.v3;
|
||||
}
|
||||
if constexpr (i == 4)
|
||||
{
|
||||
return values.v4;
|
||||
}
|
||||
if constexpr (i == 5)
|
||||
{
|
||||
return values.v5;
|
||||
}
|
||||
if constexpr (i == 6)
|
||||
{
|
||||
return values.v6;
|
||||
}
|
||||
if constexpr (i == 7)
|
||||
{
|
||||
return values.v7;
|
||||
}
|
||||
if constexpr (i == 8)
|
||||
{
|
||||
return values.v8;
|
||||
}
|
||||
MFEM_UNREACHABLE();
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief a function intended to be used for extracting the ith type from a tuple.
|
||||
*
|
||||
* @note type<i>(my_tuple) returns a value, whereas get<i>(my_tuple) returns a reference
|
||||
*
|
||||
* @tparam i the index of the tuple to query
|
||||
* @tparam T the types stored in the tuple
|
||||
* @param values the tuple of values
|
||||
* @return a copy of the ith entry of the input
|
||||
*/
|
||||
template <int i, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto type(const tuple<T...>& values)
|
||||
{
|
||||
static_assert(i < sizeof...(T));
|
||||
if constexpr (i == 0)
|
||||
{
|
||||
return values.v0;
|
||||
}
|
||||
if constexpr (i == 1)
|
||||
{
|
||||
return values.v1;
|
||||
}
|
||||
if constexpr (i == 2)
|
||||
{
|
||||
return values.v2;
|
||||
}
|
||||
if constexpr (i == 3)
|
||||
{
|
||||
return values.v3;
|
||||
}
|
||||
if constexpr (i == 4)
|
||||
{
|
||||
return values.v4;
|
||||
}
|
||||
if constexpr (i == 5)
|
||||
{
|
||||
return values.v5;
|
||||
}
|
||||
if constexpr (i == 6)
|
||||
{
|
||||
return values.v6;
|
||||
}
|
||||
if constexpr (i == 7)
|
||||
{
|
||||
return values.v7;
|
||||
}
|
||||
if constexpr (i == 8)
|
||||
{
|
||||
return values.v8;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the + operator of tuples
|
||||
*
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param y tuple of values
|
||||
* @return the returned tuple sum
|
||||
*/
|
||||
template <typename... S, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto plus_helper(const tuple<S...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) + get<i>(y)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief return a tuple of values defined by elementwise sum of x and y
|
||||
*/
|
||||
template <typename... S, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator+(const tuple<S...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
static_assert(sizeof...(S) == sizeof...(T));
|
||||
return plus_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the += operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuples x and y
|
||||
* @tparam i integer sequence used to index the tuples
|
||||
* @param x tuple of values to be incremented
|
||||
* @param y tuple of increment values
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr void plus_equals_helper(tuple<T...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
((get<i>(x) += get<i>(y)), ...);
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuples x and y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief add values contained in y, to the tuple x
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator+=(tuple<T...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
return plus_equals_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the -= operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuples x and y
|
||||
* @tparam i integer sequence used to index the tuples
|
||||
* @param x tuple of values to be subracted from
|
||||
* @param y tuple of values to subtract from x
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr void minus_equals_helper(tuple<T...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
((get<i>(x) -= get<i>(y)), ...);
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuples x and y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief add values contained in y, to the tuple x
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator-=(tuple<T...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
return minus_equals_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the - operator of tuples
|
||||
*
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param y tuple of values
|
||||
* @return the returned tuple difference
|
||||
*/
|
||||
template <typename... S, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto minus_helper(const tuple<S...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) - get<i>(y)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief return a tuple of values defined by elementwise difference of x and y
|
||||
*/
|
||||
template <typename... S, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator-(const tuple<S...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
static_assert(sizeof...(S) == sizeof...(T));
|
||||
return minus_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the - operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @return the returned tuple difference
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto unary_minus_helper(const tuple<T...>& x,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{-get<i>(x)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of values
|
||||
* @brief return a tuple of values defined by applying the unary minus operator to each element of x
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator-(const tuple<T...>& x)
|
||||
{
|
||||
return unary_minus_helper(x,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the / operator of tuples
|
||||
*
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param y tuple of values
|
||||
* @return the returned tuple ratio
|
||||
*/
|
||||
template <typename... S, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto div_helper(const tuple<S...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) / get<i>(y)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief return a tuple of values defined by elementwise division of x by y
|
||||
*/
|
||||
template <typename... S, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator/(const tuple<S...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
static_assert(sizeof...(S) == sizeof...(T));
|
||||
return div_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the / operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param a the constant numerator
|
||||
* @return the returned tuple ratio
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto div_helper(const real_t a,
|
||||
const tuple<T...>& x, std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{a / get<i>(x)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the / operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param a the constant denomenator
|
||||
* @return the returned tuple ratio
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto div_helper(const tuple<T...>& x,
|
||||
const real_t a, std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) / a...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple x
|
||||
* @param a the numerator
|
||||
* @param x a tuple of denominator values
|
||||
* @brief return a tuple of values defined by division of a by the elements of x
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator/(const real_t a, const tuple<T...>& x)
|
||||
{
|
||||
return div_helper(a, x,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of numerator values
|
||||
* @param a a denominator
|
||||
* @brief return a tuple of values defined by elementwise division of x by a
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator/(const tuple<T...>& x, const real_t a)
|
||||
{
|
||||
return div_helper(x, a,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the * operator of tuples
|
||||
*
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param y tuple of values
|
||||
* @return the returned tuple product
|
||||
*/
|
||||
template <typename... S, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto mult_helper(const tuple<S...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) * get<i>(y)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief return a tuple of values defined by elementwise multiplication of x and y
|
||||
*/
|
||||
template <typename... S, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator*(const tuple<S...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
static_assert(sizeof...(S) == sizeof...(T));
|
||||
return mult_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the * operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param a a constant multiplier
|
||||
* @return the returned tuple product
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto mult_helper(const real_t a,
|
||||
const tuple<T...>& x, std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{a * get<i>(x)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the * operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param a a constant multiplier
|
||||
* @return the returned tuple product
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto mult_helper(const tuple<T...>& x,
|
||||
const real_t a, std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) * a...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @param a a scaling factor
|
||||
* @param x the tuple object
|
||||
* @brief multiply each component of x by the value a on the left
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator*(const real_t a, const tuple<T...>& x)
|
||||
{
|
||||
return mult_helper(a, x,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @param x the tuple object
|
||||
* @param a a scaling factor
|
||||
* @brief multiply each component of x by the value a on the right
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator*(const tuple<T...>& x, const real_t a)
|
||||
{
|
||||
return mult_helper(x, a,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @tparam i a list of indices used to acces each element of the tuple
|
||||
* @param out the ostream to write the output to
|
||||
* @param A the tuple of values
|
||||
* @brief helper used to implement printing a tuple of values
|
||||
*/
|
||||
template <typename... T, std::size_t... i>
|
||||
auto& print_helper(std::ostream& out, const tuple<T...>& A,
|
||||
std::integer_sequence<size_t, i...>)
|
||||
{
|
||||
out << "tuple{";
|
||||
(..., (out << (i == 0 ? "" : ", ") << get<i>(A)));
|
||||
out << "}";
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @param out the ostream to write the output to
|
||||
* @param A the tuple of values
|
||||
* @brief print a tuple of values
|
||||
*/
|
||||
template <typename... T>
|
||||
auto& operator<<(std::ostream& out, const tuple<T...>& A)
|
||||
{
|
||||
return print_helper(out, A, std::make_integer_sequence<size_t, sizeof...(T)>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper to apply a lambda to a tuple
|
||||
*
|
||||
* @tparam lambda The functor type
|
||||
* @tparam T The tuple types
|
||||
* @tparam i The integer sequence to i
|
||||
* @param f The functor to apply to the tuple
|
||||
* @param args The input tuple
|
||||
* @return The functor output
|
||||
*/
|
||||
template <typename lambda, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE auto apply_helper(lambda f, tuple<T...>& args,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return f(get<i>(args)...);
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam lambda a callable type
|
||||
* @tparam T the types of arguments to be passed in to f
|
||||
* @param f the callable object
|
||||
* @param args a tuple of arguments
|
||||
* @brief a way of passing an n-tuple to a function that expects n separate arguments
|
||||
*
|
||||
* e.g. foo(bar, baz) is equivalent to apply(foo, mfem::tuple(bar,baz));
|
||||
*/
|
||||
template <typename lambda, typename... T>
|
||||
MFEM_HOST_DEVICE auto apply(lambda f, tuple<T...>& args)
|
||||
{
|
||||
return apply_helper(f, std::move(args),
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @overload
|
||||
*/
|
||||
template <typename lambda, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE auto apply_helper(lambda f, const tuple<T...>& args,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return f(get<i>(args)...);
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam lambda a callable type
|
||||
* @tparam T the types of arguments to be passed in to f
|
||||
* @param f the callable object
|
||||
* @param args a tuple of arguments
|
||||
* @brief a way of passing an n-tuple to a function that expects n separate arguments
|
||||
*
|
||||
* e.g. foo(bar, baz) is equivalent to apply(foo, mfem::tuple(bar,baz));
|
||||
*/
|
||||
template <typename lambda, typename... T>
|
||||
MFEM_HOST_DEVICE auto apply(lambda f, const tuple<T...>& args)
|
||||
{
|
||||
return apply_helper(f, std::move(args),
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief a struct used to determine the type at index I of a tuple
|
||||
*
|
||||
* @note see: https://en.cppreference.com/w/cpp/utility/tuple/tuple_element
|
||||
*
|
||||
* @tparam I the index of the desired type
|
||||
* @tparam T a tuple of different types
|
||||
*/
|
||||
template <size_t I, class T>
|
||||
struct tuple_element;
|
||||
|
||||
// recursive case
|
||||
/// @overload
|
||||
template <size_t I, class Head, class... Tail>
|
||||
struct tuple_element<I, tuple<Head, Tail...>> : tuple_element<I - 1,
|
||||
tuple<Tail...>>
|
||||
{
|
||||
};
|
||||
|
||||
// base case
|
||||
/// @overload
|
||||
template <class Head, class... Tail>
|
||||
struct tuple_element<0, tuple<Head, Tail...>>
|
||||
{
|
||||
using type = Head; ///< the type at the specified index
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Trait for checking if a type is a @p mfem::tuple
|
||||
*/
|
||||
template <typename T>
|
||||
struct is_tuple : std::false_type
|
||||
{
|
||||
};
|
||||
|
||||
/// @overload
|
||||
template <typename... T>
|
||||
struct is_tuple<tuple<T...>> : std::true_type
|
||||
{
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Trait for checking if a type if a @p mfem::tuple containing only @p mfem::tuple
|
||||
*/
|
||||
template <typename T>
|
||||
struct is_tuple_of_tuples : std::false_type
|
||||
{
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Trait for checking if a type if a @p mfem::tuple containing only @p mfem::tuple
|
||||
*/
|
||||
template <typename... T>
|
||||
struct is_tuple_of_tuples<tuple<T...>>
|
||||
{
|
||||
static constexpr bool value = (is_tuple<T>::value &&
|
||||
...); ///< true/false result of type check
|
||||
};
|
||||
|
||||
/** @brief Auxiliary template function that merges (concatenates) two
|
||||
mfem::future::tuple types into a single std::tuple that is empty, i.e. it is
|
||||
value initialized. */
|
||||
template <typename... T1s, typename... T2s>
|
||||
constexpr auto merge_mfem_tuples_as_empty_std_tuple(
|
||||
const mfem::future::tuple<T1s...> &,
|
||||
const mfem::future::tuple<T2s...> &)
|
||||
{
|
||||
return std::tuple<T1s..., T2s...> {};
|
||||
}
|
||||
|
||||
} // namespace mfem::future
|
||||
-2298
File diff suppressed because it is too large
Load Diff
+52
-65
@@ -17,10 +17,7 @@
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct DGMassInvKernels { DGMassInvKernels(); };
|
||||
|
||||
DGMassInverse::DGMassInverse(const FiniteElementSpace &fes_orig,
|
||||
Coefficient *coeff,
|
||||
DGMassInverse::DGMassInverse(FiniteElementSpace &fes_orig, Coefficient *coeff,
|
||||
const IntegrationRule *ir,
|
||||
int btype)
|
||||
: Solver(fes_orig.GetTrueVSize()),
|
||||
@@ -30,8 +27,6 @@ DGMassInverse::DGMassInverse(const FiniteElementSpace &fes_orig,
|
||||
fes_orig.GetTypicalFE()->GetMapType()),
|
||||
fes(fes_orig.GetMesh(), &fec)
|
||||
{
|
||||
static DGMassInvKernels kernels;
|
||||
|
||||
MFEM_VERIFY(fes.IsDGSpace(), "Space must be DG.");
|
||||
MFEM_VERIFY(!fes.IsVariableOrder(), "Variable orders not supported.");
|
||||
|
||||
@@ -51,7 +46,7 @@ DGMassInverse::DGMassInverse(const FiniteElementSpace &fes_orig,
|
||||
const FiniteElement &fe = *fes.GetTypicalFE();
|
||||
d2q = &fe_orig.GetDofToQuad(fe.GetNodes(), mode);
|
||||
|
||||
const int n = d2q->ndof;
|
||||
int n = d2q->ndof;
|
||||
Array<real_t> B_inv = d2q->B; // deep copy
|
||||
Array<int> ipiv(n);
|
||||
// solver basis to original
|
||||
@@ -76,7 +71,7 @@ DGMassInverse::DGMassInverse(const FiniteElementSpace &fes_orig,
|
||||
// Only need transformed RHS if basis is different
|
||||
if (btype_orig != btype) { b2_.SetSize(height); }
|
||||
|
||||
M.reset(new BilinearForm(&fes));
|
||||
M = new BilinearForm(&fes);
|
||||
M->AddDomainIntegrator(m); // M assumes ownership of m
|
||||
M->SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
|
||||
@@ -84,19 +79,19 @@ DGMassInverse::DGMassInverse(const FiniteElementSpace &fes_orig,
|
||||
Update();
|
||||
}
|
||||
|
||||
DGMassInverse::DGMassInverse(const FiniteElementSpace &fes_, Coefficient &coeff,
|
||||
DGMassInverse::DGMassInverse(FiniteElementSpace &fes_, Coefficient &coeff,
|
||||
int btype)
|
||||
: DGMassInverse(fes_, &coeff, nullptr, btype) { }
|
||||
|
||||
DGMassInverse::DGMassInverse(const FiniteElementSpace &fes_, Coefficient &coeff,
|
||||
DGMassInverse::DGMassInverse(FiniteElementSpace &fes_, Coefficient &coeff,
|
||||
const IntegrationRule &ir, int btype)
|
||||
: DGMassInverse(fes_, &coeff, &ir, btype) { }
|
||||
|
||||
DGMassInverse::DGMassInverse(const FiniteElementSpace &fes_,
|
||||
DGMassInverse::DGMassInverse(FiniteElementSpace &fes_,
|
||||
const IntegrationRule &ir, int btype)
|
||||
: DGMassInverse(fes_, nullptr, &ir, btype) { }
|
||||
|
||||
DGMassInverse::DGMassInverse(const FiniteElementSpace &fes_, int btype)
|
||||
DGMassInverse::DGMassInverse(FiniteElementSpace &fes_, int btype)
|
||||
: DGMassInverse(fes_, nullptr, nullptr, btype) { }
|
||||
|
||||
void DGMassInverse::SetOperator(const Operator &op)
|
||||
@@ -117,7 +112,10 @@ void DGMassInverse::Update()
|
||||
diag_inv.Reciprocal();
|
||||
}
|
||||
|
||||
DGMassInverse::~DGMassInverse() = default;
|
||||
DGMassInverse::~DGMassInverse()
|
||||
{
|
||||
delete M;
|
||||
}
|
||||
|
||||
template<int DIM, int D1D, int Q1D>
|
||||
void DGMassInverse::DGMassCGIteration(const Vector &b_, Vector &u_) const
|
||||
@@ -271,58 +269,47 @@ void DGMassInverse::Mult(const Vector &Mu, Vector &u) const
|
||||
const int d1d = m->dofs1D;
|
||||
const int q1d = m->quad1D;
|
||||
|
||||
CGKernels::Run(dim, d1d, q1d, *this, Mu, u);
|
||||
const int id = (d1d << 4) | q1d;
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x11: return DGMassCGIteration<2,1,1>(Mu, u);
|
||||
case 0x22: return DGMassCGIteration<2,2,2>(Mu, u);
|
||||
case 0x33: return DGMassCGIteration<2,3,3>(Mu, u);
|
||||
case 0x35: return DGMassCGIteration<2,3,5>(Mu, u);
|
||||
case 0x44: return DGMassCGIteration<2,4,4>(Mu, u);
|
||||
case 0x46: return DGMassCGIteration<2,4,6>(Mu, u);
|
||||
case 0x55: return DGMassCGIteration<2,5,5>(Mu, u);
|
||||
case 0x57: return DGMassCGIteration<2,5,7>(Mu, u);
|
||||
case 0x66: return DGMassCGIteration<2,6,6>(Mu, u);
|
||||
case 0x68: return DGMassCGIteration<2,6,8>(Mu, u);
|
||||
default: return DGMassCGIteration<2>(Mu, u); // Fallback
|
||||
}
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x22: return DGMassCGIteration<3,2,2>(Mu, u);
|
||||
case 0x23: return DGMassCGIteration<3,2,3>(Mu, u);
|
||||
case 0x33: return DGMassCGIteration<3,3,3>(Mu, u);
|
||||
case 0x34: return DGMassCGIteration<3,3,4>(Mu, u);
|
||||
case 0x35: return DGMassCGIteration<3,3,5>(Mu, u);
|
||||
case 0x44: return DGMassCGIteration<3,4,4>(Mu, u);
|
||||
case 0x45: return DGMassCGIteration<3,4,5>(Mu, u);
|
||||
case 0x46: return DGMassCGIteration<3,4,6>(Mu, u);
|
||||
case 0x48: return DGMassCGIteration<3,4,8>(Mu, u);
|
||||
case 0x55: return DGMassCGIteration<3,5,5>(Mu, u);
|
||||
case 0x56: return DGMassCGIteration<3,5,6>(Mu, u);
|
||||
case 0x57: return DGMassCGIteration<3,5,7>(Mu, u);
|
||||
case 0x58: return DGMassCGIteration<3,5,8>(Mu, u);
|
||||
case 0x66: return DGMassCGIteration<3,6,6>(Mu, u);
|
||||
case 0x67: return DGMassCGIteration<3,6,7>(Mu, u);
|
||||
default: return DGMassCGIteration<3>(Mu, u); // Fallback
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DGMassInvKernels::DGMassInvKernels()
|
||||
{
|
||||
using k = DGMassInverse::CGKernels;
|
||||
// 2D
|
||||
k::Specialization<2,1,1>::Add();
|
||||
k::Specialization<2,2,2>::Add();
|
||||
k::Specialization<2,3,3>::Add();
|
||||
k::Specialization<2,3,5>::Add();
|
||||
k::Specialization<2,4,4>::Add();
|
||||
k::Specialization<2,4,6>::Add();
|
||||
k::Specialization<2,5,5>::Add();
|
||||
k::Specialization<2,5,7>::Add();
|
||||
k::Specialization<2,6,6>::Add();
|
||||
k::Specialization<2,6,8>::Add();
|
||||
// 3D
|
||||
k::Specialization<3,2,2>::Add();
|
||||
k::Specialization<3,2,3>::Add();
|
||||
k::Specialization<3,3,3>::Add();
|
||||
k::Specialization<3,3,4>::Add();
|
||||
k::Specialization<3,3,5>::Add();
|
||||
k::Specialization<3,4,4>::Add();
|
||||
k::Specialization<3,4,5>::Add();
|
||||
k::Specialization<3,4,6>::Add();
|
||||
k::Specialization<3,4,8>::Add();
|
||||
k::Specialization<3,5,5>::Add();
|
||||
k::Specialization<3,5,6>::Add();
|
||||
k::Specialization<3,5,7>::Add();
|
||||
k::Specialization<3,5,8>::Add();
|
||||
k::Specialization<3,6,6>::Add();
|
||||
k::Specialization<3,6,7>::Add();
|
||||
}
|
||||
|
||||
/// @cond Suppress_Doxygen_warnings
|
||||
|
||||
template <int DIM, int D1D, int Q1D>
|
||||
DGMassInverse::CGKernelType DGMassInverse::CGKernels::Kernel()
|
||||
{
|
||||
return &DGMassInverse::DGMassCGIteration<DIM,D1D,Q1D>;
|
||||
}
|
||||
|
||||
DGMassInverse::CGKernelType DGMassInverse::CGKernels::Fallback(
|
||||
int dim, int, int)
|
||||
{
|
||||
if (dim == 1) { return &DGMassInverse::DGMassCGIteration<1>; }
|
||||
else if (dim == 2) { return &DGMassInverse::DGMassCGIteration<2>; }
|
||||
else if (dim == 3) { return &DGMassInverse::DGMassCGIteration<3>; }
|
||||
else { MFEM_ABORT("Unsupported dimension."); }
|
||||
}
|
||||
|
||||
/// @endcond
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
+9
-15
@@ -14,8 +14,6 @@
|
||||
|
||||
#include "../linalg/operator.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "kernel_dispatch.hpp"
|
||||
#include <memory>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -34,7 +32,7 @@ protected:
|
||||
const DofToQuad *d2q; ///< Change of basis. Not owned.
|
||||
Array<real_t> B_; ///< Inverse of change of basis.
|
||||
Array<real_t> Bt_; ///< Inverse of change of basis, transposed.
|
||||
std::unique_ptr<class BilinearForm> M; ///< Mass bilinear form.
|
||||
class BilinearForm *M; ///< Mass bilinear form, owned.
|
||||
class MassIntegrator *m; ///< Mass integrator, owned by the form @ref M.
|
||||
Vector diag_inv; ///< Jacobi preconditioner.
|
||||
real_t rel_tol = 1e-12; ///< Relative CG tolerance.
|
||||
@@ -50,7 +48,7 @@ protected:
|
||||
///
|
||||
/// Custom coefficient and integration rule are used if @a coeff and @a ir
|
||||
/// are non-NULL.
|
||||
DGMassInverse(const FiniteElementSpace &fes_, Coefficient *coeff,
|
||||
DGMassInverse(FiniteElementSpace &fes_, Coefficient *coeff,
|
||||
const IntegrationRule *ir, int btype);
|
||||
public:
|
||||
/// @brief Construct the DG inverse mass operator for @a fes_.
|
||||
@@ -63,37 +61,36 @@ public:
|
||||
/// The solution and right-hand side used for the solver are not affected by
|
||||
/// this basis (they correspond to the basis of @a fes_). @a btype is only
|
||||
/// used internally, and only has an effect on the convergence rate.
|
||||
DGMassInverse(const FiniteElementSpace &fes_,
|
||||
int btype=BasisType::GaussLegendre);
|
||||
DGMassInverse(FiniteElementSpace &fes_, int btype=BasisType::GaussLegendre);
|
||||
/// @brief Construct the DG inverse mass operator for @a fes_ with
|
||||
/// Coefficient @a coeff.
|
||||
///
|
||||
/// @sa DGMassInverse(FiniteElementSpace&, int) for information about @a
|
||||
/// btype.
|
||||
DGMassInverse(const FiniteElementSpace &fes_, Coefficient &coeff,
|
||||
DGMassInverse(FiniteElementSpace &fes_, Coefficient &coeff,
|
||||
int btype=BasisType::GaussLegendre);
|
||||
/// @brief Construct the DG inverse mass operator for @a fes_ with
|
||||
/// Coefficient @a coeff and IntegrationRule @a ir.
|
||||
///
|
||||
/// @sa DGMassInverse(FiniteElementSpace&, int) for information about @a
|
||||
/// btype.
|
||||
DGMassInverse(const FiniteElementSpace &fes_, Coefficient &coeff,
|
||||
DGMassInverse(FiniteElementSpace &fes_, Coefficient &coeff,
|
||||
const IntegrationRule &ir, int btype=BasisType::GaussLegendre);
|
||||
/// @brief Construct the DG inverse mass operator for @a fes_ with
|
||||
/// IntegrationRule @a ir.
|
||||
///
|
||||
/// @sa DGMassInverse(FiniteElementSpace&, int) for information about @a
|
||||
/// btype.
|
||||
DGMassInverse(const FiniteElementSpace &fes_, const IntegrationRule &ir,
|
||||
DGMassInverse(FiniteElementSpace &fes_, const IntegrationRule &ir,
|
||||
int btype=BasisType::GaussLegendre);
|
||||
/// @brief Solve the system M b = u.
|
||||
///
|
||||
/// If @ref iterative_mode is @a true, @a u is used as an initial guess.
|
||||
void Mult(const Vector &b, Vector &u) const override;
|
||||
void Mult(const Vector &b, Vector &u) const;
|
||||
/// Same as Mult() since the mass matrix is symmetric.
|
||||
void MultTranspose(const Vector &b, Vector &u) const override { Mult(b, u); }
|
||||
void MultTranspose(const Vector &b, Vector &u) const { Mult(b, u); }
|
||||
/// Not implemented. Aborts.
|
||||
void SetOperator(const Operator &op) override;
|
||||
void SetOperator(const Operator &op);
|
||||
/// Set the relative tolerance.
|
||||
void SetRelTol(const real_t rel_tol_);
|
||||
/// Set the absolute tolerance.
|
||||
@@ -110,9 +107,6 @@ public:
|
||||
/// extended lambda used in an mfem::forall kernel (nvcc limitation)
|
||||
template<int DIM, int D1D = 0, int Q1D = 0>
|
||||
void DGMassCGIteration(const Vector &b_, Vector &u_) const;
|
||||
|
||||
using CGKernelType = void(DGMassInverse::*)(const Vector &b_, Vector &u) const;
|
||||
MFEM_REGISTER_KERNELS(CGKernels, CGKernelType, (int, int, int));
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -37,13 +37,6 @@ void DGMassApply(const int e,
|
||||
constexpr bool use_smem = (D1D > 0 && Q1D > 0);
|
||||
constexpr bool ACCUM = false;
|
||||
constexpr int NBZ = 1;
|
||||
|
||||
if (DIM == 1)
|
||||
{
|
||||
PAMassApply1D_Element<ACCUM>(e, NE, B, Bt, pa_data, x, y, d1d, q1d);
|
||||
return;
|
||||
}
|
||||
|
||||
if (use_smem)
|
||||
{
|
||||
// cannot specialize functions below with D1D or Q1D equal to zero
|
||||
@@ -179,43 +172,6 @@ real_t DGMassDot(const int e,
|
||||
return s_dot[0];
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void DGMassBasis1D(const int e,
|
||||
const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *x_,
|
||||
real_t *y_,
|
||||
const int d1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
|
||||
const auto b = Reshape(b_, D1D, D1D);
|
||||
const auto x = Reshape(x_, D1D, NE);
|
||||
auto y = Reshape(y_, D1D, NE);
|
||||
|
||||
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
real_t Y[MD1];
|
||||
|
||||
MFEM_FOREACH_THREAD(i,x,D1D)
|
||||
{
|
||||
real_t val = 0.0;
|
||||
for (int j = 0; j < D1D; ++j)
|
||||
{
|
||||
val += b(i,j)*x(j,e);
|
||||
}
|
||||
Y[i] = val;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
if (MFEM_THREAD_ID(y) == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(i,x,D1D)
|
||||
{
|
||||
y(i,e) = Y[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void DGMassBasis2D(const int e,
|
||||
@@ -313,11 +269,7 @@ void DGMassBasis(const int e,
|
||||
real_t *y_,
|
||||
const int d1d = 0)
|
||||
{
|
||||
if (DIM == 1)
|
||||
{
|
||||
DGMassBasis1D<T_D1D>(e, NE, b_, x_, y_, d1d);
|
||||
}
|
||||
else if (DIM == 2)
|
||||
if (DIM == 2)
|
||||
{
|
||||
DGMassBasis2D<T_D1D>(e, NE, b_, x_, y_, d1d);
|
||||
}
|
||||
|
||||
+43
-75
@@ -16,7 +16,9 @@ namespace mfem
|
||||
|
||||
void DofTransformation::TransformPrimal(real_t *v) const
|
||||
{
|
||||
if (IsIdentity()) { return; }
|
||||
MFEM_ASSERT(dof_trans_,
|
||||
"DofTransformation has no local transformation, call "
|
||||
"SetDofTransformation first!");
|
||||
int size = dof_trans_->Size();
|
||||
|
||||
if (vdim_ == 1 || (Ordering::Type)ordering_ == Ordering::byNODES)
|
||||
@@ -46,7 +48,9 @@ void DofTransformation::TransformPrimal(real_t *v) const
|
||||
|
||||
void DofTransformation::InvTransformPrimal(real_t *v) const
|
||||
{
|
||||
if (IsIdentity()) { return; }
|
||||
MFEM_ASSERT(dof_trans_,
|
||||
"DofTransformation has no local transformation, call "
|
||||
"SetDofTransformation first!");
|
||||
int size = dof_trans_->Height();
|
||||
|
||||
if (vdim_ == 1 || (Ordering::Type)ordering_ == Ordering::byNODES)
|
||||
@@ -76,7 +80,9 @@ void DofTransformation::InvTransformPrimal(real_t *v) const
|
||||
|
||||
void DofTransformation::TransformDual(real_t *v) const
|
||||
{
|
||||
if (IsIdentity()) { return; }
|
||||
MFEM_ASSERT(dof_trans_,
|
||||
"DofTransformation has no local transformation, call "
|
||||
"SetDofTransformation first!");
|
||||
int size = dof_trans_->Size();
|
||||
|
||||
if (vdim_ == 1 || (Ordering::Type)ordering_ == Ordering::byNODES)
|
||||
@@ -106,7 +112,9 @@ void DofTransformation::TransformDual(real_t *v) const
|
||||
|
||||
void DofTransformation::InvTransformDual(real_t *v) const
|
||||
{
|
||||
if (IsIdentity()) { return; }
|
||||
MFEM_ASSERT(dof_trans_,
|
||||
"DofTransformation has no local transformation, call "
|
||||
"SetDofTransformation first!");
|
||||
int size = dof_trans_->Size();
|
||||
|
||||
if (vdim_ == 1 || (Ordering::Type)ordering_ == Ordering::byNODES)
|
||||
@@ -134,33 +142,33 @@ void DofTransformation::InvTransformDual(real_t *v) const
|
||||
}
|
||||
}
|
||||
|
||||
void TransformPrimal(const DofTransformation &ran_dof_trans,
|
||||
const DofTransformation &dom_dof_trans,
|
||||
void TransformPrimal(const DofTransformation *ran_dof_trans,
|
||||
const DofTransformation *dom_dof_trans,
|
||||
DenseMatrix &elmat)
|
||||
{
|
||||
// No action if both transformations are NULL
|
||||
if (!ran_dof_trans.IsIdentity())
|
||||
if (ran_dof_trans)
|
||||
{
|
||||
ran_dof_trans.TransformPrimalCols(elmat);
|
||||
ran_dof_trans->TransformPrimalCols(elmat);
|
||||
}
|
||||
if (!dom_dof_trans.IsIdentity())
|
||||
if (dom_dof_trans)
|
||||
{
|
||||
dom_dof_trans.TransformDualRows(elmat);
|
||||
dom_dof_trans->TransformDualRows(elmat);
|
||||
}
|
||||
}
|
||||
|
||||
void TransformDual(const DofTransformation &ran_dof_trans,
|
||||
const DofTransformation &dom_dof_trans,
|
||||
void TransformDual(const DofTransformation *ran_dof_trans,
|
||||
const DofTransformation *dom_dof_trans,
|
||||
DenseMatrix &elmat)
|
||||
{
|
||||
// No action if both transformations are NULL
|
||||
if (!ran_dof_trans.IsIdentity())
|
||||
if (ran_dof_trans)
|
||||
{
|
||||
ran_dof_trans.TransformDualCols(elmat);
|
||||
ran_dof_trans->TransformDualCols(elmat);
|
||||
}
|
||||
if (!dom_dof_trans.IsIdentity())
|
||||
if (dom_dof_trans)
|
||||
{
|
||||
dom_dof_trans.TransformDualRows(elmat);
|
||||
dom_dof_trans->TransformDualRows(elmat);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -193,16 +201,13 @@ const DenseTensor ND_DofTransformation
|
||||
::TInv(const_cast<real_t *>(TInv_data), 2, 2, 6);
|
||||
|
||||
ND_DofTransformation::ND_DofTransformation(int size, int p, int num_edges,
|
||||
int num_faces,
|
||||
int face_types[])
|
||||
int num_tri_faces)
|
||||
: StatelessDofTransformation(size)
|
||||
, order(p)
|
||||
, nedofs(p)
|
||||
, ntdofs(p*(p-1))
|
||||
, nqdofs(2*p*(p-1))
|
||||
, nfdofs(p*(p-1))
|
||||
, nedges(num_edges)
|
||||
, nfaces(num_faces)
|
||||
, ftypes(face_types)
|
||||
, nfaces(num_tri_faces)
|
||||
{
|
||||
}
|
||||
|
||||
@@ -216,7 +221,6 @@ void ND_DofTransformation::TransformPrimal(const Array<int> & Fo,
|
||||
"Face orientation array is shorter than the number of faces in "
|
||||
"ND_DofTransformation");
|
||||
|
||||
int of = 0;
|
||||
real_t data[2];
|
||||
Vector v2(data, 2);
|
||||
DenseMatrix T2;
|
||||
@@ -224,19 +228,11 @@ void ND_DofTransformation::TransformPrimal(const Array<int> & Fo,
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<nfaces; f++)
|
||||
{
|
||||
if (ftypes[f] == Geometry::TRIANGLE)
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
for (int i=0; i<ntdofs/2; i++)
|
||||
{
|
||||
v2 = &v[nedges*nedofs + of + 2*i];
|
||||
T2.UseExternalData(const_cast<real_t *>(T.GetData(Fo[f])), 2, 2);
|
||||
T2.Mult(v2, &v[nedges*nedofs + of + 2*i]);
|
||||
}
|
||||
of += ntdofs;
|
||||
}
|
||||
else
|
||||
{
|
||||
of += nqdofs;
|
||||
v2 = &v[nedges*nedofs + f*nfdofs + 2*i];
|
||||
T2.UseExternalData(const_cast<real_t *>(T.GetData(Fo[f])), 2, 2);
|
||||
T2.Mult(v2, &v[nedges*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -251,7 +247,6 @@ void ND_DofTransformation::InvTransformPrimal(const Array<int> & Fo,
|
||||
"Face orientation array is shorter than the number of faces in "
|
||||
"ND_DofTransformation");
|
||||
|
||||
int of = 0;
|
||||
real_t data[2];
|
||||
Vector v2(data, 2);
|
||||
DenseMatrix T2Inv;
|
||||
@@ -259,19 +254,11 @@ void ND_DofTransformation::InvTransformPrimal(const Array<int> & Fo,
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<nfaces; f++)
|
||||
{
|
||||
if (ftypes[f] == Geometry::TRIANGLE)
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
for (int i=0; i<ntdofs/2; i++)
|
||||
{
|
||||
v2 = &v[nedges*nedofs + of + 2*i];
|
||||
T2Inv.UseExternalData(const_cast<real_t *>(TInv.GetData(Fo[f])), 2, 2);
|
||||
T2Inv.Mult(v2, &v[nedges*nedofs + of + 2*i]);
|
||||
}
|
||||
of += ntdofs;
|
||||
}
|
||||
else
|
||||
{
|
||||
of += nqdofs;
|
||||
v2 = &v[nedges*nedofs + f*nfdofs + 2*i];
|
||||
T2Inv.UseExternalData(const_cast<real_t *>(TInv.GetData(Fo[f])), 2, 2);
|
||||
T2Inv.Mult(v2, &v[nedges*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -285,7 +272,6 @@ void ND_DofTransformation::TransformDual(const Array<int> & Fo, real_t *v) const
|
||||
"Face orientation array is shorter than the number of faces in "
|
||||
"ND_DofTransformation");
|
||||
|
||||
int of = 0;
|
||||
real_t data[2];
|
||||
Vector v2(data, 2);
|
||||
DenseMatrix T2Inv;
|
||||
@@ -293,21 +279,12 @@ void ND_DofTransformation::TransformDual(const Array<int> & Fo, real_t *v) const
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<nfaces; f++)
|
||||
{
|
||||
if (ftypes[f] == Geometry::TRIANGLE)
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
for (int i=0; i<ntdofs/2; i++)
|
||||
{
|
||||
v2 = &v[nedges*nedofs + of + 2*i];
|
||||
T2Inv.UseExternalData(const_cast<real_t *>(TInv.GetData(Fo[f])), 2, 2);
|
||||
T2Inv.MultTranspose(v2, &v[nedges*nedofs + of + 2*i]);
|
||||
}
|
||||
of += ntdofs;
|
||||
v2 = &v[nedges*nedofs + f*nfdofs + 2*i];
|
||||
T2Inv.UseExternalData(const_cast<real_t *>(TInv.GetData(Fo[f])), 2, 2);
|
||||
T2Inv.MultTranspose(v2, &v[nedges*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
else
|
||||
{
|
||||
of += nqdofs;
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
@@ -321,7 +298,6 @@ void ND_DofTransformation::InvTransformDual(const Array<int> & Fo,
|
||||
"Face orientation array is shorter than the number of faces in "
|
||||
"ND_DofTransformation");
|
||||
|
||||
int of = 0;
|
||||
real_t data[2];
|
||||
Vector v2(data, 2);
|
||||
DenseMatrix T2;
|
||||
@@ -329,19 +305,11 @@ void ND_DofTransformation::InvTransformDual(const Array<int> & Fo,
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<nfaces; f++)
|
||||
{
|
||||
if (ftypes[f] == Geometry::TRIANGLE)
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
for (int i=0; i<ntdofs/2; i++)
|
||||
{
|
||||
v2 = &v[nedges*nedofs + of + 2*i];
|
||||
T2.UseExternalData(const_cast<real_t *>(T.GetData(Fo[f])), 2, 2);
|
||||
T2.MultTranspose(v2, &v[nedges*nedofs + of + 2*i]);
|
||||
}
|
||||
of += ntdofs;
|
||||
}
|
||||
else
|
||||
{
|
||||
of += nqdofs;
|
||||
v2 = &v[nedges*nedofs + f*nfdofs + 2*i];
|
||||
T2.UseExternalData(const_cast<real_t *>(T.GetData(Fo[f])), 2, 2);
|
||||
T2.MultTranspose(v2, &v[nedges*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+17
-40
@@ -201,19 +201,19 @@ public:
|
||||
inline int NumRows() const { return dof_trans_->NumRows(); }
|
||||
inline int Width() const { return dof_trans_->Width(); }
|
||||
inline int NumCols() const { return dof_trans_->NumCols(); }
|
||||
inline bool IsIdentity() const { return !dof_trans_ || dof_trans_->IsIdentity(); }
|
||||
inline bool IsIdentity() const { return dof_trans_->IsIdentity(); }
|
||||
|
||||
/** Transform local DoFs to align with the global DoFs. For example, this
|
||||
transformation can be used to map the local vector computed by
|
||||
FiniteElement::Project() to the transformed vector stored within a
|
||||
GridFunction object. */
|
||||
void TransformPrimal(real_t *v) const;
|
||||
inline void TransformPrimal(Vector &v) const { TransformPrimal(v.GetData()); }
|
||||
inline void TransformPrimal(Vector &v) const
|
||||
{ TransformPrimal(v.GetData()); }
|
||||
|
||||
/// Transform groups of DoFs stored as dense matrices
|
||||
inline void TransformPrimalCols(DenseMatrix &V) const
|
||||
{
|
||||
if (IsIdentity()) { return; }
|
||||
for (int c=0; c<V.Width(); c++)
|
||||
{
|
||||
TransformPrimal(V.GetColumn(c));
|
||||
@@ -251,7 +251,6 @@ public:
|
||||
/// Transform rows of a dense matrix containing dual DoFs
|
||||
inline void TransformDualRows(DenseMatrix &V) const
|
||||
{
|
||||
if (IsIdentity()) { return; }
|
||||
Vector row;
|
||||
for (int r=0; r<V.Height(); r++)
|
||||
{
|
||||
@@ -264,7 +263,6 @@ public:
|
||||
/// Transform columns of a dense matrix containing dual DoFs
|
||||
inline void TransformDualCols(DenseMatrix &V) const
|
||||
{
|
||||
if (IsIdentity()) { return; }
|
||||
for (int c=0; c<V.Width(); c++)
|
||||
{
|
||||
TransformDual(V.GetColumn(c));
|
||||
@@ -276,16 +274,16 @@ public:
|
||||
computed by a DiscreteInterpolator before copying into a
|
||||
DiscreteLinearOperator.
|
||||
*/
|
||||
void TransformPrimal(const DofTransformation &ran_dof_trans,
|
||||
const DofTransformation &dom_dof_trans,
|
||||
void TransformPrimal(const DofTransformation *ran_dof_trans,
|
||||
const DofTransformation *dom_dof_trans,
|
||||
DenseMatrix &elmat);
|
||||
|
||||
/** Transform a matrix of dual DoFs entries from different finite element spaces
|
||||
as computed by a BilinearFormIntegrator before summing into a
|
||||
MixedBilinearForm object.
|
||||
*/
|
||||
void TransformDual(const DofTransformation &ran_dof_trans,
|
||||
const DofTransformation &dom_dof_trans,
|
||||
void TransformDual(const DofTransformation *ran_dof_trans,
|
||||
const DofTransformation *dom_dof_trans,
|
||||
DenseMatrix &elmat);
|
||||
|
||||
/** Abstract base class for high-order Nedelec spaces on elements with
|
||||
@@ -308,16 +306,13 @@ private:
|
||||
static const DenseTensor T, TInv;
|
||||
|
||||
protected:
|
||||
const int order; // basis function order
|
||||
const int nedofs; // number of DoFs per edge
|
||||
const int ntdofs; // number of DoFs per triangular face
|
||||
const int nqdofs; // number of DoFs per quadrilateral face
|
||||
const int nedges; // number of edges per element
|
||||
const int nfaces; // number of faces per element
|
||||
const int *ftypes; // Pointer to array of Geometry::Type for each face
|
||||
const int order; // basis function order
|
||||
const int nedofs; // number of DoFs per edge
|
||||
const int nfdofs; // number of DoFs per face
|
||||
const int nedges; // number of edges per element
|
||||
const int nfaces; // number of triangular faces per element
|
||||
|
||||
ND_DofTransformation(int size, int order, int num_edges, int num_faces,
|
||||
int *face_types);
|
||||
ND_DofTransformation(int size, int order, int num_edges, int num_tri_faces);
|
||||
|
||||
public:
|
||||
// Return the 2x2 transformation operator for the given face orientation
|
||||
@@ -327,7 +322,7 @@ public:
|
||||
static const DenseMatrix & GetFaceInverseTransform(int ori)
|
||||
{ return TInv(ori); }
|
||||
|
||||
bool IsIdentity() const override { return ntdofs < 2; }
|
||||
bool IsIdentity() const override { return nfdofs < 2; }
|
||||
|
||||
void TransformPrimal(const Array<int> & Fo, real_t *v) const override;
|
||||
void InvTransformPrimal(const Array<int> & Fo, real_t *v) const override;
|
||||
@@ -339,11 +334,9 @@ public:
|
||||
/// triangles
|
||||
class ND_TriDofTransformation : public ND_DofTransformation
|
||||
{
|
||||
private:
|
||||
const int face_type[1] = { Geometry::TRIANGLE };
|
||||
public:
|
||||
ND_TriDofTransformation(int order)
|
||||
: ND_DofTransformation(order*(order + 2), order, 3, 1, (int *)face_type)
|
||||
: ND_DofTransformation(order*(order + 2), order, 3, 1)
|
||||
{}
|
||||
};
|
||||
|
||||
@@ -352,9 +345,7 @@ class ND_TetDofTransformation : public ND_DofTransformation
|
||||
{
|
||||
public:
|
||||
ND_TetDofTransformation(int order)
|
||||
: ND_DofTransformation(order*(order + 2)*(order + 3)/2, order, 6, 4,
|
||||
(int *)Geometry::Constants<Geometry::TETRAHEDRON>::
|
||||
FaceTypes)
|
||||
: ND_DofTransformation(order*(order + 2)*(order + 3)/2, order, 6, 4)
|
||||
{}
|
||||
};
|
||||
|
||||
@@ -364,21 +355,7 @@ class ND_WedgeDofTransformation : public ND_DofTransformation
|
||||
public:
|
||||
ND_WedgeDofTransformation(int order)
|
||||
: ND_DofTransformation(3 * order * ((order + 1) * (order + 2))/2,
|
||||
order, 9, 5,
|
||||
(int *)Geometry::Constants<Geometry::PRISM>::
|
||||
FaceTypes)
|
||||
{}
|
||||
};
|
||||
|
||||
/// DoF transformation implementation for the Nedelec basis on pyramid elements
|
||||
class ND_PyramidDofTransformation : public ND_DofTransformation
|
||||
{
|
||||
public:
|
||||
ND_PyramidDofTransformation(int order)
|
||||
: ND_DofTransformation(2 * order * (order * (order + 1) + 2),
|
||||
order, 8, 5,
|
||||
(int *)Geometry::Constants<Geometry::PYRAMID>::
|
||||
FaceTypes)
|
||||
order, 9, 2)
|
||||
{}
|
||||
};
|
||||
|
||||
|
||||
+10
-53
@@ -11,8 +11,6 @@
|
||||
|
||||
#include "../mesh/mesh_headers.hpp"
|
||||
#include "fem.hpp"
|
||||
#include "eltrans/eltrans_basis.hpp"
|
||||
|
||||
#include <cmath>
|
||||
|
||||
namespace mfem
|
||||
@@ -68,6 +66,7 @@ const DenseMatrix &ElementTransformation::EvalInverseJ()
|
||||
return invJ;
|
||||
}
|
||||
|
||||
|
||||
int InverseElementTransformation::FindClosestPhysPoint(
|
||||
const Vector& pt, const IntegrationRule &ir)
|
||||
{
|
||||
@@ -181,13 +180,12 @@ int InverseElementTransformation::NewtonSolve(const Vector &pt,
|
||||
const int dim = T->GetDimension();
|
||||
const int sdim = T->GetSpaceDim();
|
||||
IntegrationPoint xip, prev_xip;
|
||||
real_t xd[3], yd[3], dxd[3], dxpd[3], dx_norm = -1.0, err_phys,
|
||||
real_dx_norm = -1.0;
|
||||
Vector x(xd, dim), y(yd, sdim), dx(dxd, dim), dx_prev(dxpd, dim);
|
||||
real_t xd[3], yd[3], dxd[3], dx_norm = -1.0, err_phys, real_dx_norm = -1.0;
|
||||
Vector x(xd, dim), y(yd, sdim), dx(dxd, dim);
|
||||
bool hit_bdr = false, prev_hit_bdr = false;
|
||||
|
||||
// Use ip0 as initial guess:
|
||||
xip = ip0;
|
||||
xip = *ip0;
|
||||
xip.Get(xd, dim); // xip -> x
|
||||
if (print_level >= 3)
|
||||
{
|
||||
@@ -344,18 +342,16 @@ int InverseElementTransformation::Transform(const Vector &pt,
|
||||
switch (init_guess_type)
|
||||
{
|
||||
case Center:
|
||||
ip0 = Geometries.GetCenter(T->GetGeometryType());
|
||||
ip0 = &Geometries.GetCenter(T->GetGeometryType());
|
||||
break;
|
||||
|
||||
case ClosestPhysNode:
|
||||
case ClosestRefNode:
|
||||
{
|
||||
const int order = qpts_order >= 0
|
||||
? qpts_order
|
||||
: std::max(T->Order() + rel_qpts_order, 0);
|
||||
const int order = std::max(T->Order()+rel_qpts_order, 0);
|
||||
if (order == 0)
|
||||
{
|
||||
ip0 = Geometries.GetCenter(T->GetGeometryType());
|
||||
ip0 = &Geometries.GetCenter(T->GetGeometryType());
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -363,45 +359,11 @@ int InverseElementTransformation::Transform(const Vector &pt,
|
||||
int closest_idx = (init_guess_type == ClosestPhysNode) ?
|
||||
FindClosestPhysPoint(pt, RefG.RefPts) :
|
||||
FindClosestRefPoint(pt, RefG.RefPts);
|
||||
ip0 = RefG.RefPts.IntPoint(closest_idx);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case EdgeScan:
|
||||
{
|
||||
const int order = qpts_order >= 0
|
||||
? qpts_order
|
||||
: std::max(T->Order() + rel_qpts_order, 0);
|
||||
if (order == 0)
|
||||
{
|
||||
ip0 = Geometries.GetCenter(T->GetGeometryType());
|
||||
}
|
||||
else
|
||||
{
|
||||
auto &ir = *refiner.EdgeScan(T->GetGeometryType(), order + 1);
|
||||
int res = Outside;
|
||||
int npts = ir.GetNPoints();
|
||||
// will return Inside if any test point reports Inside, Outside if
|
||||
// all points report Outside, else Unknown
|
||||
for (int i = 0; i < npts; ++i)
|
||||
{
|
||||
ip0 = ir.IntPoint(i);
|
||||
int tmp_res = NewtonSolve(pt, ip);
|
||||
switch (tmp_res)
|
||||
{
|
||||
case Inside:
|
||||
return Inside;
|
||||
case Outside:
|
||||
break;
|
||||
case Unknown:
|
||||
res = Unknown;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return res;
|
||||
ip0 = &RefG.RefPts.IntPoint(closest_idx);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case GivenPoint:
|
||||
break;
|
||||
|
||||
@@ -482,8 +444,6 @@ int IsoparametricTransformation::OrderJ() const
|
||||
return (FElem->GetOrder()-1);
|
||||
case FunctionSpace::Qk:
|
||||
return (FElem->GetOrder());
|
||||
case FunctionSpace::Uk:
|
||||
return (FElem->GetOrder());
|
||||
default:
|
||||
MFEM_ABORT("unsupported finite element");
|
||||
}
|
||||
@@ -498,8 +458,6 @@ int IsoparametricTransformation::OrderW() const
|
||||
return (FElem->GetOrder() - 1) * FElem->GetDim();
|
||||
case FunctionSpace::Qk:
|
||||
return (FElem->GetOrder() * FElem->GetDim() - 1);
|
||||
case FunctionSpace::Uk:
|
||||
return (FElem->GetOrder() * FElem->GetDim() - 1);
|
||||
default:
|
||||
MFEM_ABORT("unsupported finite element");
|
||||
}
|
||||
@@ -519,8 +477,6 @@ int IsoparametricTransformation::OrderGrad(const FiniteElement *fe) const
|
||||
return ((k-1)*(d-1)+(l-1));
|
||||
case FunctionSpace::Qk:
|
||||
return (k*(d-1)+(l-1));
|
||||
case FunctionSpace::Uk:
|
||||
return (k*(d-1)+(l-1));
|
||||
default:
|
||||
MFEM_ABORT("unsupported finite element");
|
||||
}
|
||||
@@ -756,4 +712,5 @@ real_t FaceElementTransformations::CheckConsistency(int print_level,
|
||||
|
||||
return max_dist;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+30
-281
@@ -17,13 +17,9 @@
|
||||
#include "intrules.hpp"
|
||||
#include "fe.hpp"
|
||||
|
||||
#include "kernel_dispatch.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class GridFunction;
|
||||
|
||||
class ElementTransformation
|
||||
{
|
||||
protected:
|
||||
@@ -202,46 +198,40 @@ public:
|
||||
/// Algorithms for selecting an initial guess.
|
||||
enum InitGuessType
|
||||
{
|
||||
Center = 0, ///< Use the center of the reference element.
|
||||
Center = 0, ///< Use the center of the reference element.
|
||||
ClosestPhysNode = 1, /**<
|
||||
Use the point returned by FindClosestPhysPoint() from a reference-space
|
||||
grid of type and size controlled by SetInitGuessPointsType() and
|
||||
SetInitGuessRelOrder(), respectively. */
|
||||
ClosestRefNode = 2, /**<
|
||||
Use the point returned by FindClosestRefPoint() from a reference-space
|
||||
grid of type and size controlled by SetInitGuessPointsType() and
|
||||
SetInitGuessRelOrder(), respectively. */
|
||||
GivenPoint = 3, ///< Use a specific point, set with SetInitialGuess().
|
||||
EdgeScan =
|
||||
4, /**< Performs full solves on multiple points along the r/s/t=0 edges
|
||||
of the element. It is recommended that SetInitGuessRelOrder() is
|
||||
chosen such that max(trans_order+order,0)+1 <= 4 with
|
||||
SetInitGuessPointsType() as Quadrature1D::ClosedUniform. @see
|
||||
GeometryRefiner::EdgeScan */
|
||||
Use the point returned by FindClosestPhysPoint() from a reference-space
|
||||
grid of type and size controlled by SetInitGuessPointsType() and
|
||||
SetInitGuessRelOrder(), respectively. */
|
||||
ClosestRefNode = 2, /**<
|
||||
Use the point returned by FindClosestRefPoint() from a reference-space
|
||||
grid of type and size controlled by SetInitGuessPointsType() and
|
||||
SetInitGuessRelOrder(), respectively. */
|
||||
GivenPoint = 3 ///< Use a specific point, set with SetInitialGuess().
|
||||
};
|
||||
|
||||
/// Solution strategy.
|
||||
enum SolverType
|
||||
{
|
||||
Newton = 0, /**<
|
||||
Use Newton's algorithm, without restricting the reference-space points
|
||||
(iterates) to the reference element. */
|
||||
Newton = 0, /**<
|
||||
Use Newton's algorithm, without restricting the reference-space points
|
||||
(iterates) to the reference element. */
|
||||
NewtonSegmentProject = 1, /**<
|
||||
Use Newton's algorithm, restricting the reference-space points to the
|
||||
reference element by scaling back the Newton increments, i.e.
|
||||
projecting new iterates, x_new, lying outside the element, to the
|
||||
intersection of the line segment [x_old, x_new] with the boundary. */
|
||||
NewtonElementProject = 2, /**<
|
||||
Use Newton's algorithm, restricting the reference-space points to the
|
||||
reference element by projecting new iterates, x_new, lying outside the
|
||||
element, to the point on the boundary closest (in reference-space) to
|
||||
x_new. */
|
||||
Use Newton's algorithm, restricting the reference-space points to the
|
||||
reference element by scaling back the Newton increments, i.e.
|
||||
projecting new iterates, x_new, lying outside the element, to the
|
||||
intersection of the line segment [x_old, x_new] with the boundary. */
|
||||
NewtonElementProject = 2 /**<
|
||||
Use Newton's algorithm, restricting the reference-space points to the
|
||||
reference element by projecting new iterates, x_new, lying outside the
|
||||
element, to the point on the boundary closest (in reference-space) to
|
||||
x_new. */
|
||||
};
|
||||
|
||||
/// Values returned by Transform().
|
||||
enum TransformResult
|
||||
{
|
||||
Inside = 0, ///< The point is inside the element
|
||||
Inside = 0, ///< The point is inside the element
|
||||
Outside = 1, ///< The point is _probably_ outside the element
|
||||
Unknown = 2 ///< The algorithm failed to determine where the point is
|
||||
};
|
||||
@@ -251,11 +241,9 @@ protected:
|
||||
ElementTransformation *T;
|
||||
|
||||
// Parameters of the inversion algorithms:
|
||||
IntegrationPoint ip0;
|
||||
const IntegrationPoint *ip0;
|
||||
int init_guess_type; // algorithm to use
|
||||
GeometryRefiner refiner; // geometry refiner for initial guess
|
||||
int qpts_order; // num_1D_qpts = rel_qpts_order + 1, or < 0 to use
|
||||
// rel_qpts_order.
|
||||
int rel_qpts_order; // num_1D_qpts = max(trans_order+rel_qpts_order,0)+1
|
||||
int solver_type; // solution strategy to use
|
||||
int max_iter; // max. number of Newton iterations
|
||||
@@ -296,19 +284,19 @@ public:
|
||||
tolerances. */
|
||||
InverseElementTransformation(ElementTransformation *Trans = NULL)
|
||||
: T(Trans),
|
||||
ip0(NULL),
|
||||
init_guess_type(Center),
|
||||
refiner(Quadrature1D::OpenHalfUniform),
|
||||
qpts_order(-1),
|
||||
rel_qpts_order(-1),
|
||||
solver_type(NewtonElementProject),
|
||||
max_iter(16),
|
||||
#ifdef MFEM_USE_DOUBLE
|
||||
ref_tol(1e-15),
|
||||
phys_rtol(4e-15),
|
||||
phys_rtol(1e-15),
|
||||
ip_tol(1e-8),
|
||||
#elif defined(MFEM_USE_SINGLE)
|
||||
ref_tol(4e-7),
|
||||
phys_rtol(1e-6),
|
||||
ref_tol(1e-7),
|
||||
phys_rtol(1e-7),
|
||||
ip_tol(1e-4),
|
||||
#endif
|
||||
print_level(-1)
|
||||
@@ -326,28 +314,16 @@ public:
|
||||
/** @brief Set the initial guess for subsequent calls to Transform(),
|
||||
switching to the #GivenPoint #InitGuessType at the same time. */
|
||||
void SetInitialGuess(const IntegrationPoint &init_ip)
|
||||
{ ip0 = init_ip; SetInitialGuessType(GivenPoint); }
|
||||
{ ip0 = &init_ip; SetInitialGuessType(GivenPoint); }
|
||||
|
||||
/// Set the Quadrature1D type used for the `Closest*` and `EdgeScan` initial
|
||||
/// guess types.
|
||||
/// Set the Quadrature1D type used for the `Closest*` initial guess types.
|
||||
void SetInitGuessPointsType(int q_type) { refiner.SetType(q_type); }
|
||||
|
||||
/// Set the relative order used for the `Closest*` initial guess types.
|
||||
/** The number of points in each spatial direction is given by the formula
|
||||
max(trans_order+order,0)+1, where trans_order is the order of the current
|
||||
ElementTransformation. */
|
||||
void SetInitGuessRelOrder(int order)
|
||||
{
|
||||
qpts_order = -1;
|
||||
rel_qpts_order = order;
|
||||
}
|
||||
|
||||
/** The number of points in each spatial direction is given by the formula
|
||||
order+1. */
|
||||
void SetInitGuessOrder(int order)
|
||||
{
|
||||
qpts_order = order;
|
||||
}
|
||||
void SetInitGuessRelOrder(int order) { rel_qpts_order = order; }
|
||||
|
||||
/** @brief Specify which algorithm to use for solving the transformation
|
||||
equation, i.e. when calling the Transform() method. */
|
||||
@@ -397,233 +373,6 @@ public:
|
||||
virtual int Transform(const Vector &pt, IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Performs batch inverse element transforms. Currently only supports
|
||||
* non-mixed meshes with SEGMENT, SQUARE, or CUBE geometries. Mixed
|
||||
* element order meshes are projected onto an equivalent uniform order mesh.
|
||||
*/
|
||||
class BatchInverseElementTransformation
|
||||
{
|
||||
// nodes grid function, not owned
|
||||
const GridFunction *gf_ = nullptr;
|
||||
// initial guess algorithm to use
|
||||
InverseElementTransformation::InitGuessType init_guess_type =
|
||||
InverseElementTransformation::ClosestPhysNode;
|
||||
int qpts_order = -1; // num_1D_qpts = rel_qpts_order + 1, or < 0 to use
|
||||
// rel_qpts_order.
|
||||
// num_1D_qpts = max(trans_order+rel_qpts_order,0)+1
|
||||
int rel_qpts_order = 0;
|
||||
// solution strategy to use
|
||||
InverseElementTransformation::SolverType solver_type =
|
||||
InverseElementTransformation::NewtonElementProject;
|
||||
// basis type stored in points1d
|
||||
int basis_type = BasisType::Invalid;
|
||||
// initial guess points type. Quadrature1D::Invalid is used for match
|
||||
// basis_type.
|
||||
int guess_points_type = Quadrature1D::Invalid;
|
||||
// max. number of Newton iterations
|
||||
int max_iter = 16;
|
||||
// internal element node locations cache
|
||||
Vector node_pos;
|
||||
#ifdef MFEM_USE_DOUBLE
|
||||
// reference space tolerance
|
||||
real_t ref_tol = 1e-15;
|
||||
// physical space tolerance (relative)
|
||||
real_t phys_rtol = 4e-15;
|
||||
#else
|
||||
// reference space tolerance
|
||||
real_t ref_tol = 4e-7;
|
||||
// physical space tolerance (relative)
|
||||
real_t phys_rtol = 1e-6;
|
||||
#endif
|
||||
// not owned, location of tensor product basis nodes in reference space
|
||||
const Array<real_t> *points1d = nullptr;
|
||||
|
||||
public:
|
||||
/// Uninitialized BatchInverseElementTransformation. Users must call
|
||||
/// UpdateNodes before Transform.
|
||||
BatchInverseElementTransformation();
|
||||
///
|
||||
/// Constructs a BatchInverseElementTransformation given @a nodes representing
|
||||
/// the mesh nodes.
|
||||
///
|
||||
BatchInverseElementTransformation(const GridFunction &nodes,
|
||||
MemoryType d_mt = MemoryType::DEFAULT);
|
||||
///
|
||||
/// Constructs a BatchInverseElementTransformation for a given @a mesh.
|
||||
/// mesh.GetNodes() must not be null.
|
||||
///
|
||||
BatchInverseElementTransformation(const Mesh &mesh,
|
||||
MemoryType d_mt = MemoryType::DEFAULT);
|
||||
|
||||
~BatchInverseElementTransformation();
|
||||
|
||||
/** @brief Choose how the initial guesses for subsequent calls to Transform()
|
||||
will be selected. ClosestRefNode is currently not supported. */
|
||||
void SetInitialGuessType(InverseElementTransformation::InitGuessType itype)
|
||||
{
|
||||
MFEM_ASSERT(itype != InverseElementTransformation::ClosestRefNode,
|
||||
"ClosestRefNode is currently not supported");
|
||||
init_guess_type = itype;
|
||||
}
|
||||
|
||||
/// Set the Quadrature1D type used for the `Closest*` and `EdgeScan` initial
|
||||
/// guess types.
|
||||
void SetInitGuessPointsType(int q_type) { guess_points_type = q_type; }
|
||||
|
||||
/// Set the relative order used for the `Closest*` initial guess types.
|
||||
/** The number of points in each spatial direction is given by the formula
|
||||
max(trans_order+order,0)+1, where trans_order is the order of the
|
||||
current ElementTransformation. */
|
||||
void SetInitGuessRelOrder(int order)
|
||||
{
|
||||
qpts_order = -1;
|
||||
rel_qpts_order = order;
|
||||
}
|
||||
|
||||
/** The number of points in each spatial direction is given by the formula
|
||||
order+1. */
|
||||
void SetInitGuessOrder(int order) { qpts_order = order; }
|
||||
|
||||
/// @b Gets the basis type nodes are projected onto, or BasisType::Invalid if
|
||||
/// uninitialized.
|
||||
int GetBasisType() const { return basis_type; }
|
||||
|
||||
/** @brief Specify which algorithm to use for solving the transformation
|
||||
equation, i.e. when calling the Transform() method. NewtonSegmentProject
|
||||
is currently not supported. */
|
||||
void SetSolverType(InverseElementTransformation::SolverType stype)
|
||||
{
|
||||
MFEM_ASSERT(stype != InverseElementTransformation::NewtonSegmentProject,
|
||||
"NewtonSegmentProject is currently not supported");
|
||||
solver_type = stype;
|
||||
}
|
||||
|
||||
/// Set the maximum number of iterations when solving for a reference point.
|
||||
void SetMaxIter(int max_it) { max_iter = max_it; }
|
||||
|
||||
/// Set the reference-space convergence tolerance.
|
||||
void SetReferenceTol(real_t ref_sp_tol) { ref_tol = ref_sp_tol; }
|
||||
|
||||
/// Set the relative physical-space convergence tolerance.
|
||||
void SetPhysicalRelTol(real_t phys_rel_tol) { phys_rtol = phys_rel_tol; }
|
||||
|
||||
/**
|
||||
* @brief Updates internal datastructures if @a nodes change. Some version
|
||||
* of UpdateNodes must be called at least once before calls to Transform if
|
||||
* nodes have changed.
|
||||
*/
|
||||
void UpdateNodes(const GridFunction &nodes,
|
||||
MemoryType d_mt = MemoryType::DEFAULT);
|
||||
/**
|
||||
* @brief Updates internal datastructures if @a mesh nodes change. Some version
|
||||
* of UpdateNodes must be called at least once before calls to Transform if
|
||||
* mesh nodes have changed. mesh.GetNodes() must not be null.
|
||||
*/
|
||||
void UpdateNodes(const Mesh &mesh, MemoryType d_mt = MemoryType::DEFAULT);
|
||||
|
||||
/** @brief Performs a batch request of a set of points belonging to the given
|
||||
elements.
|
||||
@a pts list of physical point coordinates ordered by
|
||||
Ordering::Type::byNODES.
|
||||
@a elems which element index to search for each corresponding point in
|
||||
@a pts
|
||||
@a types output search classification (@see
|
||||
InverseElementTransformation::TransformResult).
|
||||
@a refs result reference point coordinates ordered by
|
||||
Ordering::Type::byNODES. If using InitGuessType::GivenPoint, this should
|
||||
contain the initial guess for each point.
|
||||
@a use_device hint for if device acceleration should be used.
|
||||
Device acceleration is currently only implemented for meshes containing
|
||||
only a single tensor product basis element type.
|
||||
@a iters optional array storing how many iterations was spent on each
|
||||
tested point
|
||||
*/
|
||||
void Transform(const Vector &pts, const Array<int> &elems, Array<int> &types,
|
||||
Vector &refs, bool use_device = true,
|
||||
Array<int> *iters = nullptr) const;
|
||||
|
||||
using ClosestPhysPointKernelType = void (*)(int, int, int, int,
|
||||
const real_t *, const real_t *,
|
||||
const int *, const real_t *,
|
||||
const real_t *, real_t *);
|
||||
|
||||
// specialization params: Geom, SDim, use_device
|
||||
MFEM_REGISTER_KERNELS(FindClosestPhysPoint, ClosestPhysPointKernelType,
|
||||
(int, int, bool));
|
||||
|
||||
using ClosestPhysDofKernelType = void (*)(int, int, int,
|
||||
const real_t *, const real_t *,
|
||||
const int *, const real_t *,
|
||||
real_t *);
|
||||
|
||||
// specialization params: Geom, SDim, use_device
|
||||
MFEM_REGISTER_KERNELS(FindClosestPhysDof, ClosestPhysDofKernelType,
|
||||
(int, int, bool));
|
||||
|
||||
using ClosestRefDofKernelType = void (*)(int, int, int, const real_t *,
|
||||
const real_t *, const int *,
|
||||
const real_t *, real_t *);
|
||||
|
||||
// specialization params: Geom, SDim, use_device
|
||||
MFEM_REGISTER_KERNELS(FindClosestRefDof, ClosestRefDofKernelType,
|
||||
(int, int, bool));
|
||||
|
||||
using ClosestRefPointKernelType = void (*)(int, int, int, int, const real_t *,
|
||||
const real_t *, const int *,
|
||||
const real_t *, const real_t *,
|
||||
real_t *);
|
||||
|
||||
// specialization params: Geom, SDim, use_device
|
||||
MFEM_REGISTER_KERNELS(FindClosestRefPoint, ClosestRefPointKernelType,
|
||||
(int, int, bool));
|
||||
|
||||
using NewtonKernelType = void (*)(real_t, real_t, int, int, int, int,
|
||||
const real_t *, const real_t *,
|
||||
const int *, const real_t *, int *, int*,
|
||||
real_t *);
|
||||
|
||||
// specialization params: Geom, SDim, SolverType, use_device
|
||||
MFEM_REGISTER_KERNELS(NewtonSolve, NewtonKernelType,
|
||||
(int, int, InverseElementTransformation::SolverType,
|
||||
bool));
|
||||
|
||||
using NewtonEdgeScanKernelType = void (*)(real_t, real_t, int, int, int, int,
|
||||
const real_t *, const real_t *,
|
||||
const int *, const real_t *,
|
||||
const real_t *, int, int *, int *,
|
||||
real_t *);
|
||||
|
||||
// specialization params: Geom, SDim, SolverType, use_device
|
||||
MFEM_REGISTER_KERNELS(NewtonEdgeScan, NewtonEdgeScanKernelType,
|
||||
(int, int, InverseElementTransformation::SolverType,
|
||||
bool));
|
||||
|
||||
struct Kernels { Kernels(); };
|
||||
|
||||
template <int Dim, int SDim>
|
||||
static void AddFindClosestSpecialization()
|
||||
{
|
||||
FindClosestPhysPoint::Specialization<Dim, SDim, true>::Add();
|
||||
FindClosestRefPoint::Specialization<Dim, SDim, true>::Add();
|
||||
FindClosestPhysPoint::Specialization<Dim, SDim, false>::Add();
|
||||
FindClosestRefPoint::Specialization<Dim, SDim, false>::Add();
|
||||
FindClosestPhysDof::Specialization<Dim, SDim, true>::Add();
|
||||
FindClosestRefDof::Specialization<Dim, SDim, true>::Add();
|
||||
FindClosestPhysDof::Specialization<Dim, SDim, false>::Add();
|
||||
FindClosestRefDof::Specialization<Dim, SDim, false>::Add();
|
||||
}
|
||||
|
||||
template <int Dim, int SDim, InverseElementTransformation::SolverType SType>
|
||||
static void AddNewtonSolveSpecialization()
|
||||
{
|
||||
NewtonSolve::Specialization<Dim, SDim, SType, true>::Add();
|
||||
NewtonEdgeScan::Specialization<Dim, SDim, SType, true>::Add();
|
||||
NewtonSolve::Specialization<Dim, SDim, SType, false>::Add();
|
||||
NewtonEdgeScan::Specialization<Dim, SDim, SType, false>::Add();
|
||||
}
|
||||
};
|
||||
|
||||
/// A standard isoparametric element transformation
|
||||
class IsoparametricTransformation : public ElementTransformation
|
||||
{
|
||||
|
||||
@@ -1,187 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_ELTRANS_BASIS
|
||||
#define MFEM_ELTRANS_BASIS
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
|
||||
#include "../geom.hpp"
|
||||
|
||||
// this file contains utilities for computing nodal basis functions and their
|
||||
// derivatives in device kernels
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace eltrans
|
||||
{
|
||||
|
||||
/// Various utilities for working with different element geometries
|
||||
template <int GeomType> struct GeometryUtils;
|
||||
|
||||
template <> struct GeometryUtils<Geometry::SEGMENT>
|
||||
{
|
||||
static constexpr MFEM_HOST_DEVICE int Dimension() { return 1; }
|
||||
|
||||
/// @b true if the given point x in ref space is inside the element
|
||||
static bool MFEM_HOST_DEVICE inside(real_t x) { return x >= 0 && x <= 1; }
|
||||
|
||||
/// @b Bound the reference coordinate @a x += dx to be inside the segment.
|
||||
/// @a dx is updated to be dx = project(x+dx) - x
|
||||
/// @return true if x + dx hit a boundary
|
||||
static bool MFEM_HOST_DEVICE project(real_t &x, real_t &dx)
|
||||
{
|
||||
real_t tmp = x;
|
||||
x += dx;
|
||||
if (x < 0)
|
||||
{
|
||||
x = 0;
|
||||
dx = x - tmp;
|
||||
return true;
|
||||
}
|
||||
if (x > 1)
|
||||
{
|
||||
x = 1;
|
||||
dx = x - tmp;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
template <> struct GeometryUtils<Geometry::SQUARE>
|
||||
{
|
||||
static constexpr MFEM_HOST_DEVICE int Dimension() { return 2; }
|
||||
/// @b true if the given point (x,y) in ref space is inside the element
|
||||
static bool MFEM_HOST_DEVICE inside(real_t x, real_t y)
|
||||
{
|
||||
return (x >= 0) && (x <= 1) && (y >= 0) && (y <= 1);
|
||||
}
|
||||
|
||||
/// @b Bound the reference coordinate @a (x,y) += (dx,dy) to be inside the
|
||||
/// square.
|
||||
/// @a dx and @a dy are updated to be (dx,dy) = project(x+dx,y+dy) - (x,y)
|
||||
/// @return true if (x,y) + (dx,dy) hit a boundary
|
||||
static bool MFEM_HOST_DEVICE project(real_t &x, real_t &y, real_t &dx,
|
||||
real_t &dy)
|
||||
{
|
||||
bool x_cond = GeometryUtils<Geometry::SEGMENT>::project(x, dx);
|
||||
bool y_cond = GeometryUtils<Geometry::SEGMENT>::project(y, dy);
|
||||
return x_cond || y_cond;
|
||||
}
|
||||
};
|
||||
|
||||
template <> struct GeometryUtils<Geometry::CUBE>
|
||||
{
|
||||
static constexpr MFEM_HOST_DEVICE int Dimension() { return 3; }
|
||||
/// @b true if the given point (x,y,z) in ref space is inside the element
|
||||
static bool MFEM_HOST_DEVICE inside(real_t x, real_t y, real_t z)
|
||||
{
|
||||
return (x >= 0) && (x <= 1) && (y >= 0) && (y <= 1) && (z >= 0) && (z <= 1);
|
||||
}
|
||||
|
||||
/// @b Bound the reference coordinate @a (x,y,z) += (dx,dy,dz) to be inside
|
||||
/// the cube.
|
||||
/// @a dx, @a dy, and @ dz are updated to be
|
||||
/// (dx,dy,dz) = project(x+dx,y+dy,z+dz) - (x,y,z)
|
||||
/// @return true if (x,y,z) + (dx,dy,dz) hit a boundary
|
||||
static bool MFEM_HOST_DEVICE project(real_t &x, real_t &y, real_t &z,
|
||||
real_t &dx, real_t &dy, real_t &dz)
|
||||
{
|
||||
bool x_cond = GeometryUtils<Geometry::SEGMENT>::project(x, dx);
|
||||
bool y_cond = GeometryUtils<Geometry::SEGMENT>::project(y, dy);
|
||||
bool z_cond = GeometryUtils<Geometry::SEGMENT>::project(z, dz);
|
||||
return x_cond || y_cond || z_cond;
|
||||
}
|
||||
};
|
||||
|
||||
/// 1D Lagrange basis from [0, 1]
|
||||
class Lagrange
|
||||
{
|
||||
public:
|
||||
/// interpolant node locations, in reference space
|
||||
const real_t *z;
|
||||
|
||||
/// number of points
|
||||
int pN;
|
||||
|
||||
/// @b Evaluates the @a i'th Lagrange polynomial at @a x
|
||||
real_t MFEM_HOST_DEVICE eval(real_t x, int i) const
|
||||
{
|
||||
real_t u0 = 1;
|
||||
real_t den = 1;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
real_t d_j = (x - z[j]);
|
||||
u0 = d_j * u0;
|
||||
den *= (z[i] - z[j]);
|
||||
}
|
||||
}
|
||||
den = 1 / den;
|
||||
return u0 * den;
|
||||
}
|
||||
|
||||
/// @b Evaluates the @a i'th Lagrange polynomial and its first derivative at
|
||||
/// @a x
|
||||
void MFEM_HOST_DEVICE eval_d1(real_t &p, real_t &d1, real_t x, int i) const
|
||||
{
|
||||
real_t u0 = 1;
|
||||
real_t u1 = 0;
|
||||
real_t den = 1;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
real_t d_j = (x - z[j]);
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
den *= (z[i] - z[j]);
|
||||
}
|
||||
}
|
||||
den = 1 / den;
|
||||
p = u0 * den;
|
||||
d1 = u1 * den;
|
||||
}
|
||||
|
||||
/// @b Evaluates the @a i'th Lagrange polynomial and its first and second
|
||||
/// derivatives at @a x
|
||||
void MFEM_HOST_DEVICE eval_d2(real_t &p, real_t &d1, real_t &d2, real_t x,
|
||||
int i) const
|
||||
{
|
||||
real_t u0 = 1;
|
||||
real_t u1 = 0;
|
||||
real_t u2 = 0;
|
||||
real_t den = 1;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
real_t d_j = (x - z[j]);
|
||||
u2 = d_j * u2 + u1;
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
den *= (z[i] - z[j]);
|
||||
}
|
||||
}
|
||||
den = 1 / den;
|
||||
p = den * u0;
|
||||
d1 = den * u1;
|
||||
d2 = 2 * den * u2;
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace eltrans
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
+2
-2
@@ -623,7 +623,7 @@ private:
|
||||
|
||||
public:
|
||||
/** @brief Construct a new KellyErrorEstimator object for a scalar field.
|
||||
@param di_ The bilinear form to compute the interface flux.
|
||||
@param di_ The bilinearform to compute the interface flux.
|
||||
@param sol_ The solution field whose error is to be estimated.
|
||||
@param flux_fes_ The finite element space for the interface flux.
|
||||
@param attributes_ The attributes of the subdomain(s) for which the
|
||||
@@ -635,7 +635,7 @@ public:
|
||||
const Array<int> &attributes_ = Array<int>());
|
||||
|
||||
/** @brief Construct a new KellyErrorEstimator object for a scalar field.
|
||||
@param di_ The bilinear form to compute the interface flux.
|
||||
@param di_ The bilinearform to compute the interface flux.
|
||||
@param sol_ The solution field whose error is to be estimated.
|
||||
@param flux_fes_ The finite element space for the interface flux.
|
||||
@param attributes_ The attributes of the subdomain(s) for which the
|
||||
|
||||
+17
-41
@@ -69,16 +69,16 @@ inline int ToLexOrdering2D(const int face_id, const int size1d, const int i)
|
||||
}
|
||||
|
||||
/// @brief Given a face DOF index on a shared face, ordered lexicographically
|
||||
/// relative to element the element (where the local face is face_id), and
|
||||
/// return the corresponding face DOF index ordered lexicographically relative
|
||||
/// to the face itself.
|
||||
/// relative to element 1, return the corresponding face DOF index ordered
|
||||
/// lexicographically relative to element 2.
|
||||
MFEM_HOST_DEVICE
|
||||
inline int PermuteFace2D(const int face_id, const int orientation,
|
||||
const int size1d, const int index)
|
||||
inline int PermuteFace2D(const int face_id1, const int face_id2,
|
||||
const int orientation, const int size1d,
|
||||
const int index)
|
||||
{
|
||||
int new_index;
|
||||
// Convert from element 1 lex ordering to native ordering
|
||||
if (face_id == 2 || face_id == 3)
|
||||
if (face_id1 == 2 || face_id1 == 3)
|
||||
{
|
||||
new_index = size1d-1-index;
|
||||
}
|
||||
@@ -91,18 +91,7 @@ inline int PermuteFace2D(const int face_id, const int orientation,
|
||||
{
|
||||
new_index = size1d-1-new_index;
|
||||
}
|
||||
return new_index;
|
||||
}
|
||||
|
||||
/// @brief Given a face DOF index on a shared face, ordered lexicographically
|
||||
/// relative to element 1, return the corresponding face DOF index ordered
|
||||
/// lexicographically relative to element 2.
|
||||
MFEM_HOST_DEVICE
|
||||
inline int PermuteFace2D(const int face_id1, const int face_id2,
|
||||
const int orientation, const int size1d,
|
||||
const int index)
|
||||
{
|
||||
const int new_index = PermuteFace2D(face_id1, orientation, size1d, index);
|
||||
// Covert to element 2 lex ordering
|
||||
return ToLexOrdering2D(face_id2, size1d, new_index);
|
||||
}
|
||||
|
||||
@@ -127,22 +116,26 @@ inline int ToLexOrdering3D(const int face_id, const int size1d, const int i,
|
||||
}
|
||||
}
|
||||
|
||||
/// @brief Given the index of a face DOF in lexicographic ordering relative the
|
||||
/// element (where the local face id is @a face_id), permute the index so that
|
||||
/// it is lexicographically ordered relative to the face itself.
|
||||
/// @brief Given the index of a face DOF in lexicographic ordering relative
|
||||
/// element 1, permute the index so that it is lexicographically ordered
|
||||
/// relative to element 2.
|
||||
///
|
||||
/// The given face corresponds to local face index @a face_id1 relative to
|
||||
/// element 1, and @a face_id2 (with @a orientation) relative to element 2.
|
||||
MFEM_HOST_DEVICE
|
||||
inline int PermuteFace3D(const int face_id, const int orientation,
|
||||
inline int PermuteFace3D(const int face_id1, const int face_id2,
|
||||
const int orientation,
|
||||
const int size1d, const int index)
|
||||
{
|
||||
int i=0, j=0, new_i=0, new_j=0;
|
||||
i = index%size1d;
|
||||
j = index/size1d;
|
||||
// Convert from lex ordering
|
||||
if (face_id==3 || face_id==4)
|
||||
if (face_id1==3 || face_id1==4)
|
||||
{
|
||||
i = size1d-1-i;
|
||||
}
|
||||
else if (face_id==0)
|
||||
else if (face_id1==0)
|
||||
{
|
||||
j = size1d-1-j;
|
||||
}
|
||||
@@ -182,23 +175,6 @@ inline int PermuteFace3D(const int face_id, const int orientation,
|
||||
new_j = (size1d-1-j);
|
||||
break;
|
||||
}
|
||||
return new_i + new_j*size1d;
|
||||
}
|
||||
|
||||
/// @brief Given the index of a face DOF in lexicographic ordering relative
|
||||
/// element 1, permute the index so that it is lexicographically ordered
|
||||
/// relative to element 2.
|
||||
///
|
||||
/// The given face corresponds to local face index @a face_id1 relative to
|
||||
/// element 1, and @a face_id2 (with @a orientation) relative to element 2.
|
||||
MFEM_HOST_DEVICE
|
||||
inline int PermuteFace3D(const int face_id1, const int face_id2,
|
||||
const int orientation,
|
||||
const int size1d, const int index)
|
||||
{
|
||||
const int new_index = PermuteFace3D(face_id1, orientation, size1d, index);
|
||||
const int new_i = new_index%size1d;
|
||||
const int new_j = new_index/size1d;
|
||||
return ToLexOrdering3D(face_id2, size1d, new_i, new_j);
|
||||
}
|
||||
|
||||
|
||||
+62
-95
@@ -973,22 +973,6 @@ void NodalFiniteElement::ProjectDiv(
|
||||
}
|
||||
}
|
||||
|
||||
void NodalFiniteElement::ReorderLexToNative(int ncomp,
|
||||
Vector &dofs) const
|
||||
{
|
||||
MFEM_ASSERT(lex_ordering.Size() == dof, "Permutation is not defined by FE.");
|
||||
MFEM_ASSERT(dofs.Size() == ncomp * dof, "Wrong input size.");
|
||||
|
||||
Vector dofs_native(ncomp * dof);
|
||||
for (int i = 0; i < dof; i++)
|
||||
{
|
||||
for (int c = 0; c < ncomp; c++)
|
||||
{
|
||||
dofs_native(c*dof + lex_ordering[i]) = dofs(c*dof + i);
|
||||
}
|
||||
}
|
||||
dofs = dofs_native;
|
||||
}
|
||||
|
||||
VectorFiniteElement::VectorFiniteElement(int D, Geometry::Type G,
|
||||
int Do, int O, int M, int F)
|
||||
@@ -2184,64 +2168,6 @@ void Poly_1D::CalcDBinomTerms(const int p, const real_t x, const real_t y,
|
||||
}
|
||||
}
|
||||
|
||||
void Poly_1D::CalcDxBinomTerms(const int p, const real_t x, const real_t y,
|
||||
real_t *u)
|
||||
{
|
||||
if (p == 0)
|
||||
{
|
||||
u[0] = 0.;
|
||||
}
|
||||
else
|
||||
{
|
||||
int i;
|
||||
const int *b = Binom(p);
|
||||
real_t z = 1.;
|
||||
|
||||
for (i = 1; i < p; i++)
|
||||
{
|
||||
u[i] = i * b[i]*z;
|
||||
z *= x;
|
||||
}
|
||||
u[p] = i * z;
|
||||
z = y;
|
||||
for (i--; i > 0; i--)
|
||||
{
|
||||
u[i] *= z;
|
||||
z *= y;
|
||||
}
|
||||
u[0] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
void Poly_1D::CalcDyBinomTerms(const int p, const real_t x, const real_t y,
|
||||
real_t *u)
|
||||
{
|
||||
if (p == 0)
|
||||
{
|
||||
u[0] = 0.;
|
||||
}
|
||||
else
|
||||
{
|
||||
int i;
|
||||
const int *b = Binom(p);
|
||||
real_t z = x;
|
||||
|
||||
for (i = 1; i < p; i++)
|
||||
{
|
||||
u[i] = b[i]*z;
|
||||
z *= x;
|
||||
}
|
||||
u[p] = 0.;
|
||||
z = 1.;
|
||||
for (i--; i > 0; i--)
|
||||
{
|
||||
u[i] *= (p - i) * z;
|
||||
z *= y;
|
||||
}
|
||||
u[0] = p * z;
|
||||
}
|
||||
}
|
||||
|
||||
void Poly_1D::CalcLegendre(const int p, const real_t x, real_t *u)
|
||||
{
|
||||
// use the recursive definition for [-1,1]:
|
||||
@@ -2341,58 +2267,99 @@ void Poly_1D::CalcChebyshev(const int p, const real_t x, real_t *u, real_t *d,
|
||||
}
|
||||
}
|
||||
|
||||
const Array<real_t>* Poly_1D::GetPointsArray(const int p, const int btype)
|
||||
const real_t *Poly_1D::GetPoints(const int p, const int btype)
|
||||
{
|
||||
Array<real_t> *val;
|
||||
Array<real_t*> *pts;
|
||||
BasisType::Check(btype);
|
||||
const int qtype = BasisType::GetQuadrature1D(btype);
|
||||
if (qtype == Quadrature1D::Invalid) { return nullptr; }
|
||||
if (qtype == Quadrature1D::Invalid) { return NULL; }
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
#pragma omp critical (Poly1DGetPoints)
|
||||
#endif
|
||||
{
|
||||
std::pair<int, int> key(btype, p);
|
||||
auto it = points_container.find(key);
|
||||
if (it == points_container.end())
|
||||
auto it = points_container.find(btype);
|
||||
if (it != points_container.end())
|
||||
{
|
||||
it = points_container.emplace(key, new Array<real_t>(p + 1, h_mt)).first;
|
||||
val = it->second.get();
|
||||
real_t* hptr = val->HostWrite();
|
||||
quad_func.GivePolyPoints(p + 1, hptr, qtype);
|
||||
pts = it->second;
|
||||
}
|
||||
else
|
||||
{
|
||||
val = it->second.get();
|
||||
pts = new Array<real_t*>(h_mt);
|
||||
points_container[btype] = pts;
|
||||
}
|
||||
if (pts->Size() <= p)
|
||||
{
|
||||
pts->SetSize(p + 1, NULL);
|
||||
}
|
||||
if ((*pts)[p] == NULL)
|
||||
{
|
||||
(*pts)[p] = new real_t[p + 1];
|
||||
quad_func.GivePolyPoints(p + 1, (*pts)[p], qtype);
|
||||
}
|
||||
}
|
||||
return val;
|
||||
return (*pts)[p];
|
||||
}
|
||||
|
||||
Poly_1D::Basis &Poly_1D::GetBasis(const int p, const int btype)
|
||||
{
|
||||
Array<Basis*> *bases;
|
||||
BasisType::Check(btype);
|
||||
Basis* val;
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
#pragma omp critical (Poly1DGetBasis)
|
||||
#endif
|
||||
{
|
||||
std::pair<int, int> key(btype, p);
|
||||
auto it = bases_container.find(key);
|
||||
if (it == bases_container.end())
|
||||
auto it = bases_container.find(btype);
|
||||
if (it != bases_container.end())
|
||||
{
|
||||
bases = it->second;
|
||||
}
|
||||
else
|
||||
{
|
||||
// we haven't been asked for basis or points of this type yet
|
||||
bases = new Array<Basis*>(h_mt);
|
||||
bases_container[btype] = bases;
|
||||
}
|
||||
if (bases->Size() <= p)
|
||||
{
|
||||
bases->SetSize(p + 1, NULL);
|
||||
}
|
||||
if ((*bases)[p] == NULL)
|
||||
{
|
||||
EvalType etype;
|
||||
if (btype == BasisType::Positive) { etype = Positive; }
|
||||
else if (btype == BasisType::IntegratedGLL) { etype = Integrated; }
|
||||
else { etype = Barycentric; }
|
||||
it = bases_container
|
||||
.emplace(key, new Basis(p, GetPoints(p, btype), etype))
|
||||
.first;
|
||||
(*bases)[p] = new Basis(p, GetPoints(p, btype), etype);
|
||||
}
|
||||
val = it->second.get();
|
||||
}
|
||||
return *val;
|
||||
return *(*bases)[p];
|
||||
}
|
||||
|
||||
Poly_1D::~Poly_1D()
|
||||
{
|
||||
for (PointsMap::iterator it = points_container.begin();
|
||||
it != points_container.end() ; ++it)
|
||||
{
|
||||
Array<real_t*>& pts = *it->second;
|
||||
for (int i = 0; i < pts.Size(); ++i)
|
||||
{
|
||||
delete [] pts[i];
|
||||
}
|
||||
delete it->second;
|
||||
}
|
||||
|
||||
for (BasisMap::iterator it = bases_container.begin();
|
||||
it != bases_container.end() ; ++it)
|
||||
{
|
||||
Array<Basis*>& bases = *it->second;
|
||||
for (int i = 0; i < bases.Size(); ++i)
|
||||
{
|
||||
delete bases[i];
|
||||
}
|
||||
delete it->second;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user