Compare commits
233
Commits
multiapp-rm-vec
...
fixes
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c11b2df36f | ||
|
|
1299e13e65 | ||
|
|
907a629f82 | ||
|
|
e032c15aef | ||
|
|
10ceb3e66b | ||
|
|
efa30a4a62 | ||
|
|
3ef9a5c668 | ||
|
|
7b85e1e9c1 | ||
|
|
775195b887 | ||
|
|
89adf27a44 | ||
|
|
a7dbea190f | ||
|
|
12e9b66eae | ||
|
|
713edd670d | ||
|
|
e2d6f5fb3b | ||
|
|
45b0e6e02c | ||
|
|
d37b7867ec | ||
|
|
73779b1de6 | ||
|
|
8307a751db | ||
|
|
9f12aee475 | ||
|
|
366157036e | ||
|
|
8afc1d1e36 | ||
|
|
2bc734468d | ||
|
|
c07c534f42 | ||
|
|
ccade73917 | ||
|
|
d169312edd | ||
|
|
8812081cfc | ||
|
|
b20051c06b | ||
|
|
d66068b754 | ||
|
|
362ca5b66d | ||
|
|
f9282b38f6 | ||
|
|
aab2e1ebf8 | ||
|
|
8c2a8580b6 | ||
|
|
9141e85e15 | ||
|
|
e57b63c660 | ||
|
|
5d1958cfdf | ||
|
|
614a355c04 | ||
|
|
79a88dfef5 | ||
|
|
bd13f53db1 | ||
|
|
eb738baebe | ||
|
|
46c5aed37b | ||
|
|
6b8f53308f | ||
|
|
1369d61457 | ||
|
|
e7d6b370dc | ||
|
|
ba07e91128 | ||
|
|
ebdf68a1c3 | ||
|
|
5f31928c2b | ||
|
|
0cf5aca53e | ||
|
|
2357771384 | ||
|
|
7efeb617b1 | ||
|
|
fe025de316 | ||
|
|
76b5f341cc | ||
|
|
ea8468ea95 | ||
|
|
2ea59935d8 | ||
|
|
4e5ebe6451 | ||
|
|
04f23f353c | ||
|
|
0a76b8bfb2 | ||
|
|
6116b49933 | ||
|
|
09f6023468 | ||
|
|
610a8f9c0b | ||
|
|
8f01292a45 | ||
|
|
f7056be951 | ||
|
|
790848019e | ||
|
|
01b146ab01 | ||
|
|
ae87b89f16 | ||
|
|
28e0f3569a | ||
|
|
3c9ee8ff42 | ||
|
|
f49b9a58e8 | ||
|
|
8bfac662f4 | ||
|
|
905696021a | ||
|
|
dc6e1ff4ea | ||
|
|
ed563f3090 | ||
|
|
51f2f5dd78 | ||
|
|
4e828b9240 | ||
|
|
d627b19f06 | ||
|
|
9b35464986 | ||
|
|
f14a9bb53f | ||
|
|
4eafaaa628 | ||
|
|
bcab63b41c | ||
|
|
28c1f905b6 | ||
|
|
66dbe60cb1 | ||
|
|
f898d0bcde | ||
|
|
b8fcd640e5 | ||
|
|
c566a165b2 | ||
|
|
733d0bd177 | ||
|
|
6e2bd88274 | ||
|
|
7a0a7bd1da | ||
|
|
e59487bf14 | ||
|
|
647750ffa9 | ||
|
|
db42eb3255 | ||
|
|
1c19aba72a | ||
|
|
1631ec67fa | ||
|
|
eceb502df3 | ||
|
|
bfdaf07a19 | ||
|
|
d4b59fe357 | ||
|
|
cdf077b560 | ||
|
|
50ce940dee | ||
|
|
d3307a6957 | ||
|
|
bde4cbbccd | ||
|
|
c8b64fef23 | ||
|
|
4c16395398 | ||
|
|
cfb05a4a60 | ||
|
|
e9cce62beb | ||
|
|
7bb2d30100 | ||
|
|
51a0058f65 | ||
|
|
fd223f68b5 | ||
|
|
eac57686c5 | ||
|
|
ad962de425 | ||
|
|
c44c2f0cdf | ||
|
|
25a1c8f4a4 | ||
|
|
a60ba38833 | ||
|
|
2fa81463ae | ||
|
|
0909dc634a | ||
|
|
dc995c4aa0 | ||
|
|
b9c960cc0d | ||
|
|
0aa392a4ea | ||
|
|
ab6d0d9777 | ||
|
|
3ce8b9e250 | ||
|
|
ffa3d0789b | ||
|
|
eaf91c9c08 | ||
|
|
8330565463 | ||
|
|
6a6e9b5d6b | ||
|
|
609a9c0e3b | ||
|
|
340fe85001 | ||
|
|
4c6291f018 | ||
|
|
c6378788af | ||
|
|
d6fffff08c | ||
|
|
d65409fdc0 | ||
|
|
8b14357249 | ||
|
|
cc1c6daed3 | ||
|
|
28b6c85b44 | ||
|
|
0a43f3ca1f | ||
|
|
212edacfd1 | ||
|
|
45cd0db146 | ||
|
|
f0505ec6eb | ||
|
|
d3fda1ed30 | ||
|
|
f63e95a7a1 | ||
|
|
24e63e6802 | ||
|
|
e8961b32ff | ||
|
|
9c4fa75530 | ||
|
|
b904dd0131 | ||
|
|
24652e2a36 | ||
|
|
529209bcf2 | ||
|
|
501e37d105 | ||
|
|
d48384f9f4 | ||
|
|
94a815d9c9 | ||
|
|
1a03792398 | ||
|
|
118e97772c | ||
|
|
f1138eae7a | ||
|
|
449199525b | ||
|
|
17ecabf915 | ||
|
|
de1a876e39 | ||
|
|
538711c13f | ||
|
|
412cc42685 | ||
|
|
c8b1dcad70 | ||
|
|
fa006da71e | ||
|
|
1e5f9e4d6b | ||
|
|
b0cc0a9b8c | ||
|
|
e839a5e8ab | ||
|
|
9d40c8b40c | ||
|
|
069c618def | ||
|
|
e6d5e98a06 | ||
|
|
cfa3440178 | ||
|
|
ed9a29130f | ||
|
|
9a80c8cd14 | ||
|
|
143d7bf31b | ||
|
|
8358ee93fa | ||
|
|
eb38d6ecd8 | ||
|
|
5f80fb1eb7 | ||
|
|
52efc31130 | ||
|
|
b7dc53af15 | ||
|
|
2b5c0c6fe4 | ||
|
|
7b8af2b05f | ||
|
|
1433d4aec4 | ||
|
|
74d1579371 | ||
|
|
ea83267885 | ||
|
|
49201d41c3 | ||
|
|
a53c446dd7 | ||
|
|
3c8c8c21a9 | ||
|
|
50d58159bd | ||
|
|
bab4314cf3 | ||
|
|
e59d1835c3 | ||
|
|
fbb0e44dce | ||
|
|
3cdaebdcaa | ||
|
|
9e8a7c456f | ||
|
|
b39719984a | ||
|
|
a95278fe72 | ||
|
|
f2f366efa2 | ||
|
|
cc585df285 | ||
|
|
c7f2950458 | ||
|
|
068b61eb3f | ||
|
|
3419a50655 | ||
|
|
f4ad8b8f92 | ||
|
|
e04c90b678 | ||
|
|
abbfe7cf71 | ||
|
|
7d91917d7a | ||
|
|
6c2a78d5bd | ||
|
|
82f03e136d | ||
|
|
46a84f6417 | ||
|
|
f6b333681f | ||
|
|
644b4ef141 | ||
|
|
08d6dd777a | ||
|
|
879413e774 | ||
|
|
63627acf30 | ||
|
|
ce80de49d0 | ||
|
|
45a62e8bcd | ||
|
|
4ee2e40d34 | ||
|
|
86af0f883c | ||
|
|
c529d34eea | ||
|
|
742d043ead | ||
|
|
a67c93d0b8 | ||
|
|
43532923f7 | ||
|
|
9980f767f8 | ||
|
|
3b89be0ec6 | ||
|
|
4c12e3815b | ||
|
|
44d2d0c75b | ||
|
|
dbb5fe2f0e | ||
|
|
620e49aea6 | ||
|
|
94da954917 | ||
|
|
006855bec2 | ||
|
|
96f9456a7d | ||
|
|
6ce18b2005 | ||
|
|
c26f1937a9 | ||
|
|
8431604228 | ||
|
|
c09b6d8a1d | ||
|
|
19d9175833 | ||
|
|
e01d5afadb | ||
|
|
24bc9d48a1 | ||
|
|
27deb9cdd2 | ||
|
|
a7b30bed56 | ||
|
|
fd63847904 | ||
|
|
5269fc2bf2 | ||
|
|
4036a7d0c2 | ||
|
|
8099ca947e |
@@ -260,6 +260,7 @@ miniapps/meshing/polar-nc
|
||||
miniapps/meshing/mesh-quality
|
||||
miniapps/meshing/hpref
|
||||
miniapps/meshing/phpref
|
||||
miniapps/meshing/pref321
|
||||
miniapps/meshing/mobius-strip.mesh
|
||||
miniapps/meshing/klein-bottle.mesh
|
||||
miniapps/meshing/toroid-*.mesh
|
||||
|
||||
@@ -46,8 +46,19 @@ Discretization improvements
|
||||
|
||||
- Extend FindPointsGSLIB to support surface meshes.
|
||||
|
||||
- Added support for complex-valued mixed bilinear forms via the new classes
|
||||
MixedSesquilinearForm and ParMixedSesquilinearForm, mirroring the existing
|
||||
SesquilinearForm classes. Rectangular complex operators are now also
|
||||
handled correctly by ComplexSparseMatrix::GetSystemMatrix and
|
||||
ComplexHypreParMatrix::GetSystemMatrix, which previously assumed equal
|
||||
trial and test spaces.
|
||||
|
||||
Meshing improvements
|
||||
--------------------
|
||||
- Added support for nonuniform anisotropic mesh refinement on parallel quad/hex
|
||||
meshes with arbitrary spacing in each direction. This enables in particular
|
||||
3:1 refinement in parallel, as demonstrated in the new meshing miniapp pref321.
|
||||
|
||||
- Added option to guarantee mesh validity during TMOP-based r-adaptivity, using
|
||||
bounds on the determinant of the mesh transformation Jacobian.
|
||||
|
||||
@@ -68,6 +79,11 @@ Linear and nonlinear solvers
|
||||
PRefinement multigrid methods for problems posed on trace spaces (see e.g. the
|
||||
DPG miniapps).
|
||||
|
||||
- Added new class MultiVector: an array of Vectors of different sizes where each
|
||||
Vector can be allocated independently. Also, added associated methods in class
|
||||
Operator: MultMV, MultTransposeMV, and GetGradientMV, that use MultiVector
|
||||
objects for input and/or output parameters. [PR #5249]
|
||||
|
||||
GPU computing
|
||||
-------------
|
||||
- Improved partial assembly for VectorDivergenceIntegrator with shared-memory
|
||||
@@ -81,15 +97,35 @@ GPU computing
|
||||
|
||||
- Added device assembly support for 3D H(curl) VectorFEDomainLFIntegrator.
|
||||
|
||||
- Added partial assembly support for MixedScalarWeakGradientIntegrator.
|
||||
|
||||
- Added partial assembly support for MixedDotProductIntegrator.
|
||||
|
||||
- Added partial assembly support for MixedScalarCrossProductIntegrator.
|
||||
|
||||
- Added partial assembly support for MixedScalarWeakCrossProductIntegrator.
|
||||
|
||||
- Added support for device partial assembly CurlInterpolator.
|
||||
This supports 2D and 3D variants:
|
||||
2D H1 (out-of-plane) to RT (in-plane)
|
||||
2D ND (in-plane) to Integral L2 (out-of-plane)
|
||||
3D ND to RT
|
||||
|
||||
- Added NVIDIA cuDSS library interface. Implementation examples have been
|
||||
added to ex1 and ex1p. See https://developer.nvidia.com/cudss for more
|
||||
details. Supported versions >= 0.6.0.
|
||||
|
||||
- Allow specifying GPU kernel launch bounds for native and RAJA GPU backends.
|
||||
|
||||
- Changed VectorFEMassIntegrator to use kernel specialization dispatch for
|
||||
partial assembly.
|
||||
|
||||
- Added support for FiniteElement::MapType::INTEGRAL spaces to
|
||||
QuadratureInterpolator.
|
||||
|
||||
- Added support for FiniteElement::MapType::INTEGRAL spaces to
|
||||
MixedScalarCurlIntegrator.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- The Lorentz miniapp (in miniapps/electromagnetics) has been updated to
|
||||
@@ -104,6 +140,12 @@ Miscellaneous
|
||||
using the new method ApplyDofSigns() in class ParFiniteElementSpace: the
|
||||
method will return immediately if no sign flips are needed.
|
||||
|
||||
API changes
|
||||
-----------
|
||||
- Removed ProjectGrad from 2D RT elements. Users should use ProjectCurl instead.
|
||||
This also fixes a bug where ProjectCurl was returning the negative curl,
|
||||
identical to ProjectGrad.
|
||||
|
||||
|
||||
Version 4.9, released on Dec 11, 2025
|
||||
=====================================
|
||||
|
||||
+3
-12
@@ -88,18 +88,9 @@ if (MFEM_USE_STRUMPACK OR MFEM_USE_MUMPS)
|
||||
# Just needed to find the MPI_Fortran libraries to link with
|
||||
set(XSDK_ENABLE_Fortran ON)
|
||||
endif()
|
||||
# Ginkgo requires C++17:
|
||||
if ((MFEM_USE_GINKGO) AND ("${CMAKE_CXX_STANDARD}" LESS "17"))
|
||||
set(CMAKE_CXX_STANDARD 17 CACHE STRING "C++ standard to use." FORCE)
|
||||
# Google Benchmark, SUNDIALS, STRUMPACK, Tribol, RAJA and Umpire require C++14:
|
||||
elseif ((MFEM_USE_BENCHMARK OR
|
||||
MFEM_USE_SUNDIALS OR
|
||||
MFEM_USE_STRUMPACK OR
|
||||
MFEM_USE_TRIBOL OR
|
||||
MFEM_USE_RAJA OR
|
||||
MFEM_USE_UMPIRE) AND
|
||||
("${CMAKE_CXX_STANDARD}" LESS "14"))
|
||||
set(CMAKE_CXX_STANDARD 14 CACHE STRING "C++ standard to use." FORCE)
|
||||
# RAJA requires C++20:
|
||||
if ((MFEM_USE_UMPIRE OR MFEM_USE_RAJA) AND ("${CMAKE_CXX_STANDARD}" LESS "20"))
|
||||
set(CMAKE_CXX_STANDARD 20 CACHE STRING "C++ standard to use." FORCE)
|
||||
endif()
|
||||
|
||||
# Include xSDK default CMake file.
|
||||
|
||||
+33
-8
@@ -28,11 +28,8 @@ MPICXX = mpicxx
|
||||
BASE_FLAGS = -std=c++17
|
||||
OPTIM_FLAGS = -O3 $(BASE_FLAGS)
|
||||
|
||||
# Shadow warnings for clang only; GCC's -Wshadow flags more.
|
||||
SHADOW_WARNING_FLAG = $(if $(findstring clang,\
|
||||
$(shell $(MFEM_HOST_CXX) --version 2>/dev/null)),-Wshadow,)
|
||||
WARNING_FLAGS = -pedantic -Wall $(SHADOW_WARNING_FLAG)
|
||||
|
||||
# The variable WARNING_FLAGS depends on which compiler is used, and is defined
|
||||
# later in this file.
|
||||
DEBUG_FLAGS = $(strip -g $(addprefix $(XCOMPILER),$(WARNING_FLAGS)) $(BASE_FLAGS))
|
||||
|
||||
# Prefixes for passing flags to the compiler and linker when using CXX or MPICXX
|
||||
@@ -52,6 +49,10 @@ SHARED = NO
|
||||
#
|
||||
# If you set MFEM_USE_ENZYME=YES, must use CUDA_CXX=clang++
|
||||
CUDA_CXX = nvcc
|
||||
# CUDA compute capability used during compilation, e.g. sm_60. Multiple
|
||||
# architectures can be requested as a comma-separated list, e.g. sm_70,sm_80.
|
||||
# A single value may also be one of the nvcc special values "all",
|
||||
# "all-major", or "native".
|
||||
CUDA_ARCH = sm_60
|
||||
# Base CUDA install directory, only needed if building with clang+cuda:
|
||||
# The default setting is:
|
||||
@@ -60,11 +61,23 @@ CUDA_ARCH = sm_60
|
||||
# 3. Use /usr/local/cuda
|
||||
CUDA_DIR = $(or $(CUDA_HOME),$(patsubst %/,%,$(dir \
|
||||
$(patsubst %/,%,$(dir $(shell command -v nvcc))))),/usr/local/cuda)
|
||||
# Derive nvcc/clang architecture flags from CUDA_ARCH. A comma-separated list
|
||||
# expands into one -gencode / --cuda-gpu-arch flag per architecture; otherwise
|
||||
# use the -arch / --cuda-gpu-arch shorthand.
|
||||
MFEM_COMMA := ,
|
||||
CUDA_ARCH_NUMS = $(patsubst sm_%,%,$(subst $(MFEM_COMMA), ,$(CUDA_ARCH)))
|
||||
NVCC_ARCH_FLAGS = $(strip $(if $(findstring $(MFEM_COMMA),$(CUDA_ARCH)),\
|
||||
$(foreach arch,$(CUDA_ARCH_NUMS),\
|
||||
-gencode arch=compute_$(arch)$(MFEM_COMMA)code=sm_$(arch)),\
|
||||
-arch=$(CUDA_ARCH)))
|
||||
CLANG_ARCH_FLAGS = $(strip $(if $(findstring $(MFEM_COMMA),$(CUDA_ARCH)),\
|
||||
$(foreach arch,$(CUDA_ARCH_NUMS),--cuda-gpu-arch=sm_$(arch)),\
|
||||
--cuda-gpu-arch=$(CUDA_ARCH)))
|
||||
# flags for clang+cuda
|
||||
CLANG_CUDA_FLAGS = -xcuda --cuda-path=$(CUDA_DIR) --cuda-gpu-arch=$(CUDA_ARCH)
|
||||
CLANG_CUDA_FLAGS = -xcuda --cuda-path=$(CUDA_DIR) $(CLANG_ARCH_FLAGS)
|
||||
# flags for nvcc
|
||||
NVCC_FLAGS = -x=cu --expt-extended-lambda --expt-relaxed-constexpr \
|
||||
-arch=$(CUDA_ARCH) -isystem "$(CUDA_DIR)/include"
|
||||
$(NVCC_ARCH_FLAGS) -isystem "$(CUDA_DIR)/include"
|
||||
# Prefixes for passing flags to the host compiler and linker when using
|
||||
# CUDA_CXX=nvcc
|
||||
CUDA_XCOMPILER = -Xcompiler=
|
||||
@@ -382,7 +395,7 @@ CUDSS_LIBRARY_DIR = $(CUDSS_DIR)/lib
|
||||
CUDSS_OPT = -I$(CUDSS_INCLUDE_DIR)
|
||||
CUDSS_LIB = \
|
||||
$(XLINKER)-rpath,$(CUDSS_LIBRARY_DIR) -L$(CUDSS_LIBRARY_DIR) -lcudss
|
||||
# The cuDSS communication and threading libraries.
|
||||
# The cuDSS communication and threading libraries.
|
||||
MFEM_CUDSS_COMM_LIB = $(abspath $(wildcard $(or $(CUDSS_COMM_LIB),\
|
||||
$(subst @MFEM_DIR@,$(MFEM_DIR), $(CUDSS_LIBRARY_DIR)/libcudss_commlayer_openmpi.so))))
|
||||
MFEM_CUDSS_THREADING_LIB = $(abspath $(wildcard $(or $(CUDSS_THREADING_LIB),\
|
||||
@@ -665,3 +678,15 @@ VERBOSE = NO
|
||||
|
||||
# Optional build tag
|
||||
MFEM_BUILD_TAG = $(shell uname -snm)
|
||||
|
||||
# Enable -pedantic flag only for gcc or clang. nvcc complains with -pedantic
|
||||
# because of line directives.
|
||||
PEDANTIC_FLAG = $(if \
|
||||
$(findstring NVIDIA,$(shell $(MFEM_CXX) --version 2>&1)),, \
|
||||
$(if $(or \
|
||||
$(findstring gcc version,$(shell $(MFEM_CXX) -v 2>&1)), \
|
||||
$(findstring clang version,$(shell $(MFEM_CXX) -v 2>&1))),-pedantic,))
|
||||
# Enable shadow warnings for clang only; GCC's -Wshadow flags more.
|
||||
SHADOW_WARNING_FLAG = $(if $(findstring clang,\
|
||||
$(shell $(MFEM_HOST_CXX) --version 2>/dev/null)),-Wshadow,)
|
||||
WARNING_FLAGS = $(PEDANTIC_FLAG) -Wall $(SHADOW_WARNING_FLAG)
|
||||
|
||||
@@ -1083,7 +1083,8 @@ EXCLUDE_PATTERNS =
|
||||
# ANamespace::AClass, ANamespace::*Test
|
||||
|
||||
EXCLUDE_SYMBOLS = mfem::internal \
|
||||
mfem::kernels::internal
|
||||
mfem::kernels::internal \
|
||||
mfem::future::detail
|
||||
|
||||
# The EXAMPLE_PATH tag can be used to specify one or more files or directories
|
||||
# that contain example code fragments that are included (see the \include
|
||||
|
||||
@@ -201,6 +201,7 @@ namespace mfem {
|
||||
* - <a class="el" href="nurbs__naca__cmesh_8cpp_source.html">NURBS NACA Mesher</a>: generate NURBS based mesh around a NACA foil
|
||||
* - <a class="el" href="nurbs__printfunc_8cpp_source.html">NURBS Printer</a>: print the NURBS-basis
|
||||
* - <a class="el" href="nurbs__mesh_info_8cpp_source.html">NURBS Mesh info</a>: print the info of a NURBS mesh
|
||||
* - <a class="el" href="nurbs__surface_8cpp_source.html">NURBS Surface</a>: interpolate a 3D Surface in a NURBS Patch
|
||||
*
|
||||
* <H3>Miniapps</H3>
|
||||
* - <a class="el" href="volta_8cpp_source.html">Volta</a>: simple electrostatics simulation code
|
||||
@@ -245,6 +246,9 @@ namespace mfem {
|
||||
* - <a class="el" href="pdiffusion_8cpp_source.html">DPG Diffusion example</a>: DPG formulation for the diffusion problem
|
||||
* - <a class="el" href="pmaxwell_8cpp_source.html">DPG Maxwell example</a>: DPG formulation for the indefinite Maxwell problem
|
||||
* - <a class="el" href="lor__elast_8cpp_source.html">LOR Elasticity</a>: solve linear elasticity with LOR preconditioning on GPUs
|
||||
* - <a class="el" href="reflector_8cpp_source.html">Reflector Miniapp</a>: reflect a mesh about a plane
|
||||
* - <a class="el" href="ref321_8cpp_source.html">3:1 Refinement Miniapp</a>: perform 3:1 anisotropic mesh refinements
|
||||
* - <a class="el" href="pref321_8cpp_source.html">3:1 Refinement Miniapp</a>: parallel 3:1 anisotropic mesh refinements
|
||||
*
|
||||
* See also the <a class="el" href="https://mfem.org/examples/">examples documentation</a> online.
|
||||
*/
|
||||
|
||||
@@ -1255,6 +1255,31 @@ void BilinearForm::Mult(const Vector &x, Vector &y) const
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::AddMult(const Vector &x, Vector &y, const real_t a) const
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
ext->AddMult(x, y, a);
|
||||
}
|
||||
else
|
||||
{
|
||||
mat->AddMult(x, y, a);
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::AddMultTranspose(const Vector &x, Vector &y,
|
||||
const real_t a) const
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
ext->AddMultTranspose(x, y, a);
|
||||
}
|
||||
else
|
||||
{
|
||||
mat->AddMultTranspose(x, y, a);
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::MultTranspose(const Vector & x, Vector & y) const
|
||||
{
|
||||
if (ext)
|
||||
|
||||
@@ -307,8 +307,8 @@ public:
|
||||
{ mat->Mult(x, y); mat_e->AddMult(x, y); }
|
||||
|
||||
/// Add the matrix vector multiple to a vector: $ y += a M x $
|
||||
void AddMult(const Vector &x, Vector &y, const real_t a = 1.0) const override
|
||||
{ mat -> AddMult (x, y, a); }
|
||||
void AddMult(const Vector &x, Vector &y,
|
||||
const real_t a = 1.0) const override;
|
||||
|
||||
/** @brief Add the original uneliminated matrix vector multiple to a vector.
|
||||
The original matrix is $ M + Me $ so we have:
|
||||
@@ -318,8 +318,7 @@ public:
|
||||
|
||||
/// Add the matrix transpose vector multiplication: $ y += a M^T x $
|
||||
void AddMultTranspose(const Vector & x, Vector & y,
|
||||
const real_t a = 1.0) const override
|
||||
{ mat->AddMultTranspose(x, y, a); }
|
||||
const real_t a = 1.0) const override;
|
||||
|
||||
/** @brief Add the original uneliminated matrix transpose vector
|
||||
multiple to a vector. The original matrix is $ M + M_e $
|
||||
|
||||
@@ -1997,7 +1997,11 @@ void PADiscreteLinearOperatorExtension::Assemble()
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem_error("A real ElementRestriction is required in this setting!");
|
||||
const L2ElementRestriction* l2_elem_restrict =
|
||||
dynamic_cast<const L2ElementRestriction*>(elem_restrict_test);
|
||||
MFEM_VERIFY(l2_elem_restrict,
|
||||
"A real ElementRestriction is required in this setting!");
|
||||
test_multiplicity = 1.0;
|
||||
}
|
||||
|
||||
auto tm = test_multiplicity.ReadWrite();
|
||||
@@ -2036,7 +2040,13 @@ void PADiscreteLinearOperatorExtension::AddMult(
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem_error("In this setting you need a real ElementRestriction!");
|
||||
const L2ElementRestriction* l2_elem_restrict =
|
||||
dynamic_cast<const L2ElementRestriction*>(elem_restrict_test);
|
||||
MFEM_VERIFY(l2_elem_restrict,
|
||||
"In this setting you need a real ElementRestriction!");
|
||||
tempY.SetSize(y.Size());
|
||||
l2_elem_restrict->MultTranspose(localTest, tempY);
|
||||
y += tempY;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+464
-331
File diff suppressed because it is too large
Load Diff
+961
-138
File diff suppressed because it is too large
Load Diff
@@ -392,6 +392,9 @@ private:
|
||||
bool RealInteg();
|
||||
bool ImagInteg();
|
||||
|
||||
void BuildComplexOperator(OperatorHandle &A_r, OperatorHandle &A_i,
|
||||
OperatorHandle &A) const;
|
||||
|
||||
public:
|
||||
SesquilinearForm(FiniteElementSpace *fes,
|
||||
ComplexOperator::Convention
|
||||
@@ -505,6 +508,186 @@ public:
|
||||
virtual ~SesquilinearForm();
|
||||
};
|
||||
|
||||
/** Class for a mixed sesquilinear form
|
||||
|
||||
A mixed sesquilinear form is a generalization of a mixed bilinear form to
|
||||
complex-valued fields. Mixed sesquilinear forms are linear in the second
|
||||
argument but the first argument involves a complex conjugate in the sense
|
||||
that:
|
||||
|
||||
a(alpha u, beta v) = conj(alpha) beta a(u, v)
|
||||
|
||||
The @a convention argument in the class's constructor is documented in the
|
||||
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
|
||||
|
||||
When supplying integrators to the MixedSesquilinearForm either the real or
|
||||
imaginary integrator can be NULL. This indicates that the corresponding
|
||||
portion of the complex-valued material coefficient is equal to zero.
|
||||
*/
|
||||
class MixedSesquilinearForm
|
||||
{
|
||||
private:
|
||||
ComplexOperator::Convention conv;
|
||||
|
||||
MixedBilinearForm * mblfr;
|
||||
MixedBilinearForm * mblfi;
|
||||
|
||||
/* These methods check if the real/imag parts of the sesqulinear form are not
|
||||
empty */
|
||||
bool RealInteg();
|
||||
bool ImagInteg();
|
||||
|
||||
public:
|
||||
MixedSesquilinearForm(
|
||||
FiniteElementSpace * trial_fes,
|
||||
FiniteElementSpace * test_fes,
|
||||
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
|
||||
|
||||
/** @brief Create a MixedSesquilinearForm on the given trial and test
|
||||
FiniteElementSpaces, using the same integrators as the
|
||||
MixedBilinearForms @a bfr and @a bfi.
|
||||
|
||||
The FiniteElementSpace pointers are not owned by the newly constructed
|
||||
object.
|
||||
|
||||
The integrators are copied as pointers and they are not owned by the
|
||||
newly constructed MixedSesquilinearForm. */
|
||||
MixedSesquilinearForm(
|
||||
FiniteElementSpace * trial_fes,
|
||||
FiniteElementSpace * test_fes,
|
||||
MixedBilinearForm * bfr,
|
||||
MixedBilinearForm * bfi,
|
||||
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
|
||||
|
||||
ComplexOperator::Convention GetConvention() const { return conv; }
|
||||
void SetConvention(const ComplexOperator::Convention & convention) { conv = convention; }
|
||||
|
||||
/// Set the desired assembly level.
|
||||
/** Valid choices are:
|
||||
|
||||
- AssemblyLevel::LEGACY (default)
|
||||
- AssemblyLevel::FULL
|
||||
- AssemblyLevel::PARTIAL
|
||||
- AssemblyLevel::ELEMENT
|
||||
- AssemblyLevel::NONE
|
||||
|
||||
This method must be called before assembly. */
|
||||
void SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
{
|
||||
mblfr->SetAssemblyLevel(assembly_level);
|
||||
mblfi->SetAssemblyLevel(assembly_level);
|
||||
}
|
||||
|
||||
MixedBilinearForm & real() { return *mblfr; }
|
||||
MixedBilinearForm & imag() { return *mblfi; }
|
||||
const MixedBilinearForm & real() const { return *mblfr; }
|
||||
const MixedBilinearForm & imag() const { return *mblfi; }
|
||||
|
||||
/// Adds new Domain Integrator.
|
||||
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds new Domain Integrator, restricted to specific attributes.
|
||||
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & elem_marker);
|
||||
|
||||
/// Adds new Boundary Integrator.
|
||||
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/** @brief Adds new boundary Integrator, restricted to specific boundary
|
||||
attributes.
|
||||
|
||||
Assumes ownership of @a bfi.
|
||||
|
||||
The mfem::array @a bdr_marker is stored internally as a pointer to the given
|
||||
mfem::Array<int> object. */
|
||||
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & bdr_marker);
|
||||
|
||||
/// Adds new interior Face Integrator. Assumes ownership of @a bfi.
|
||||
void AddInteriorFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds new boundary Face Integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/** @brief Adds new boundary Face Integrator, restricted to specific boundary
|
||||
attributes.
|
||||
|
||||
Assumes ownership of @a bfi.
|
||||
|
||||
The mfem::array @a bdr_marker is stored internally as a pointer to the given
|
||||
mfem::Array<int> object. */
|
||||
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & bdr_marker);
|
||||
|
||||
/** @brief Add a trace face integrator. Assumes ownership of @a bfi.
|
||||
|
||||
This type of integrator assembles terms over all faces of the mesh using
|
||||
the face FE from the trial space and the two adjacent volume FEs from
|
||||
the test space. */
|
||||
void AddTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> &bdr_marker);
|
||||
|
||||
/// Assemble the local matrix
|
||||
void Assemble(int skip_zeros = 1);
|
||||
|
||||
/// Finalizes the matrix initialization.
|
||||
void Finalize(int skip_zeros = 1);
|
||||
|
||||
/// Updates the internal mixed forms with the new finite element space.
|
||||
virtual void Update();
|
||||
|
||||
/** @brief Return a ComplexSparseMatrix wrapping the local (L-dof) real
|
||||
and imaginary matrices of the form.
|
||||
|
||||
The returned wrapper has to be deleted by the caller, but it does not
|
||||
own the wrapped real and imaginary matrices, which remain owned by
|
||||
this form. */
|
||||
ComplexSparseMatrix *AssembleComplexSparseMatrix();
|
||||
|
||||
/// Return the trial FE space associated with the MixedSesquilinearForm.
|
||||
FiniteElementSpace *TrialFESpace() { return mblfr->TrialFESpace(); }
|
||||
|
||||
/// Read-only access to the associated trial FiniteElementSpace.
|
||||
const FiniteElementSpace *TrialFESpace() const { return mblfr->TrialFESpace(); }
|
||||
|
||||
/// Return the test FE space associated with the MixedSesquilinearForm.
|
||||
FiniteElementSpace *TestFESpace() { return mblfr->TestFESpace(); }
|
||||
|
||||
/// Read-only access to the associated test FiniteElementSpace.
|
||||
const FiniteElementSpace *TestFESpace() const { return mblfr->TestFESpace(); }
|
||||
|
||||
|
||||
void FormRectangularLinearSystem(const Array<int> & ess_trial_tdof_list,
|
||||
const Array<int> & ess_test_tdof_list,
|
||||
Vector & x,
|
||||
Vector & b,
|
||||
OperatorHandle & A,
|
||||
Vector & X,
|
||||
Vector & B);
|
||||
|
||||
void FormRectangularSystemMatrix(const Array<int> & ess_trial_tdof_list,
|
||||
const Array<int> & ess_test_tdof_list,
|
||||
OperatorHandle & A);
|
||||
|
||||
virtual ~MixedSesquilinearForm();
|
||||
};
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
/// Class for parallel complex-valued grid function - real + imaginary part
|
||||
@@ -806,6 +989,12 @@ private:
|
||||
bool RealInteg();
|
||||
bool ImagInteg();
|
||||
|
||||
void SetImaginaryEssentialDiagonalToZero(
|
||||
const Array<int> &ess_tdof_list, OperatorHandle &A);
|
||||
|
||||
void BuildComplexOperator(OperatorHandle &A_r, OperatorHandle &A_i,
|
||||
OperatorHandle &A) const;
|
||||
|
||||
public:
|
||||
ParSesquilinearForm(ParFiniteElementSpace *pf,
|
||||
ComplexOperator::Convention
|
||||
@@ -921,6 +1110,169 @@ public:
|
||||
virtual ~ParSesquilinearForm();
|
||||
};
|
||||
|
||||
/** Class for a parallel mixed sesquilinear form
|
||||
|
||||
A mixed sesquilinear form is a generalization of a mixed bilinear form to
|
||||
complex-valued fields. Mixed sesquilinear forms are linear in the second
|
||||
argument but the first argument involves a complex conjugate in the sense
|
||||
that:
|
||||
|
||||
a(alpha u, beta v) = conj(alpha) beta a(u, v)
|
||||
|
||||
The @a convention argument in the class's constructor is documented in the
|
||||
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
|
||||
|
||||
When supplying integrators to the ParMixedSesquilinearForm either the real
|
||||
or imaginary integrator can be NULL. This indicates that the corresponding
|
||||
portion of the complex-valued material coefficient is equal to zero.
|
||||
*/
|
||||
class ParMixedSesquilinearForm
|
||||
{
|
||||
private:
|
||||
ComplexOperator::Convention conv;
|
||||
|
||||
ParMixedBilinearForm * pmblfr;
|
||||
ParMixedBilinearForm * pmblfi;
|
||||
|
||||
/* These methods check if the real/imag parts of the sesqulinear form are
|
||||
not empty */
|
||||
bool RealInteg();
|
||||
bool ImagInteg();
|
||||
|
||||
public:
|
||||
ParMixedSesquilinearForm(
|
||||
ParFiniteElementSpace * trial_fes,
|
||||
ParFiniteElementSpace * test_fes,
|
||||
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
|
||||
|
||||
/** @brief Create a ParMixedSesquilinearForm on the given trial and test
|
||||
ParFiniteElementSpaces, using the same integrators as the
|
||||
ParMixedBilinearForms @a pbfr and @a pbfi.
|
||||
|
||||
The ParFiniteElementSpace pointers are not owned by the newly
|
||||
constructed object.
|
||||
|
||||
The integrators are copied as pointers and they are not owned by the
|
||||
newly constructed ParMixedSesquilinearForm. */
|
||||
ParMixedSesquilinearForm(
|
||||
ParFiniteElementSpace * trial_fes,
|
||||
ParFiniteElementSpace * test_fes,
|
||||
ParMixedBilinearForm * pbfr,
|
||||
ParMixedBilinearForm * pbfi,
|
||||
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
|
||||
|
||||
ComplexOperator::Convention GetConvention() const { return conv; }
|
||||
void SetConvention(const ComplexOperator::Convention & convention) { conv = convention; }
|
||||
|
||||
/// Set the desired assembly level.
|
||||
/** Valid choices are:
|
||||
|
||||
- AssemblyLevel::LEGACY (default)
|
||||
- AssemblyLevel::FULL
|
||||
- AssemblyLevel::PARTIAL
|
||||
- AssemblyLevel::ELEMENT
|
||||
- AssemblyLevel::NONE
|
||||
|
||||
This method must be called before assembly. */
|
||||
void SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
{
|
||||
pmblfr->SetAssemblyLevel(assembly_level);
|
||||
pmblfi->SetAssemblyLevel(assembly_level);
|
||||
}
|
||||
|
||||
ParMixedBilinearForm & real() { return *pmblfr; }
|
||||
ParMixedBilinearForm & imag() { return *pmblfi; }
|
||||
const ParMixedBilinearForm & real() const { return *pmblfr; }
|
||||
const ParMixedBilinearForm & imag() const { return *pmblfi; }
|
||||
|
||||
/// Adds new Domain Integrator.
|
||||
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds new Domain Integrator, restricted to specific attributes.
|
||||
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & elem_marker);
|
||||
|
||||
/// Adds new Boundary Integrator.
|
||||
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/** @brief Adds new boundary Integrator, restricted to specific boundary
|
||||
attributes.
|
||||
|
||||
Assumes ownership of @a bfi.
|
||||
|
||||
The mfem::array @a bdr_marker is stored internally as a pointer to the given
|
||||
mfem::Array<int> object. */
|
||||
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & bdr_marker);
|
||||
|
||||
/// Adds new interior Face Integrator. Assumes ownership of @a bfi.
|
||||
void AddInteriorFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds new boundary Face Integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/** @brief Adds new boundary Face Integrator, restricted to specific boundary
|
||||
attributes.
|
||||
|
||||
Assumes ownership of @a bfi.
|
||||
|
||||
The mfem::array @a bdr_marker is stored internally as a pointer to the given
|
||||
mfem::Array<int> object. */
|
||||
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & bdr_marker);
|
||||
|
||||
/** @brief Add a trace face integrator. Assumes ownership of @a bfi.
|
||||
|
||||
This type of integrator assembles terms over all faces of the mesh using
|
||||
the face FE from the trial space and the two adjacent volume FEs from
|
||||
the test space. */
|
||||
void AddTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> &bdr_marker);
|
||||
|
||||
/// Assemble the local matrix
|
||||
void Assemble(int skip_zeros = 1);
|
||||
|
||||
/// Finalizes the matrix initialization.
|
||||
void Finalize(int skip_zeros = 1);
|
||||
|
||||
/// Updates the internal mixed forms with the new finite element space.
|
||||
virtual void Update();
|
||||
|
||||
/// Returns the matrix assembled on the true dofs, i.e. P^t A P.
|
||||
/** The returned matrix has to be deleted by the caller. */
|
||||
ComplexHypreParMatrix * ParallelAssemble();
|
||||
|
||||
void FormRectangularLinearSystem(const Array<int> & ess_trial_tdof_list,
|
||||
const Array<int> & ess_test_tdof_list,
|
||||
Vector & x,
|
||||
Vector & b,
|
||||
OperatorHandle & A,
|
||||
Vector & X,
|
||||
Vector & B);
|
||||
|
||||
void FormRectangularSystemMatrix(const Array<int> & ess_trial_tdof_list,
|
||||
const Array<int> & ess_test_tdof_list,
|
||||
OperatorHandle & A);
|
||||
|
||||
virtual ~ParMixedSesquilinearForm();
|
||||
};
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
}
|
||||
|
||||
@@ -51,4 +51,52 @@ DifferentiableOperator::DifferentiableOperator(
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void FDJacobian::Mult(const Vector &v, Vector &y) const
|
||||
{
|
||||
// See [1] for choice of eps.
|
||||
//
|
||||
// [1] Woodward, C.S., Gardner, D.J. and Evans, K.J., 2015. On the use of
|
||||
// finite difference matrix-vector products in Newton-Krylov solvers for
|
||||
// implicit climate dynamics with spectral elements. Procedia Computer
|
||||
// Science, 51, pp.2036-2045.
|
||||
real_t eps;
|
||||
if (fixed_eps > 0.0)
|
||||
{
|
||||
eps = fixed_eps;
|
||||
}
|
||||
else
|
||||
{
|
||||
const real_t vnorm_local = v.Norml2();
|
||||
real_t vnorm;
|
||||
MPI_Allreduce(&vnorm_local, &vnorm, 1, MPITypeMap<real_t>::mpi_type, MPI_SUM,
|
||||
MPI_COMM_WORLD);
|
||||
eps = lambda * (lambda + xnorm / vnorm);
|
||||
}
|
||||
|
||||
// x + eps * v
|
||||
{
|
||||
const auto d_v = v.Read();
|
||||
const auto d_x = x.Read();
|
||||
auto d_xpev = xpev.Write();
|
||||
mfem::forall(x.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_xpev[i] = d_x[i] + eps * d_v[i];
|
||||
});
|
||||
}
|
||||
|
||||
// y = f(x + eps * v)
|
||||
op.Mult(xpev, y);
|
||||
|
||||
// y = (f(x + eps * v) - f(x)) / eps
|
||||
{
|
||||
const auto d_f = f.Read();
|
||||
auto d_y = y.ReadWrite();
|
||||
mfem::forall(f.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_y[i] = (d_y[i] - d_f[i]) / eps;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
+23
-22
@@ -697,17 +697,18 @@ void DifferentiableOperator::AddIntegrator(
|
||||
|
||||
// The explicit captures are necessary to avoid dependency on
|
||||
// the specific instance of this class (this pointer).
|
||||
restriction_callback =
|
||||
[=, solutions = this->solutions, parameters = this->parameters]
|
||||
(std::vector<Vector> &sol,
|
||||
const std::vector<Vector> &par,
|
||||
std::vector<Vector> &f)
|
||||
restriction_callback = [element_dof_ordering,
|
||||
solutions_ = this->solutions,
|
||||
parameters_ = this->parameters]
|
||||
(std::vector<Vector> &sol,
|
||||
const std::vector<Vector> &par,
|
||||
std::vector<Vector> &f)
|
||||
{
|
||||
restriction<entity_t>(solutions, sol, f,
|
||||
restriction<entity_t>(solutions_, sol, f,
|
||||
element_dof_ordering);
|
||||
restriction<entity_t>(parameters, par, f,
|
||||
restriction<entity_t>(parameters_, par, f,
|
||||
element_dof_ordering,
|
||||
solutions.size());
|
||||
solutions_.size());
|
||||
};
|
||||
|
||||
prolongation_transpose = get_prolongation_transpose(
|
||||
@@ -835,19 +836,19 @@ void DifferentiableOperator::AddIntegrator(
|
||||
|
||||
// capture by ref:
|
||||
&restriction_cb = this->restriction_callback,
|
||||
&fields_e = this->fields_e,
|
||||
&residual_e = this->residual_e,
|
||||
&output_restriction_transpose = this->output_restriction_transpose
|
||||
&fields_e_ = this->fields_e,
|
||||
&residual_e_ = this->residual_e,
|
||||
&output_restriction_transpose_ = this->output_restriction_transpose
|
||||
]
|
||||
(std::vector<Vector> &sol, const std::vector<Vector> &par, Vector &res)
|
||||
mutable // mutable: needed to modify 'shmem_cache'
|
||||
{
|
||||
restriction_cb(sol, par, fields_e);
|
||||
restriction_cb(sol, par, fields_e_);
|
||||
|
||||
residual_e = 0.0;
|
||||
auto ye = Reshape(residual_e.ReadWrite(), test_vdim, num_test_dof, num_entities);
|
||||
residual_e_ = 0.0;
|
||||
auto ye = Reshape(residual_e_.ReadWrite(), test_vdim, num_test_dof, num_entities);
|
||||
|
||||
auto wrapped_fields_e = wrap_fields(fields_e,
|
||||
auto wrapped_fields_e = wrap_fields(fields_e_,
|
||||
action_shmem_info.field_sizes,
|
||||
num_entities);
|
||||
|
||||
@@ -878,7 +879,7 @@ void DifferentiableOperator::AddIntegrator(
|
||||
y, fhat, output_fop, output_dtq_shmem[0],
|
||||
scratch_shmem, dimension, use_sum_factorization);
|
||||
}, num_entities, thread_blocks, action_shmem_info.total_size, shmem_cache.ReadWrite());
|
||||
output_restriction_transpose(residual_e, res);
|
||||
output_restriction_transpose_(residual_e_, res);
|
||||
});
|
||||
|
||||
// Without this compile-time check, some valid instantiations of this method
|
||||
@@ -1193,7 +1194,7 @@ void DifferentiableOperator::AddIntegrator(
|
||||
|
||||
// capture by ref:
|
||||
&qpdc_mem = derivative_qp_caches_ref,
|
||||
&fields = fields_ref
|
||||
&fields_ = fields_ref
|
||||
](std::vector<Vector> &f_e, SparseMatrix *&A) mutable
|
||||
{
|
||||
auto wrapped_fields_e = wrap_fields(f_e, shmem_info.field_sizes,
|
||||
@@ -1241,14 +1242,14 @@ void DifferentiableOperator::AddIntegrator(
|
||||
{
|
||||
if (input_is_dependent[s])
|
||||
{
|
||||
trial_field = &fields[input_to_field[s]];
|
||||
trial_field = &fields_[input_to_field[s]];
|
||||
}
|
||||
}
|
||||
|
||||
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
|
||||
(&trial_field->data);
|
||||
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
|
||||
(&fields[output_to_field[0]].data);
|
||||
(&fields_[output_to_field[0]].data);
|
||||
|
||||
A = new SparseMatrix(test_fes->GetVSize(), trial_fes->GetVSize());
|
||||
|
||||
@@ -1334,7 +1335,7 @@ void DifferentiableOperator::AddIntegrator(
|
||||
input_to_field,
|
||||
output_to_field,
|
||||
&spmatcb = assemble_derivative_sparsematrix_callbacks_ref,
|
||||
&fields = fields_ref
|
||||
&fields_ = fields_ref
|
||||
](std::vector<Vector> &f_e, HypreParMatrix *&A) mutable
|
||||
{
|
||||
SparseMatrix *spmat = nullptr;
|
||||
@@ -1366,14 +1367,14 @@ void DifferentiableOperator::AddIntegrator(
|
||||
{
|
||||
if (input_is_dependent[s])
|
||||
{
|
||||
trial_field = &fields[input_to_field[s]];
|
||||
trial_field = &fields_[input_to_field[s]];
|
||||
}
|
||||
}
|
||||
|
||||
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
|
||||
(&trial_field->data);
|
||||
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
|
||||
(&fields[output_to_field[0]].data);
|
||||
(&fields_[output_to_field[0]].data);
|
||||
|
||||
if (same_test_and_trial)
|
||||
{
|
||||
|
||||
+742
-768
File diff suppressed because it is too large
Load Diff
+9
-52
@@ -597,7 +597,7 @@ struct ThreadBlocks
|
||||
int z = 1;
|
||||
};
|
||||
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP)
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP_LANG)
|
||||
template <typename func_t>
|
||||
__global__ void forall_kernel_shmem(func_t f, int n)
|
||||
{
|
||||
@@ -617,10 +617,11 @@ void forall(func_t f,
|
||||
int num_shmem = 0,
|
||||
real_t *shmem = nullptr)
|
||||
{
|
||||
if (Device::Allows(Backend::CUDA_MASK) ||
|
||||
Device::Allows(Backend::HIP_MASK))
|
||||
internal::RequireKernelCompilation();
|
||||
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP_LANG)
|
||||
if (Device::Allows(Backend::CUDA_MASK | Backend::HIP_MASK))
|
||||
{
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP)
|
||||
// int gridsize = (N + Z - 1) / Z;
|
||||
int num_bytes = num_shmem * sizeof(decltype(shmem));
|
||||
dim3 block_size(blocks.x, blocks.y, blocks.z);
|
||||
@@ -631,9 +632,10 @@ void forall(func_t f,
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
#endif
|
||||
MFEM_DEVICE_SYNC;
|
||||
#endif
|
||||
return;
|
||||
}
|
||||
else if (Device::Allows(Backend::CPU_MASK))
|
||||
#endif
|
||||
if (Device::Allows(Backend::CPU_MASK))
|
||||
{
|
||||
MFEM_ASSERT(!((bool)num_shmem != (bool)shmem),
|
||||
"Backend::CPU needs a pre-allocated shared memory block");
|
||||
@@ -671,52 +673,7 @@ public:
|
||||
MPI_COMM_WORLD);
|
||||
}
|
||||
|
||||
void Mult(const Vector &v, Vector &y) const override
|
||||
{
|
||||
// See [1] for choice of eps.
|
||||
//
|
||||
// [1] Woodward, C.S., Gardner, D.J. and Evans, K.J., 2015. On the use of
|
||||
// finite difference matrix-vector products in Newton-Krylov solvers for
|
||||
// implicit climate dynamics with spectral elements. Procedia Computer
|
||||
// Science, 51, pp.2036-2045.
|
||||
real_t eps;
|
||||
if (fixed_eps > 0.0)
|
||||
{
|
||||
eps = fixed_eps;
|
||||
}
|
||||
else
|
||||
{
|
||||
const real_t vnorm_local = v.Norml2();
|
||||
real_t vnorm;
|
||||
MPI_Allreduce(&vnorm_local, &vnorm, 1, MPITypeMap<real_t>::mpi_type, MPI_SUM,
|
||||
MPI_COMM_WORLD);
|
||||
eps = lambda * (lambda + xnorm / vnorm);
|
||||
}
|
||||
|
||||
// x + eps * v
|
||||
{
|
||||
const auto d_v = v.Read();
|
||||
const auto d_x = x.Read();
|
||||
auto d_xpev = xpev.Write();
|
||||
mfem::forall(x.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_xpev[i] = d_x[i] + eps * d_v[i];
|
||||
});
|
||||
}
|
||||
|
||||
// y = f(x + eps * v)
|
||||
op.Mult(xpev, y);
|
||||
|
||||
// y = (f(x + eps * v) - f(x)) / eps
|
||||
{
|
||||
const auto d_f = f.Read();
|
||||
auto d_y = y.ReadWrite();
|
||||
mfem::forall(f.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_y[i] = (d_y[i] - d_f[i]) / eps;
|
||||
});
|
||||
}
|
||||
}
|
||||
void Mult(const Vector &v, Vector &y) const override;
|
||||
|
||||
virtual MemoryClass GetMemoryClass() const override
|
||||
{
|
||||
|
||||
+6
-5
@@ -1316,13 +1316,14 @@ void VectorFiniteElement::Project_RT(
|
||||
}
|
||||
}
|
||||
|
||||
void VectorFiniteElement::ProjectGrad_RT(
|
||||
void VectorFiniteElement::ProjectCurl2D_RT(
|
||||
const real_t *nk, const Array<int> &d2n, const FiniteElement &fe,
|
||||
ElementTransformation &Trans, DenseMatrix &grad) const
|
||||
{
|
||||
// 2D "ProjectCurl_RT"
|
||||
if (dim != 2)
|
||||
{
|
||||
mfem_error("VectorFiniteElement::ProjectGrad_RT works only in 2D!");
|
||||
mfem_error("VectorFiniteElement::ProjectCurl2D_RT works only in 2D!");
|
||||
}
|
||||
|
||||
DenseMatrix dshape(fe.GetDof(), fe.GetDim());
|
||||
@@ -1333,8 +1334,8 @@ void VectorFiniteElement::ProjectGrad_RT(
|
||||
for (int k = 0; k < dof; k++)
|
||||
{
|
||||
fe.CalcDShape(Nodes.IntPoint(k), dshape);
|
||||
tk[0] = nk[d2n[k]*dim+1];
|
||||
tk[1] = -nk[d2n[k]*dim];
|
||||
tk[0] = -nk[d2n[k]*dim+1];
|
||||
tk[1] = nk[d2n[k]*dim];
|
||||
dshape.Mult(tk, grad_k);
|
||||
for (int j = 0; j < grad_k.Size(); j++)
|
||||
{
|
||||
@@ -1381,7 +1382,7 @@ void VectorFiniteElement::ProjectCurl_ND(
|
||||
}
|
||||
}
|
||||
|
||||
void VectorFiniteElement::ProjectCurl_RT(
|
||||
void VectorFiniteElement::ProjectCurl3D_RT(
|
||||
const real_t *nk, const Array<int> &d2n, const FiniteElement &fe,
|
||||
ElementTransformation &Trans, DenseMatrix &curl) const
|
||||
{
|
||||
|
||||
+10
-7
@@ -957,10 +957,11 @@ protected:
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &I) const;
|
||||
|
||||
// rotated gradient in 2D
|
||||
void ProjectGrad_RT(const real_t *nk, const Array<int> &d2n,
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &grad) const;
|
||||
// Input is a scalar representing the Z (out of plane) component, Output is
|
||||
// the X-Y (in-plane) RT curl
|
||||
void ProjectCurl2D_RT(const real_t *nk, const Array<int> &d2n,
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &grad) const;
|
||||
|
||||
// Compute the curl as a discrete operator from ND FE (fe) to ND FE (this).
|
||||
// The natural FE for the range is RT, so this is an approximation.
|
||||
@@ -968,9 +969,9 @@ protected:
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const;
|
||||
|
||||
void ProjectCurl_RT(const real_t *nk, const Array<int> &d2n,
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const;
|
||||
void ProjectCurl3D_RT(const real_t *nk, const Array<int> &d2n,
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const;
|
||||
|
||||
/** @brief Project a vector coefficient onto the ND basis functions
|
||||
@param tk Edge tangent vectors for this element type
|
||||
@@ -1446,6 +1447,8 @@ public:
|
||||
dof2quad_array_open);
|
||||
}
|
||||
|
||||
const Poly_1D::Basis &GetOpenBasis1D() const { return obasis1d; }
|
||||
|
||||
virtual ~VectorTensorFiniteElement();
|
||||
};
|
||||
|
||||
|
||||
+38
-1
@@ -1282,12 +1282,49 @@ ND_SegmentElement::ND_SegmentElement(const int p, const int ob_type)
|
||||
}
|
||||
}
|
||||
|
||||
void ND_SegmentElement::CalcShape(const IntegrationPoint &ip,
|
||||
Vector &shape) const
|
||||
{
|
||||
if (obasis1d.IsIntegratedType()) { obasis1d.ScaleIntegrated(false); }
|
||||
obasis1d.Eval(ip.x, shape);
|
||||
}
|
||||
|
||||
void ND_SegmentElement::CalcVShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &shape) const
|
||||
{
|
||||
Vector vshape(shape.Data(), dof);
|
||||
|
||||
obasis1d.Eval(ip.x, vshape);
|
||||
CalcShape(ip, vshape);
|
||||
}
|
||||
|
||||
void ND_SegmentElement::ProjectIntegrated(VectorCoefficient &vc,
|
||||
ElementTransformation &Trans,
|
||||
Vector &dofs) const
|
||||
{
|
||||
MFEM_ASSERT(obasis1d.IsIntegratedType(), "Not integrated type");
|
||||
real_t vk[Geometry::MaxDim];
|
||||
Vector xk(vk, vc.GetVDim());
|
||||
|
||||
const real_t *cp = poly1d.ClosedPoints(dof, BasisType::GaussLobatto);
|
||||
const IntegrationRule &ir = IntRules.Get(Geometry::SEGMENT, dof);
|
||||
IntegrationPoint ip;
|
||||
|
||||
for (int i = 0; i < dof; i++)
|
||||
{
|
||||
const real_t h = cp[i+1] - cp[i];
|
||||
real_t val = 0.0;
|
||||
|
||||
for (int q = 0; q < ir.GetNPoints(); q++)
|
||||
{
|
||||
const IntegrationPoint &ip1d = ir.IntPoint(q);
|
||||
ip.x = cp[i] + h*ip1d.x;
|
||||
Trans.SetIntPoint(&ip);
|
||||
vc.Eval(xk, Trans, ip);
|
||||
val += ip1d.weight*Trans.Jacobian().InnerProduct(tk, vk);
|
||||
}
|
||||
|
||||
dofs(i) = val*h;
|
||||
}
|
||||
}
|
||||
|
||||
const real_t ND_WedgeElement::tk[15] =
|
||||
|
||||
+10
-3
@@ -303,8 +303,7 @@ public:
|
||||
/** @brief Construct the ND_SegmentElement of order @a p and open
|
||||
BasisType @a ob_type */
|
||||
ND_SegmentElement(const int p, const int ob_type = BasisType::GaussLegendre);
|
||||
void CalcShape(const IntegrationPoint &ip, Vector &shape) const override
|
||||
{ obasis1d.Eval(ip.x, shape); }
|
||||
void CalcShape(const IntegrationPoint &ip, Vector &shape) const override;
|
||||
void CalcVShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &shape) const override;
|
||||
void CalcVShape(ElementTransformation &Trans,
|
||||
@@ -325,7 +324,10 @@ public:
|
||||
using FiniteElement::Project;
|
||||
void Project(VectorCoefficient &vc,
|
||||
ElementTransformation &Trans, Vector &dofs) const override
|
||||
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
|
||||
{
|
||||
if (obasis1d.IsIntegratedType()) { ProjectIntegrated(vc, Trans, dofs); }
|
||||
else { Project_ND(tk, dof2tk, vc, Trans, dofs); }
|
||||
}
|
||||
void ProjectMatrixCoefficient(MatrixCoefficient &mc,
|
||||
ElementTransformation &T,
|
||||
Vector &dofs) const override
|
||||
@@ -338,6 +340,11 @@ public:
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &grad) const override
|
||||
{ ProjectGrad_ND(tk, dof2tk, fe, Trans, grad); }
|
||||
|
||||
protected:
|
||||
void ProjectIntegrated(VectorCoefficient &vc,
|
||||
ElementTransformation &Trans,
|
||||
Vector &dofs) const;
|
||||
};
|
||||
|
||||
class ND_WedgeElement : public VectorFiniteElement
|
||||
|
||||
+6
-16
@@ -73,16 +73,11 @@ public:
|
||||
void Project(const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &I) const override
|
||||
{ Project_RT(nk, dof2nk, fe, Trans, I); }
|
||||
// Gradient + rotation = Curl: H1 -> H(div)
|
||||
void ProjectGrad(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &grad) const override
|
||||
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, grad); }
|
||||
// Curl = Gradient + rotation: H1 -> H(div)
|
||||
void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const override
|
||||
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectCurl2D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
|
||||
void GetFaceMap(const int face_id, Array<int> &face_map) const override;
|
||||
|
||||
@@ -148,7 +143,7 @@ public:
|
||||
void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const override
|
||||
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
|
||||
/// @brief Return the mapping from lexicographically ordered face DOFs to
|
||||
/// lexicographically ordered element DOFs corresponding to local face
|
||||
@@ -210,16 +205,11 @@ public:
|
||||
void Project(const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &I) const override
|
||||
{ Project_RT(nk, dof2nk, fe, Trans, I); }
|
||||
// Gradient + rotation = Curl: H1 -> H(div)
|
||||
void ProjectGrad(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &grad) const override
|
||||
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, grad); }
|
||||
// Curl = Gradient + rotation: H1 -> H(div)
|
||||
void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const override
|
||||
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectCurl2D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
};
|
||||
|
||||
|
||||
@@ -274,7 +264,7 @@ public:
|
||||
void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const override
|
||||
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
};
|
||||
|
||||
class RT_WedgeElement : public VectorFiniteElement
|
||||
@@ -332,7 +322,7 @@ public:
|
||||
void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const override
|
||||
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
};
|
||||
|
||||
/** Arbitrary order H(Div) basis functions defined on pyramid-shaped elements
|
||||
@@ -428,7 +418,7 @@ public:
|
||||
virtual void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const
|
||||
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
|
||||
void CalcRawVShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &shape) const;
|
||||
|
||||
@@ -100,6 +100,10 @@ public:
|
||||
return FiniteElementForGeometry(GeomType);
|
||||
}
|
||||
|
||||
/** @brief Returns a collection of the trace elements.
|
||||
|
||||
@note The collection is owned by the caller and is NOT deleted in the
|
||||
destructor. */
|
||||
virtual FiniteElementCollection *GetTraceCollection() const;
|
||||
|
||||
virtual ~FiniteElementCollection();
|
||||
|
||||
+4
-4
@@ -556,7 +556,7 @@ void obboxsurf_calc_3(Vector &bb,
|
||||
gslib::lagrange_fun *const lag = gslib::gll_lag_setup(work, n);
|
||||
lag(I0, work, n, 1, 0);
|
||||
|
||||
for (int ie = 0; ie < nel; ie++,x+=n2,y+=n2,z+=n2)
|
||||
for (int ie = 0; (unsigned)ie < nel; ie++,x+=n2,y+=n2,z+=n2)
|
||||
{
|
||||
struct gslib::dbl_range ab[3];
|
||||
struct gslib::dbl_range tb[3];
|
||||
@@ -780,7 +780,7 @@ void obboxedge_calc_2(Vector &bb,
|
||||
gslib::lagrange_fun *const lag = gslib::gll_lag_setup(work, nr);
|
||||
lag(I0r, work, nr,1, 0);
|
||||
|
||||
for (int ie = 0; ie < nel; ie++,x+=nr,y+=nr)
|
||||
for (int ie = 0; (unsigned)ie < nel; ie++,x+=nr,y+=nr)
|
||||
{
|
||||
double x0[2], A[4];
|
||||
struct gslib::dbl_range ab[2], tb[2];
|
||||
@@ -892,7 +892,7 @@ void obboxedge_calc_3(Vector &bb,
|
||||
gslib::lagrange_fun *const lag = gslib::gll_lag_setup(work, nr);
|
||||
lag(I0r, work, nr, 1, 0);
|
||||
|
||||
for (int ie = 0; ie < nel; ie++,x+=nr,y+=nr,z+=nr)
|
||||
for (int ie = 0; (unsigned)ie < nel; ie++,x+=nr,y+=nr,z+=nr)
|
||||
{
|
||||
double x0[3], A[9], Ai[9];
|
||||
struct gslib::dbl_range ab[3], tb[3];
|
||||
@@ -4518,7 +4518,7 @@ Mesh* FindPointsGSLIB::GetBoundingBoxMesh(int type)
|
||||
int eidx = 0;
|
||||
if (myid == save_rank)
|
||||
{
|
||||
for (int p = 0; p < gsl_comm->np; p++)
|
||||
for (int p = 0; (unsigned)p < gsl_comm->np; p++)
|
||||
{
|
||||
if (static_cast<unsigned int>(p) != save_rank)
|
||||
{
|
||||
|
||||
@@ -368,6 +368,8 @@ void HybridizationExtension::ConstructH()
|
||||
|
||||
CAhatInvCt = 0.0;
|
||||
|
||||
// Fill the face-to-face adjacency array. Two faces are adjacent if they are
|
||||
// incident to a common element.
|
||||
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int fi)
|
||||
{
|
||||
const int begin_f = d_face_face_offsets[fi];
|
||||
@@ -403,6 +405,12 @@ void HybridizationExtension::ConstructH()
|
||||
}
|
||||
}
|
||||
}
|
||||
// Fill unused entries with -1 to indicate invalid
|
||||
const int end_f = d_face_face_offsets[fi + 1];
|
||||
for (int i = begin_f + idx; i < end_f; ++i)
|
||||
{
|
||||
d_face_to_face[i] = -1;
|
||||
}
|
||||
});
|
||||
|
||||
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int fi)
|
||||
@@ -412,6 +420,7 @@ void HybridizationExtension::ConstructH()
|
||||
for (int idx_j = begin; idx_j < end; ++idx_j)
|
||||
{
|
||||
const int fj = d_face_to_face[idx_j];
|
||||
if (fj < 0) { break; }
|
||||
for (int ei = 0; ei < 2; ++ei)
|
||||
{
|
||||
const int e = d_face_to_el(0, ei, fi);
|
||||
|
||||
@@ -178,6 +178,8 @@ void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
// Assumes tensor-product elements
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const FiniteElement &el = *fes.GetTypicalFE();
|
||||
MFEM_VERIFY(el.GetMapType() == FiniteElement::VALUE,
|
||||
"Only value map type currently supported");
|
||||
ElementTransformation &Trans = *mesh->GetTypicalElementTransformation();
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, Trans);
|
||||
if (DeviceCanUseCeed())
|
||||
|
||||
@@ -147,18 +147,16 @@ void PAHcurlMassAssembleDiagonal3D(const int D1D,
|
||||
}); // end of element loop
|
||||
}
|
||||
|
||||
void PAHcurlMassApply2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &bo,
|
||||
const Array<real_t> &bc,
|
||||
const Array<real_t> &bot,
|
||||
const Array<real_t> &bct,
|
||||
const Vector &pa_data,
|
||||
const Vector &x,
|
||||
Vector &y)
|
||||
void PAHcurlMassApply2D(const int NE, const bool symmetric,
|
||||
[[maybe_unused]] const bool scalar_coeff,
|
||||
const Array<real_t> &bo, const Array<real_t> &bc,
|
||||
const Array<real_t> &bot, const Array<real_t> &bct,
|
||||
const Vector &pa_data, const Vector &x, Vector &y,
|
||||
const int D1D, [[maybe_unused]] const int TestD1D,
|
||||
const int Q1D)
|
||||
{
|
||||
MFEM_ASSERT(D1D == TestD1D,
|
||||
"Trial and Test space must have the same number of dofs");
|
||||
auto Bo = Reshape(bo.Read(), Q1D, D1D-1);
|
||||
auto Bc = Reshape(bc.Read(), Q1D, D1D);
|
||||
auto Bot = Reshape(bot.Read(), D1D-1, Q1D);
|
||||
@@ -277,18 +275,16 @@ void PAHcurlMassApply2D(const int D1D,
|
||||
}); // end of element loop
|
||||
}
|
||||
|
||||
void PAHcurlMassApply3D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &bo,
|
||||
const Array<real_t> &bc,
|
||||
const Array<real_t> &bot,
|
||||
const Array<real_t> &bct,
|
||||
const Vector &pa_data,
|
||||
const Vector &x,
|
||||
Vector &y)
|
||||
void PAHcurlMassApply3D(const int NE, const bool symmetric,
|
||||
[[maybe_unused]] const bool scalar_coeff,
|
||||
const Array<real_t> &bo, const Array<real_t> &bc,
|
||||
const Array<real_t> &bot, const Array<real_t> &bct,
|
||||
const Vector &pa_data, const Vector &x, Vector &y,
|
||||
const int D1D, [[maybe_unused]] const int TestD1D,
|
||||
const int Q1D)
|
||||
{
|
||||
MFEM_VERIFY(D1D == TestD1D,
|
||||
"Trial and test spaces must have same number of dofs");
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().HCURL_MAX_D1D,
|
||||
"Error: D1D > MAX_D1D");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().HCURL_MAX_Q1D,
|
||||
@@ -789,6 +785,23 @@ void PAHcurlL2Setup2D(const int Q1D,
|
||||
});
|
||||
}
|
||||
|
||||
void PAHcurlL2IntSetup2D(const int Q1D, const int NE, const Array<real_t> &w,
|
||||
Vector &coeff, const Vector &detJ, Vector &op)
|
||||
{
|
||||
const int NQ = Q1D*Q1D;
|
||||
auto W = w.Read();
|
||||
auto C = Reshape(coeff.Read(), NQ, NE);
|
||||
auto J = Reshape(detJ.Read(), NQ, NE);
|
||||
auto y = Reshape(op.Write(), NQ, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
y(q,e) = W[q] * C(q,e) / J(q,e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void PAHcurlL2Setup3D(const int NQ,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
|
||||
@@ -181,228 +181,312 @@ inline void SmemPAHcurlMassAssembleDiagonal3D(const int d1d,
|
||||
}
|
||||
|
||||
// PA H(curl) Mass Apply 2D kernel
|
||||
void PAHcurlMassApply2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &bo,
|
||||
const Array<real_t> &bc,
|
||||
const Array<real_t> &bot,
|
||||
const Array<real_t> &bct,
|
||||
const Vector &pa_data,
|
||||
const Vector &x,
|
||||
Vector &y);
|
||||
void PAHcurlMassApply2D(const int NE, const bool symmetric,
|
||||
const bool scalar_coeff, const Array<real_t> &bo,
|
||||
const Array<real_t> &bc, const Array<real_t> &bot,
|
||||
const Array<real_t> &bct, const Vector &pa_data,
|
||||
const Vector &x, Vector &y, const int TrialD1D,
|
||||
const int TestD1D, const int Q1D);
|
||||
|
||||
// PA H(curl) Mass Apply 3D kernel
|
||||
void PAHcurlMassApply3D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &bo,
|
||||
const Array<real_t> &bc,
|
||||
const Array<real_t> &bot,
|
||||
const Array<real_t> &bct,
|
||||
const Vector &pa_data,
|
||||
const Vector &x,
|
||||
Vector &y);
|
||||
void PAHcurlMassApply3D(const int NE, const bool symmetric,
|
||||
[[maybe_unused]] const bool scalar_coeff,
|
||||
const Array<real_t> &bo, const Array<real_t> &bc,
|
||||
const Array<real_t> &bot, const Array<real_t> &bct,
|
||||
const Vector &pa_data, const Vector &x, Vector &y,
|
||||
const int TrialD1D, [[maybe_unused]] const int TestD1D,
|
||||
const int Q1D);
|
||||
|
||||
// Shared memory PA H(curl) Mass Apply 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void SmemPAHcurlMassApply3D(const int d1d,
|
||||
const int q1d,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &bo,
|
||||
const Array<real_t> &bc,
|
||||
const Array<real_t> &bot,
|
||||
const Array<real_t> &bct,
|
||||
const Vector &pa_data,
|
||||
const Vector &x,
|
||||
Vector &y)
|
||||
template <int T_D1D = 0, int T_Q1D = 0, int TBATCH = 0, bool ACCUMULATE = true>
|
||||
inline void SmemPAHcurlMassApply3D(
|
||||
const int NE, const bool symmetric, [[maybe_unused]] const bool scalar_coeff,
|
||||
const Array<real_t> &bo, const Array<real_t> &bc,
|
||||
[[maybe_unused]] const Array<real_t> &bot,
|
||||
[[maybe_unused]] const Array<real_t> &bct, const Vector &pa_data,
|
||||
const Vector &x, Vector &y, const int d1d = 0,
|
||||
[[maybe_unused]] const int test_d1d = 0, const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
MFEM_VERIFY(T_D1D || d1d <= DeviceDofQuadLimits::Get().HCURL_MAX_D1D,
|
||||
"Error: d1d > HCURL_MAX_D1D");
|
||||
MFEM_VERIFY(T_Q1D || q1d <= DeviceDofQuadLimits::Get().HCURL_MAX_Q1D,
|
||||
"Error: q1d > HCURL_MAX_Q1D");
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
MFEM_ASSERT(Q1D >= D1D, "Expected Q1D >= D1D");
|
||||
const int dataSize = symmetric ? 6 : 9;
|
||||
|
||||
auto Bo = Reshape(bo.Read(), Q1D, D1D-1);
|
||||
auto Bc = Reshape(bc.Read(), Q1D, D1D);
|
||||
auto op = Reshape(pa_data.Read(), Q1D, Q1D, Q1D, dataSize, NE);
|
||||
auto X = Reshape(x.Read(), 3*(D1D-1)*D1D*D1D, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), 3*(D1D-1)*D1D*D1D, NE);
|
||||
// assume trial space == test space
|
||||
auto Bo = bo.Read();
|
||||
auto Bc = bc.Read();
|
||||
auto op =
|
||||
Reshape(pa_data.Read(), Q1D, Q1D, Q1D, dataSize, NE);
|
||||
auto X_ = Reshape(x.Read(), 3 * (D1D - 1) * D1D * D1D, NE);
|
||||
auto y_ = y.ReadWrite();
|
||||
|
||||
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
constexpr int MD_ = T_D1D ? T_D1D : DofQuadLimits::HCURL_MAX_D1D;
|
||||
constexpr int MQ_ = T_Q1D ? T_Q1D : DofQuadLimits::HCURL_MAX_Q1D;
|
||||
constexpr int MDQ_ = std::max(MD_, MQ_);
|
||||
constexpr int MB_ = TBATCH ? TBATCH : 1;
|
||||
|
||||
mfem::forall_2D_batch<MDQ_ * MDQ_ * MDQ_ * MB_>(
|
||||
NE, MDQ_ * MDQ_ * MDQ_, 1, MB_, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)
|
||||
constexpr int nbz = TBATCH ? TBATCH : 1;
|
||||
int tidz = MFEM_THREAD_ID(z);
|
||||
#else
|
||||
constexpr int nbz = 1;
|
||||
constexpr int tidz = 0;
|
||||
#endif
|
||||
|
||||
constexpr int VDIM = 3;
|
||||
constexpr int MD1D = T_D1D ? T_D1D : DofQuadLimits::HCURL_MAX_D1D;
|
||||
constexpr int MQ1D = T_Q1D ? T_Q1D : DofQuadLimits::HCURL_MAX_Q1D;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MDQ = std::max(MD1D, MQ1D);
|
||||
|
||||
MFEM_SHARED real_t sBo[MQ1D][MD1D];
|
||||
MFEM_SHARED real_t sBc[MQ1D][MD1D];
|
||||
// nvcc limit work-around: can't have Y_ be captured first in
|
||||
// if constexpr, so capture y_ and construct Y_ locally
|
||||
// only works on GPU
|
||||
auto Y = Reshape(y_, VDIM * (D1D - 1) * D1D * D1D, NE);
|
||||
|
||||
real_t op9[9];
|
||||
MFEM_SHARED real_t sop[9*MQ1D*MQ1D];
|
||||
MFEM_SHARED real_t mass[MQ1D][MQ1D][3];
|
||||
MFEM_SHARED real_t sBo[MDQ * (MD1D - 1)];
|
||||
MFEM_SHARED real_t sBc[MDQ * MD1D];
|
||||
auto BO = Reshape(sBo, Q1D, D1D - 1);
|
||||
auto BC = Reshape(sBc, Q1D, D1D);
|
||||
|
||||
MFEM_SHARED real_t sX[MD1D][MD1D][MD1D];
|
||||
MFEM_SHARED real_t sX[nbz * VDIM * (MD1D - 1) * MD1D * MD1D];
|
||||
MFEM_SHARED real_t sm0[nbz * VDIM * MDQ * MDQ * MDQ];
|
||||
MFEM_SHARED real_t sm1[nbz * VDIM * MDQ * MDQ * MDQ];
|
||||
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
real_t(*X)[nbz][(MD1D - 1) * MD1D * MD1D] =
|
||||
(real_t(*)[nbz][(MD1D - 1) * MD1D * MD1D])(sX);
|
||||
// shapes of buffers always use MQ1D to mitigate shared memory bank
|
||||
// conflicts
|
||||
real_t(*DDQ)[nbz][MQ1D][MQ1D][MQ1D] =
|
||||
(real_t(*)[nbz][MQ1D][MQ1D][MQ1D])(sm0);
|
||||
real_t(*DQQ)[nbz][MQ1D][MQ1D][MQ1D] =
|
||||
(real_t(*)[nbz][MQ1D][MQ1D][MQ1D])(sm1);
|
||||
real_t(*QQQ)[nbz][MQ1D][MQ1D][MQ1D] =
|
||||
(real_t(*)[nbz][MQ1D][MQ1D][MQ1D])(sm0);
|
||||
real_t(*QQD)[nbz][MQ1D][MQ1D][MQ1D] =
|
||||
(real_t(*)[nbz][MQ1D][MQ1D][MQ1D])(sm1);
|
||||
real_t(*QDD)[nbz][MQ1D][MQ1D][MQ1D] =
|
||||
(real_t(*)[nbz][MQ1D][MQ1D][MQ1D])(sm0);
|
||||
|
||||
// load dofs into smem
|
||||
const int offset = (D1D - 1) * D1D * D1D;
|
||||
MFEM_FOREACH_THREAD_DIRECT(ix, x, offset)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
for (int dim = 0; dim < VDIM; ++dim)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
for (int i=0; i<dataSize; ++i)
|
||||
{
|
||||
op9[i] = op(qx,qy,qz,i,e);
|
||||
}
|
||||
}
|
||||
X[dim][tidz][ix] = X_(ix + dim * offset, e);
|
||||
}
|
||||
}
|
||||
|
||||
const int tidx = MFEM_THREAD_ID(x);
|
||||
const int tidy = MFEM_THREAD_ID(y);
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
|
||||
// load basis functions data
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
MFEM_FOREACH_THREAD_DIRECT(ix, x, D1D * Q1D) { sBc[ix] = Bc[ix]; }
|
||||
MFEM_FOREACH_THREAD_DIRECT(ix, x, (D1D - 1) * Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
sBo[ix] = Bo[ix];
|
||||
}
|
||||
}
|
||||
|
||||
for (int dim0 = 0; dim0 < VDIM; ++dim0)
|
||||
{
|
||||
MFEM_SYNC_THREAD;
|
||||
// sum factor to QQQ = Q_{dim0,dim1} B X_{dim1}
|
||||
for (int dim1 = 0; dim1 < VDIM; ++dim1)
|
||||
{
|
||||
const int D1Dz = (dim1 == 2) ? D1D - 1 : D1D;
|
||||
const int D1Dy = (dim1 == 1) ? D1D - 1 : D1D;
|
||||
const int D1Dx = (dim1 == 0) ? D1D - 1 : D1D;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, Q1D, D1Dy, D1Dz,
|
||||
Q1D, Q1D, Q1D)
|
||||
{
|
||||
sBc[q][d] = Bc(q,d);
|
||||
if (d < D1D-1)
|
||||
real_t u = 0;
|
||||
for (int dx = 0; dx < D1Dx; ++dx)
|
||||
{
|
||||
sBo[q][d] = Bo(q,d);
|
||||
real_t b;
|
||||
if (dim1 == 0)
|
||||
{
|
||||
b = BO(qx, dx);
|
||||
}
|
||||
else
|
||||
{
|
||||
b = BC(qx, dx);
|
||||
}
|
||||
u += X[dim1][tidz][dx + (dy + dz * D1Dy) * D1Dx] * b;
|
||||
}
|
||||
DDQ[dim1][tidz][dz][dy][qx] = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
for (int dim1 = 0; dim1 < VDIM; ++dim1)
|
||||
{
|
||||
const int D1Dz = (dim1 == 2) ? D1D - 1 : D1D;
|
||||
const int D1Dy = (dim1 == 1) ? D1D - 1 : D1D;
|
||||
// const int D1Dx = (dim1 == 0) ? D1D - 1 : D1D;
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, Q1D, Q1D, D1Dz,
|
||||
Q1D, Q1D, Q1D)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dy = 0; dy < D1Dy; ++dy)
|
||||
{
|
||||
real_t b;
|
||||
if (dim1 == 1)
|
||||
{
|
||||
b = BO(qy, dy);
|
||||
}
|
||||
else
|
||||
{
|
||||
b = BC(qy, dy);
|
||||
}
|
||||
u += DDQ[dim1][tidz][dz][dy][qx] * b;
|
||||
}
|
||||
DQQ[dim1][tidz][dz][qy][qx] = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
for (int dim1 = 0; dim1 < VDIM; ++dim1)
|
||||
{
|
||||
const int D1Dz = (dim1 == 2) ? D1D - 1 : D1D;
|
||||
// const int D1Dy = (dim1 == 1) ? D1D - 1 : D1D;
|
||||
// const int D1Dx = (dim1 == 0) ? D1D - 1 : D1D;
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D(qx, qy, qz, x, Q1D, Q1D, Q1D)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dz = 0; dz < D1Dz; ++dz)
|
||||
{
|
||||
real_t b;
|
||||
if (dim1 == 2)
|
||||
{
|
||||
b = BO(qz, dz);
|
||||
}
|
||||
else
|
||||
{
|
||||
b = BC(qz, dz);
|
||||
}
|
||||
u += DQQ[dim1][tidz][dz][qy][qx] * b;
|
||||
}
|
||||
// pa_data is row major
|
||||
int idx;
|
||||
if (symmetric)
|
||||
{
|
||||
int row;
|
||||
int col;
|
||||
if (dim0 > dim1)
|
||||
{
|
||||
row = dim1;
|
||||
col = dim0;
|
||||
}
|
||||
else
|
||||
{
|
||||
row = dim0;
|
||||
col = dim1;
|
||||
}
|
||||
idx = col + VDIM * row - row * (row + 1) / 2;
|
||||
}
|
||||
else
|
||||
{
|
||||
idx = dim0 * VDIM + dim1;
|
||||
}
|
||||
QQQ[dim1][tidz][qz][qy][qx] = op(qx, qy, qz, idx, e) * u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// sum factor back to Y
|
||||
// Assume bot and bct == bo^t and bc^t respectively (i.e. test ==
|
||||
// trial functions), skip loading them again.
|
||||
{
|
||||
const int D1Dz = (dim0 == 2) ? D1D - 1 : D1D;
|
||||
const int D1Dy = (dim0 == 1) ? D1D - 1 : D1D;
|
||||
const int D1Dx = (dim0 == 0) ? D1D - 1 : D1D;
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, D1Dz, Q1D, Q1D,
|
||||
Q1D, Q1D, Q1D)
|
||||
{
|
||||
for (int dim1 = 0; dim1 < VDIM; ++dim1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
real_t b = 0;
|
||||
if (dim0 == 2)
|
||||
{
|
||||
b = BO(qz, dz);
|
||||
}
|
||||
else
|
||||
{
|
||||
b = BC(qz, dz);
|
||||
}
|
||||
u += QQQ[dim1][tidz][qz][qy][qx] * b;
|
||||
}
|
||||
QQD[dim1][tidz][qy][qx][dz] = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, D1Dy, D1Dz, Q1D,
|
||||
Q1D, Q1D, Q1D)
|
||||
{
|
||||
for (int dim1 = 0; dim1 < VDIM; ++dim1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
real_t b;
|
||||
if (dim0 == 1)
|
||||
{
|
||||
b = BO(qy, dy);
|
||||
}
|
||||
else
|
||||
{
|
||||
b = BC(qy, dy);
|
||||
}
|
||||
u += QQD[dim1][tidz][qy][qx][dz] * b;
|
||||
}
|
||||
QDD[dim1][tidz][qx][dz][dy] = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D(dx, dy, dz, x, D1Dx, D1Dy, D1Dz)
|
||||
{
|
||||
int ix = dx + D1Dx * (dy + D1Dy * dz);
|
||||
real_t u = 0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
real_t b;
|
||||
if (dim0 == 0)
|
||||
{
|
||||
b = BO(qx, dx);
|
||||
}
|
||||
else
|
||||
{
|
||||
b = BC(qx, dx);
|
||||
}
|
||||
for (int dim1 = 0; dim1 < VDIM; ++dim1)
|
||||
{
|
||||
u += QDD[dim1][tidz][qx][dz][dy] * b;
|
||||
}
|
||||
}
|
||||
if constexpr (ACCUMULATE)
|
||||
{
|
||||
Y(ix + dim0 * offset, e) += u;
|
||||
}
|
||||
else
|
||||
{
|
||||
Y(ix + dim0 * offset, e) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int qz=0; qz < Q1D; ++qz)
|
||||
{
|
||||
int osc = 0;
|
||||
for (int c = 0; c < VDIM; ++c) // loop over x, y, z components
|
||||
{
|
||||
const int D1Dz = (c == 2) ? D1D - 1 : D1D;
|
||||
const int D1Dy = (c == 1) ? D1D - 1 : D1D;
|
||||
const int D1Dx = (c == 0) ? D1D - 1 : D1D;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz,z,D1Dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1Dy)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1Dx)
|
||||
{
|
||||
sX[dz][dy][dx] = X(dx + ((dy + (dz * D1Dy)) * D1Dx) + osc, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
if (tidz == qz)
|
||||
{
|
||||
for (int i=0; i<dataSize; ++i)
|
||||
{
|
||||
sop[i + (dataSize*tidx) + (dataSize*Q1D*tidy)] = op9[i];
|
||||
}
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
|
||||
for (int dz = 0; dz < D1Dz; ++dz)
|
||||
{
|
||||
const real_t wz = (c == 2) ? sBo[qz][dz] : sBc[qz][dz];
|
||||
for (int dy = 0; dy < D1Dy; ++dy)
|
||||
{
|
||||
const real_t wy = (c == 1) ? sBo[qy][dy] : sBc[qy][dy];
|
||||
for (int dx = 0; dx < D1Dx; ++dx)
|
||||
{
|
||||
const real_t t = sX[dz][dy][dx];
|
||||
const real_t wx = (c == 0) ? sBo[qx][dx] : sBc[qx][dx];
|
||||
u += t * wx * wy * wz;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mass[qy][qx][c] = u;
|
||||
} // qx
|
||||
} // qy
|
||||
} // tidz == qz
|
||||
|
||||
osc += D1Dx * D1Dy * D1Dz;
|
||||
MFEM_SYNC_THREAD;
|
||||
} // c
|
||||
|
||||
MFEM_SYNC_THREAD; // Sync mass[qy][qx][d] and sop
|
||||
|
||||
osc = 0;
|
||||
for (int c = 0; c < VDIM; ++c) // loop over x, y, z components
|
||||
{
|
||||
const int D1Dz = (c == 2) ? D1D - 1 : D1D;
|
||||
const int D1Dy = (c == 1) ? D1D - 1 : D1D;
|
||||
const int D1Dx = (c == 0) ? D1D - 1 : D1D;
|
||||
|
||||
real_t dxyz = 0.0;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz,z,D1Dz)
|
||||
{
|
||||
const real_t wz = (c == 2) ? sBo[qz][dz] : sBc[qz][dz];
|
||||
|
||||
MFEM_FOREACH_THREAD(dy,y,D1Dy)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1Dx)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t wy = (c == 1) ? sBo[qy][dy] : sBc[qy][dy];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const int os = (dataSize*qx) + (dataSize*Q1D*qy);
|
||||
const int id1 = os + ((c == 0) ? 0 : ((c == 1) ? (symmetric ? 1 : 3) :
|
||||
(symmetric ? 2 : 6))); // O11, O21, O31
|
||||
const int id2 = os + ((c == 0) ? 1 : ((c == 1) ? (symmetric ? 3 : 4) :
|
||||
(symmetric ? 4 : 7))); // O12, O22, O32
|
||||
const int id3 = os + ((c == 0) ? 2 : ((c == 1) ? (symmetric ? 4 : 5) :
|
||||
(symmetric ? 5 : 8))); // O13, O23, O33
|
||||
|
||||
const real_t m_c = (sop[id1] * mass[qy][qx][0]) + (sop[id2] * mass[qy][qx][1]) +
|
||||
(sop[id3] * mass[qy][qx][2]);
|
||||
|
||||
const real_t wx = (c == 0) ? sBo[qx][dx] : sBc[qx][dx];
|
||||
dxyz += m_c * wx * wy * wz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz,z,D1Dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1Dy)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1Dx)
|
||||
{
|
||||
Y(dx + ((dy + (dz * D1Dy)) * D1Dx) + osc, e) += dxyz;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
osc += D1Dx * D1Dy * D1Dz;
|
||||
} // c loop
|
||||
} // qz
|
||||
}); // end of element loop
|
||||
}
|
||||
|
||||
@@ -1805,13 +1889,17 @@ inline void SmemPACurlCurlApply3D(const int d1d,
|
||||
ForallWrap<3>(true, NE, device_kernel, host_kernel, Q1D, Q1D, Q1D);
|
||||
}
|
||||
|
||||
// PA H(curl)-L2 Assemble 2D kernel
|
||||
// PA H(curl)-L2 value Assemble 2D kernel
|
||||
void PAHcurlL2Setup2D(const int Q1D,
|
||||
const int NE,
|
||||
const Array<real_t> &w,
|
||||
Vector &coeff,
|
||||
Vector &op);
|
||||
|
||||
// PA H(curl)-L2 integral Assemble 2D kernel
|
||||
void PAHcurlL2IntSetup2D(const int Q1D, const int NE, const Array<real_t> &w,
|
||||
Vector &coeff, const Vector &detJ, Vector &op);
|
||||
|
||||
// PA H(curl)-L2 Assemble 3D kernel
|
||||
void PAHcurlL2Setup3D(const int NQ,
|
||||
const int coeffDim,
|
||||
|
||||
@@ -62,6 +62,30 @@ void PAHcurlHdivMassApply2D(const int D1D,
|
||||
const Vector &x_,
|
||||
Vector &y_);
|
||||
|
||||
/// H(curl) test, H(div) trial
|
||||
inline void
|
||||
PAHcurlHdivMassApply2D(const int NE, const bool, const bool scalarCoeff,
|
||||
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
|
||||
const Vector &op_, const Vector &x_, Vector &y_,
|
||||
const int D1D, const int D1Dtest, const int Q1D)
|
||||
{
|
||||
return PAHcurlHdivMassApply2D(D1D, D1Dtest, Q1D, NE, scalarCoeff, false,
|
||||
false, Bo_, Bc_, Bot_, Bct_, op_, x_, y_);
|
||||
}
|
||||
|
||||
/// H(div) test, H(curl) trial
|
||||
inline void
|
||||
PAHdivHcurlMassApply2D(const int NE, const bool, const bool scalarCoeff,
|
||||
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
|
||||
const Vector &op_, const Vector &x_, Vector &y_,
|
||||
const int D1D, const int D1Dtest, const int Q1D)
|
||||
{
|
||||
return PAHcurlHdivMassApply2D(D1D, D1Dtest, Q1D, NE, scalarCoeff, true,
|
||||
false, Bo_, Bc_, Bot_, Bct_, op_, x_, y_);
|
||||
}
|
||||
|
||||
// PA H(curl)-H(div) Mass Apply 3D kernel
|
||||
void PAHcurlHdivMassApply3D(const int D1D,
|
||||
const int D1Dtest,
|
||||
@@ -78,6 +102,30 @@ void PAHcurlHdivMassApply3D(const int D1D,
|
||||
const Vector &x_,
|
||||
Vector &y_);
|
||||
|
||||
/// H(curl) test, H(div) trial
|
||||
inline void
|
||||
PAHcurlHdivMassApply3D(const int NE, const bool, const bool scalarCoeff,
|
||||
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
|
||||
const Vector &op_, const Vector &x_, Vector &y_,
|
||||
const int D1D, const int D1Dtest, const int Q1D)
|
||||
{
|
||||
PAHcurlHdivMassApply3D(D1D, D1Dtest, Q1D, NE, scalarCoeff, false, false, Bo_,
|
||||
Bc_, Bot_, Bct_, op_, x_, y_);
|
||||
}
|
||||
|
||||
/// H(div) test, H(curl) trial
|
||||
inline void
|
||||
PAHdivHcurlMassApply3D(const int NE, const bool, const bool scalarCoeff,
|
||||
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
|
||||
const Vector &op_, const Vector &x_, Vector &y_,
|
||||
const int D1D, const int D1Dtest, const int Q1D)
|
||||
{
|
||||
PAHcurlHdivMassApply3D(D1D, D1Dtest, Q1D, NE, scalarCoeff, true, false, Bo_,
|
||||
Bc_, Bot_, Bct_, op_, x_, y_);
|
||||
}
|
||||
|
||||
// PA H(curl)-H(div) Curl Apply 3D kernel
|
||||
template<int T_D1D = 0, int T_D1D_TEST = 0, int T_Q1D = 0>
|
||||
inline void PAHcurlHdivApply3D(const int d1d,
|
||||
@@ -816,8 +864,656 @@ inline void PAHcurlHdivApplyTranspose3D(const int d1d,
|
||||
}); // end of element loop
|
||||
}
|
||||
|
||||
namespace curlinterp
|
||||
{
|
||||
constexpr int NBZ3D(int ndof_o, int nquad_o, int mdq)
|
||||
{
|
||||
if (ndof_o <= 0 || nquad_o <= 0)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
int ndof_c = ndof_o + 1;
|
||||
int nquad_c = nquad_o + 1;
|
||||
// z dimension is capped at 64 on nvidia and amd gpus
|
||||
int tmp =
|
||||
std::min((128 + mdq * mdq * (mdq - 1) - 1) / (mdq * mdq * (mdq - 1)), 64);
|
||||
int smem_req =
|
||||
sizeof(mfem::real_t) *
|
||||
((3 * ndof_c * ndof_c * ndof_o + 2 * 2 * mdq * mdq * mdq) * tmp +
|
||||
ndof_c * nquad_o + ndof_c * nquad_c + ndof_o * nquad_o);
|
||||
// assume GPU has at least 48k shared memory
|
||||
return std::max(std::min(tmp, (48 * 1024 + smem_req - 1) / smem_req), 1);
|
||||
}
|
||||
}
|
||||
|
||||
template <int T_NDOF_O, int T_NQUAD_O>
|
||||
void CurlInterpolatorApply3DSmem(const int ne, const int ndof_o,
|
||||
const int nquad_o, const Vector &pa,
|
||||
const Vector &x_, Vector &y_)
|
||||
{
|
||||
constexpr int mnd_o = T_NDOF_O ? T_NDOF_O : DofQuadLimits::HCURL_MAX_D1D - 1;
|
||||
constexpr int mnq_o =
|
||||
T_NQUAD_O ? T_NQUAD_O : DofQuadLimits::HDIV_MAX_D1D - 1;
|
||||
constexpr int mndq = std::max(mnd_o + 1, mnq_o + 1);
|
||||
constexpr int tbatch = curlinterp::NBZ3D(T_NDOF_O, T_NQUAD_O, mndq);
|
||||
MFEM_VERIFY(ndof_o <= mnd_o, "Error: H(curl) order larger than supported");
|
||||
MFEM_VERIFY(nquad_o <= mnq_o, "Error: H(div) order larger than supported");
|
||||
int mnq = std::max(ndof_o + 1, nquad_o + 1);
|
||||
auto pa_data = pa.Read();
|
||||
auto x_d = x_.Read();
|
||||
auto y_d = y_.ReadWrite();
|
||||
mfem::forall_2D_batch<mndq * mndq * (mndq - 1) * tbatch>(
|
||||
ne, mnq * mnq * (mnq - 1), 1, tbatch, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MND_O =
|
||||
T_NDOF_O ? T_NDOF_O : DofQuadLimits::HCURL_MAX_D1D - 1;
|
||||
constexpr int MNQ_O =
|
||||
T_NQUAD_O ? T_NQUAD_O : DofQuadLimits::HDIV_MAX_D1D - 1;
|
||||
constexpr int MNDQ = std::max(MND_O + 1, MNQ_O + 1);
|
||||
#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)
|
||||
constexpr int nbz = curlinterp::NBZ3D(T_NDOF_O, T_NQUAD_O, MNDQ);
|
||||
int tidz = MFEM_THREAD_ID(z);
|
||||
// Make mnq a local variable since capturing would result in different
|
||||
// captures between host/device versions, and spuriously fails
|
||||
int mnq = std::max(ndof_o + 1, nquad_o + 1);
|
||||
#else
|
||||
constexpr int nbz = 1;
|
||||
constexpr int tidz = 0;
|
||||
#endif
|
||||
const int NDOF_O = T_NDOF_O ? T_NDOF_O : ndof_o;
|
||||
const int NQUAD_O = T_NQUAD_O ? T_NQUAD_O : nquad_o;
|
||||
const int NDOF_C = NDOF_O + 1;
|
||||
const int NQUAD_C = NQUAD_O + 1;
|
||||
MFEM_SHARED real_t
|
||||
sBG[(MND_O + 1) * MNQ_O + (MND_O + 1) * (MNQ_O + 1) + MND_O * MNQ_O];
|
||||
auto X_ = Reshape(x_d, 3 * NDOF_C * NDOF_C * NDOF_O, ne);
|
||||
auto Y = Reshape(y_d, 3 * NQUAD_C * NQUAD_O * NQUAD_O, ne);
|
||||
auto Gco = Reshape(sBG, NQUAD_O, NDOF_C);
|
||||
auto Bcc = Reshape(sBG + NDOF_C * NQUAD_O, NQUAD_C, NDOF_C);
|
||||
auto Boo =
|
||||
Reshape(sBG + NDOF_C * NQUAD_O + NDOF_C * NQUAD_C, NQUAD_O, NDOF_O);
|
||||
MFEM_SHARED real_t X[3][nbz][MND_O * (MND_O + 1) * (MND_O + 1)];
|
||||
MFEM_SHARED real_t sm0[nbz * 2 * MNDQ * MNDQ * MNDQ];
|
||||
MFEM_SHARED real_t sm1[nbz * 2 * MNDQ * MNDQ * MNDQ];
|
||||
|
||||
// shapes of buffers always use MNDQ to mitigate shared memory bank
|
||||
// conflicts
|
||||
real_t(*DDQ)[nbz][MNDQ][MNDQ][MNDQ] =
|
||||
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm0);
|
||||
real_t(*DQQ)[nbz][MNDQ][MNDQ][MNDQ] =
|
||||
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm1);
|
||||
real_t(*QQQ)[nbz][MNDQ][MNDQ][MNDQ] =
|
||||
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm0);
|
||||
const int offset = NDOF_O * NDOF_C * NDOF_C;
|
||||
const int offsetq = NQUAD_C * NQUAD_O * NQUAD_O;
|
||||
MFEM_FOREACH_THREAD_DIRECT(ix, x, offset)
|
||||
{
|
||||
for (int dim = 0; dim < 3; ++dim)
|
||||
{
|
||||
X[dim][tidz][ix] = X_(ix + dim * offset, e);
|
||||
}
|
||||
}
|
||||
// load basis functions data
|
||||
if (tidz == 0)
|
||||
{
|
||||
auto npts = NDOF_C * NQUAD_O + NDOF_C * NQUAD_C + NDOF_O * NQUAD_O;
|
||||
MFEM_FOREACH_THREAD(ix, x, npts) { sBG[ix] = pa_data[ix]; }
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// x: Vz Bcc Gco Boo - Vy Bcc Boo Gco
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_C, NDOF_C,
|
||||
NDOF_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dx = 0; dx < NDOF_C; ++dx)
|
||||
{
|
||||
u += X[2][tidz][dx + (dy + dz * NDOF_C) * NDOF_C] * Bcc(qx, dx);
|
||||
}
|
||||
DDQ[0][tidz][dz][dy][qx] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_C, NDOF_O,
|
||||
NDOF_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dx = 0; dx < NDOF_C; ++dx)
|
||||
{
|
||||
u += X[1][tidz][dx + (dy + dz * NDOF_O) * NDOF_C] * Bcc(qx, dx);
|
||||
}
|
||||
DDQ[1][tidz][dz][dy][qx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_C, NQUAD_O,
|
||||
NDOF_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dy = 0; dy < NDOF_C; ++dy)
|
||||
{
|
||||
u += DDQ[0][tidz][dz][dy][qx] * Gco(qy, dy);
|
||||
}
|
||||
DQQ[0][tidz][dz][qy][qx] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_C, NQUAD_O,
|
||||
NDOF_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dy = 0; dy < NDOF_O; ++dy)
|
||||
{
|
||||
u += DDQ[1][tidz][dz][dy][qx] * Boo(qy, dy);
|
||||
}
|
||||
DQQ[1][tidz][dz][qy][qx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_C, NQUAD_O,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dz = 0; dz < NDOF_O; ++dz)
|
||||
{
|
||||
u += DQQ[0][tidz][dz][qy][qx] * Boo(qz, dz);
|
||||
}
|
||||
QQQ[0][tidz][qz][qy][qx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_C, NQUAD_O,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dz = 0; dz < NDOF_C; ++dz)
|
||||
{
|
||||
u += DQQ[1][tidz][dz][qy][qx] * Gco(qz, dz);
|
||||
}
|
||||
Y(qx + (qy + qz * NQUAD_O) * NQUAD_C, e) =
|
||||
QQQ[0][tidz][qz][qy][qx] - u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// y: Vx Boo Bcc Gco - Vz Gco Bcc Boo
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_O, NDOF_C,
|
||||
NDOF_C, mnq - 1, mnq, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dx = 0; dx < NDOF_O; ++dx)
|
||||
{
|
||||
u += X[0][tidz][dx + (dy + dz * NDOF_C) * NDOF_O] * Boo(qx, dx);
|
||||
}
|
||||
DDQ[0][tidz][dz][dy][qx] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_O, NDOF_C,
|
||||
NDOF_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dx = 0; dx < NDOF_C; ++dx)
|
||||
{
|
||||
u += X[2][tidz][dx + (dy + dz * NDOF_C) * NDOF_C] * Gco(qx, dx);
|
||||
}
|
||||
DDQ[1][tidz][dz][dy][qx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_O, NQUAD_C,
|
||||
NDOF_C, mnq - 1, mnq, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dy = 0; dy < NDOF_C; ++dy)
|
||||
{
|
||||
u += DDQ[0][tidz][dz][dy][qx] * Bcc(qy, dy);
|
||||
}
|
||||
DQQ[0][tidz][dz][qy][qx] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_O, NQUAD_C,
|
||||
NDOF_O, mnq - 1, mnq, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dy = 0; dy < NDOF_C; ++dy)
|
||||
{
|
||||
u += DDQ[1][tidz][dz][dy][qx] * Bcc(qy, dy);
|
||||
}
|
||||
DQQ[1][tidz][dz][qy][qx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_O, NQUAD_C,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dz = 0; dz < NDOF_C; ++dz)
|
||||
{
|
||||
u += DQQ[0][tidz][dz][qy][qx] * Gco(qz, dz);
|
||||
}
|
||||
QQQ[0][tidz][qz][qy][qx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_O, NQUAD_C,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dz = 0; dz < NDOF_O; ++dz)
|
||||
{
|
||||
u += DQQ[1][tidz][dz][qy][qx] * Boo(qz, dz);
|
||||
}
|
||||
Y(qx + (qy + qz * NQUAD_C) * NQUAD_O + offsetq, e) =
|
||||
QQQ[0][tidz][qz][qy][qx] - u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// z: Vy Gco Boo Bcc - Vx Boo Gco Bcc
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_O, NDOF_O,
|
||||
NDOF_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dx = 0; dx < NDOF_C; ++dx)
|
||||
{
|
||||
u += X[1][tidz][dx + (dy + dz * NDOF_O) * NDOF_C] * Gco(qx, dx);
|
||||
}
|
||||
DDQ[0][tidz][dz][dy][qx] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_O, NDOF_C,
|
||||
NDOF_C, mnq - 1, mnq, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dx = 0; dx < NDOF_O; ++dx)
|
||||
{
|
||||
u += X[0][tidz][dx + (dy + dz * NDOF_C) * NDOF_O] * Boo(qx, dx);
|
||||
}
|
||||
DDQ[1][tidz][dz][dy][qx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_O, NQUAD_O,
|
||||
NDOF_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dy = 0; dy < NDOF_O; ++dy)
|
||||
{
|
||||
u += DDQ[0][tidz][dz][dy][qx] * Boo(qy, dy);
|
||||
}
|
||||
DQQ[0][tidz][dz][qy][qx] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_O, NQUAD_O,
|
||||
NDOF_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dy = 0; dy < NDOF_C; ++dy)
|
||||
{
|
||||
u += DDQ[1][tidz][dz][dy][qx] * Gco(qy, dy);
|
||||
}
|
||||
DQQ[1][tidz][dz][qy][qx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_O, NQUAD_O,
|
||||
NQUAD_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dz = 0; dz < NDOF_C; ++dz)
|
||||
{
|
||||
u += DQQ[0][tidz][dz][qy][qx] * Bcc(qz, dz);
|
||||
}
|
||||
QQQ[0][tidz][qz][qy][qx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_O, NQUAD_O,
|
||||
NQUAD_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int dz = 0; dz < NDOF_C; ++dz)
|
||||
{
|
||||
u += DQQ[1][tidz][dz][qy][qx] * Bcc(qz, dz);
|
||||
}
|
||||
Y(qx + (qy + qz * NQUAD_O) * NQUAD_O + 2 * offsetq, e) =
|
||||
QQQ[0][tidz][qz][qy][qx] - u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
});
|
||||
}
|
||||
|
||||
template <int T_NDOF_O, int T_NQUAD_O>
|
||||
void CurlInterpolatorTApply3DSmem(const int ne, const int ndof_o,
|
||||
const int nquad_o, const Vector &pa,
|
||||
const Vector &x_, Vector &y_)
|
||||
{
|
||||
constexpr int mnd_o = T_NDOF_O ? T_NDOF_O : DofQuadLimits::HCURL_MAX_D1D - 1;
|
||||
constexpr int mnq_o =
|
||||
T_NQUAD_O ? T_NQUAD_O : DofQuadLimits::HDIV_MAX_D1D - 1;
|
||||
constexpr int mndq = std::max(mnd_o + 1, mnq_o + 1);
|
||||
constexpr int tbatch = curlinterp::NBZ3D(T_NDOF_O, T_NQUAD_O, mndq);
|
||||
MFEM_VERIFY(ndof_o <= mnd_o, "Error: H(curl) order larger than supported");
|
||||
MFEM_VERIFY(nquad_o <= mnq_o, "Error: H(div) order larger than supported");
|
||||
int mnq = std::max(ndof_o + 1, nquad_o + 1);
|
||||
auto pa_data = pa.Read();
|
||||
auto x_d = x_.Read();
|
||||
auto y_d = y_.ReadWrite();
|
||||
mfem::forall_2D_batch<mndq * mndq * (mndq - 1) * tbatch>(
|
||||
ne, mnq * mnq * (mnq - 1), 1, tbatch, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MND_O =
|
||||
T_NDOF_O ? T_NDOF_O : DofQuadLimits::HCURL_MAX_D1D - 1;
|
||||
constexpr int MNQ_O =
|
||||
T_NQUAD_O ? T_NQUAD_O : DofQuadLimits::HDIV_MAX_D1D - 1;
|
||||
constexpr int MNDQ = std::max(MND_O + 1, MNQ_O + 1);
|
||||
#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)
|
||||
constexpr int nbz = curlinterp::NBZ3D(T_NDOF_O, T_NQUAD_O, MNDQ);
|
||||
int tidz = MFEM_THREAD_ID(z);
|
||||
// Make mnq a local variable since capturing would result in different
|
||||
// captures between host/device versions, and spuriously fails
|
||||
int mnq = std::max(ndof_o + 1, nquad_o + 1);
|
||||
#else
|
||||
constexpr int nbz = 1;
|
||||
constexpr int tidz = 0;
|
||||
#endif
|
||||
const int NDOF_O = T_NDOF_O ? T_NDOF_O : ndof_o;
|
||||
const int NQUAD_O = T_NQUAD_O ? T_NQUAD_O : nquad_o;
|
||||
const int NDOF_C = NDOF_O + 1;
|
||||
const int NQUAD_C = NQUAD_O + 1;
|
||||
MFEM_SHARED real_t
|
||||
sBG[(MND_O + 1) * MNQ_O + (MND_O + 1) * (MNQ_O + 1) + MND_O * MNQ_O];
|
||||
auto X_ = Reshape(x_d, 3 * NQUAD_C * NQUAD_O * NQUAD_O, ne);
|
||||
auto Y = Reshape(y_d, 3 * NDOF_C * NDOF_C * NDOF_O, ne);
|
||||
auto Gco = Reshape(sBG, NQUAD_O, NDOF_C);
|
||||
auto Bcc = Reshape(sBG + NDOF_C * NQUAD_O, NQUAD_C, NDOF_C);
|
||||
auto Boo =
|
||||
Reshape(sBG + NDOF_C * NQUAD_O + NDOF_C * NQUAD_C, NQUAD_O, NDOF_O);
|
||||
MFEM_SHARED real_t X[3][nbz][MNQ_O * MNQ_O * (MNQ_O + 1)];
|
||||
MFEM_SHARED real_t sm0[nbz * 2 * MNDQ * MNDQ * MNDQ];
|
||||
MFEM_SHARED real_t sm1[nbz * 2 * MNDQ * MNDQ * MNDQ];
|
||||
|
||||
// shapes of buffers always use MNDQ to mitigate shared memory bank
|
||||
// conflicts
|
||||
real_t(*QQD)[nbz][MNDQ][MNDQ][MNDQ] =
|
||||
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm0);
|
||||
real_t(*QDD)[nbz][MNDQ][MNDQ][MNDQ] =
|
||||
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm1);
|
||||
real_t(*DDD)[nbz][MNDQ][MNDQ][MNDQ] =
|
||||
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm0);
|
||||
const int offset = NDOF_O * NDOF_C * NDOF_C;
|
||||
const int offsetq = NQUAD_C * NQUAD_O * NQUAD_O;
|
||||
MFEM_FOREACH_THREAD_DIRECT(ix, x, offsetq)
|
||||
{
|
||||
for (int dim = 0; dim < 3; ++dim)
|
||||
{
|
||||
X[dim][tidz][ix] = X_(ix + dim * offsetq, e);
|
||||
}
|
||||
}
|
||||
// load basis functions data
|
||||
if (tidz == 0)
|
||||
{
|
||||
auto npts = NDOF_C * NQUAD_O + NDOF_C * NQUAD_C + NDOF_O * NQUAD_O;
|
||||
MFEM_FOREACH_THREAD(ix, x, npts) { sBG[ix] = pa_data[ix]; }
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// x: Vy Boo Bcc Gco - Vz Boo Gco Bcc
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_C, NQUAD_O,
|
||||
NQUAD_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qz = 0; qz < NQUAD_O; ++qz)
|
||||
{
|
||||
u += X[1][tidz][qx + (qy + qz * NQUAD_C) * NQUAD_O] * Gco(qz, dz);
|
||||
}
|
||||
QQD[0][tidz][qy][qx][dz] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_C, NQUAD_O,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qz = 0; qz < NQUAD_C; ++qz)
|
||||
{
|
||||
u += X[2][tidz][qx + (qy + qz * NQUAD_O) * NQUAD_O] * Bcc(qz, dz);
|
||||
}
|
||||
QQD[1][tidz][qy][qx][dz] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_C, NDOF_C,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qy = 0; qy < NQUAD_C; ++qy)
|
||||
{
|
||||
u += QQD[0][tidz][qy][qx][dz] * Bcc(qy, dy);
|
||||
}
|
||||
QDD[0][tidz][qx][dz][dy] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_C, NDOF_C,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qy = 0; qy < NQUAD_O; ++qy)
|
||||
{
|
||||
u += QQD[1][tidz][qy][qx][dz] * Gco(qy, dy);
|
||||
}
|
||||
QDD[1][tidz][qx][dz][dy] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_O, NDOF_C,
|
||||
NDOF_C, mnq - 1, mnq, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qx = 0; qx < NQUAD_O; ++qx)
|
||||
{
|
||||
u += QDD[0][tidz][qx][dz][dy] * Boo(qx, dx);
|
||||
}
|
||||
DDD[0][tidz][dz][dy][dx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_O, NDOF_C,
|
||||
NDOF_C, mnq - 1, mnq, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qx = 0; qx < NQUAD_O; ++qx)
|
||||
{
|
||||
u += QDD[1][tidz][qx][dz][dy] * Boo(qx, dx);
|
||||
}
|
||||
Y(dx + (dy + dz * NDOF_C) * NDOF_O, e) =
|
||||
DDD[0][tidz][dz][dy][dx] - u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// y: Vz Gco Boo Bcc - Vx Bcc Boo Gco
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_C, NQUAD_O,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qz = 0; qz < NQUAD_C; ++qz)
|
||||
{
|
||||
u += X[2][tidz][qx + (qy + qz * NQUAD_O) * NQUAD_O] * Bcc(qz, dz);
|
||||
}
|
||||
QQD[0][tidz][qy][qx][dz] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_C, NQUAD_C,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qz = 0; qz < NQUAD_O; ++qz)
|
||||
{
|
||||
u += X[0][tidz][qx + (qy + qz * NQUAD_O) * NQUAD_C] * Gco(qz, dz);
|
||||
}
|
||||
QQD[1][tidz][qy][qx][dz] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_O, NDOF_C,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qy = 0; qy < NQUAD_O; ++qy)
|
||||
{
|
||||
u += QQD[0][tidz][qy][qx][dz] * Boo(qy, dy);
|
||||
}
|
||||
QDD[0][tidz][qx][dz][dy] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_O, NDOF_C,
|
||||
NQUAD_C, mnq - 1, mnq, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qy = 0; qy < NQUAD_O; ++qy)
|
||||
{
|
||||
u += QQD[1][tidz][qy][qx][dz] * Boo(qy, dy);
|
||||
}
|
||||
QDD[1][tidz][qx][dz][dy] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_C, NDOF_O,
|
||||
NDOF_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qx = 0; qx < NQUAD_O; ++qx)
|
||||
{
|
||||
u += QDD[0][tidz][qx][dz][dy] * Gco(qx, dx);
|
||||
}
|
||||
DDD[0][tidz][dz][dy][dx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_C, NDOF_O,
|
||||
NDOF_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qx = 0; qx < NQUAD_C; ++qx)
|
||||
{
|
||||
u += QDD[1][tidz][qx][dz][dy] * Bcc(qx, dx);
|
||||
}
|
||||
Y(dx + (dy + dz * NDOF_O) * NDOF_C + offset, e) =
|
||||
DDD[0][tidz][dz][dy][dx] - u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// z: Vx Bcc Gco Boo - Vy Gco Bcc Boo
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_O, NQUAD_C,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qz = 0; qz < NQUAD_O; ++qz)
|
||||
{
|
||||
u += X[0][tidz][qx + (qy + qz * NQUAD_O) * NQUAD_C] * Boo(qz, dz);
|
||||
}
|
||||
QQD[0][tidz][qy][qx][dz] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_O, NQUAD_O,
|
||||
NQUAD_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qz = 0; qz < NQUAD_O; ++qz)
|
||||
{
|
||||
u += X[1][tidz][qx + (qy + qz * NQUAD_C) * NQUAD_O] * Boo(qz, dz);
|
||||
}
|
||||
QQD[1][tidz][qy][qx][dz] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_C, NDOF_O,
|
||||
NQUAD_C, mnq, mnq - 1, mnq)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qy = 0; qy < NQUAD_O; ++qy)
|
||||
{
|
||||
u += QQD[0][tidz][qy][qx][dz] * Gco(qy, dy);
|
||||
}
|
||||
QDD[0][tidz][qx][dz][dy] = u;
|
||||
}
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_C, NDOF_O,
|
||||
NQUAD_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qy = 0; qy < NQUAD_C; ++qy)
|
||||
{
|
||||
u += QQD[1][tidz][qy][qx][dz] * Bcc(qy, dy);
|
||||
}
|
||||
QDD[1][tidz][qx][dz][dy] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_C, NDOF_C,
|
||||
NDOF_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qx = 0; qx < NQUAD_C; ++qx)
|
||||
{
|
||||
u += QDD[0][tidz][qx][dz][dy] * Bcc(qx, dx);
|
||||
}
|
||||
DDD[0][tidz][dz][dy][dx] = u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// threads assigned to mitigate bank conflicts
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_C, NDOF_C,
|
||||
NDOF_O, mnq, mnq, mnq - 1)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qx = 0; qx < NQUAD_O; ++qx)
|
||||
{
|
||||
u += QDD[1][tidz][qx][dz][dy] * Gco(qx, dx);
|
||||
}
|
||||
Y(dx + (dy + dz * NDOF_C) * NDOF_C + 2 * offset, e) =
|
||||
DDD[0][tidz][dz][dy][dx] - u;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace internal
|
||||
|
||||
template <int DIM, int NDOF_O, int NQUAD_O>
|
||||
CurlInterpolator::ApplyKernelType
|
||||
CurlInterpolator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 3)
|
||||
{
|
||||
return internal::CurlInterpolatorApply3DSmem<NDOF_O, NQUAD_O>;
|
||||
}
|
||||
MFEM_ABORT("Bad dimension!");
|
||||
}
|
||||
|
||||
template <int DIM, int NDOF_O, int NQUAD_O>
|
||||
CurlInterpolator::ApplyKernelType
|
||||
CurlInterpolator::ApplyTPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 3)
|
||||
{
|
||||
return internal::CurlInterpolatorTApply3DSmem<NDOF_O, NQUAD_O>;
|
||||
}
|
||||
MFEM_ABORT("Bad dimension!");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
@@ -294,61 +294,14 @@ void PAHdivMassAssembleDiagonal3D(const int D1D,
|
||||
}); // end of element loop
|
||||
}
|
||||
|
||||
void PAHdivMassApply(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &Bo,
|
||||
const Array<real_t> &Bc,
|
||||
const Array<real_t> &Bot,
|
||||
const Array<real_t> &Bct,
|
||||
const Vector &op,
|
||||
const Vector &x,
|
||||
Vector &y)
|
||||
{
|
||||
const int id = (D1D << 4) | Q1D;
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x22: return SmemPAHdivMassApply2D<2,2>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
case 0x33: return SmemPAHdivMassApply2D<3,3>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
case 0x44: return SmemPAHdivMassApply2D<4,4>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
case 0x55: return SmemPAHdivMassApply2D<5,5>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
default: // fallback
|
||||
return PAHdivMassApply2D(D1D,Q1D,NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
}
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x23: return SmemPAHdivMassApply3D<2,3>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
case 0x34: return SmemPAHdivMassApply3D<3,4>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
case 0x45: return SmemPAHdivMassApply3D<4,5>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
case 0x56: return SmemPAHdivMassApply3D<5,6>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
case 0x67: return SmemPAHdivMassApply3D<6,7>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
case 0x78: return SmemPAHdivMassApply3D<7,8>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
default: // fallback
|
||||
return PAHdivMassApply3D(D1D,Q1D,NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void PAHdivMassApply2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_,
|
||||
const Array<real_t> &Bct_,
|
||||
const Vector &op_,
|
||||
const Vector &x_,
|
||||
Vector &y_)
|
||||
void PAHdivMassApply2D(const int NE, const bool symmetric, const bool,
|
||||
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
|
||||
const Vector &op_, const Vector &x_, Vector &y_,
|
||||
const int D1D, const int TestD1D, const int Q1D)
|
||||
{
|
||||
MFEM_VERIFY(D1D == TestD1D,
|
||||
"Trial and test spaces must have same number of dofs");
|
||||
auto Bo = Reshape(Bo_.Read(), Q1D, D1D-1);
|
||||
auto Bc = Reshape(Bc_.Read(), Q1D, D1D);
|
||||
auto Bot = Reshape(Bot_.Read(), D1D-1, Q1D);
|
||||
@@ -468,18 +421,14 @@ void PAHdivMassApply2D(const int D1D,
|
||||
}); // end of element loop
|
||||
}
|
||||
|
||||
void PAHdivMassApply3D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_,
|
||||
const Array<real_t> &Bct_,
|
||||
const Vector &op_,
|
||||
const Vector &x_,
|
||||
Vector &y_)
|
||||
void PAHdivMassApply3D(const int NE, const bool symmetric, const bool,
|
||||
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
|
||||
const Vector &op_, const Vector &x_, Vector &y_,
|
||||
const int D1D, const int TestD1D, const int Q1D)
|
||||
{
|
||||
MFEM_VERIFY(D1D == TestD1D,
|
||||
"Trial and test spaces must have same number of dofs");
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
|
||||
"Error: D1D > HDIV_MAX_D1D");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
|
||||
|
||||
@@ -66,58 +66,29 @@ void PAHdivMassAssembleDiagonal3D(const int D1D,
|
||||
const Vector &op_,
|
||||
Vector &diag_);
|
||||
|
||||
void PAHdivMassApply(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &Bo,
|
||||
const Array<real_t> &Bc,
|
||||
const Array<real_t> &Bot,
|
||||
const Array<real_t> &Bct,
|
||||
const Vector &op,
|
||||
const Vector &x,
|
||||
Vector &y);
|
||||
|
||||
// PA H(div) Mass Apply 2D kernel
|
||||
void PAHdivMassApply2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_,
|
||||
const Array<real_t> &Bct_,
|
||||
const Vector &op_,
|
||||
const Vector &x_,
|
||||
Vector &y_);
|
||||
void PAHdivMassApply2D(const int NE, const bool symmetric,
|
||||
const bool scalar_coeff, const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Bc_, const Array<real_t> &Bot_,
|
||||
const Array<real_t> &Bct_, const Vector &op_,
|
||||
const Vector &x_, Vector &y_, const int D1D,
|
||||
const int TestD1D, const int Q1D);
|
||||
|
||||
// PA H(div) Mass Apply 3D kernel
|
||||
void PAHdivMassApply3D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_,
|
||||
const Array<real_t> &Bct_,
|
||||
const Vector &op_,
|
||||
const Vector &x_,
|
||||
Vector &y_);
|
||||
void PAHdivMassApply3D(const int NE, const bool symmetric,
|
||||
const bool scalar_coeff, const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Bc_, const Array<real_t> &Bot_,
|
||||
const Array<real_t> &Bct_, const Vector &op_,
|
||||
const Vector &x_, Vector &y_, const int D1D,
|
||||
const int TestD1D, const int Q1D);
|
||||
|
||||
// Shared memory PA H(div) Mass Apply 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void SmemPAHdivMassApply2D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_,
|
||||
const Array<real_t> &Bct_,
|
||||
const Vector &op_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void SmemPAHdivMassApply2D(
|
||||
const int NE, const bool symmetric, const bool, const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Bc_, const Array<real_t> &Bot_,
|
||||
const Array<real_t> &Bct_, const Vector &op_, const Vector &x_, Vector &y_,
|
||||
const int d1d = 0, const int = 0, const int q1d = 0)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(Bot_);
|
||||
MFEM_CONTRACT_VAR(Bct_);
|
||||
@@ -280,18 +251,13 @@ inline void SmemPAHdivMassApply2D(const int NE,
|
||||
}
|
||||
|
||||
// Shared memory PA H(div) Mass Apply 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void SmemPAHdivMassApply3D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_,
|
||||
const Array<real_t> &Bct_,
|
||||
const Vector &op_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void
|
||||
SmemPAHdivMassApply3D(const int NE, const bool symmetric, const bool,
|
||||
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
|
||||
const Vector &op_, const Vector &x_, Vector &y_,
|
||||
const int d1d = 0, const int = 0, const int q1d = 0)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(Bot_);
|
||||
MFEM_CONTRACT_VAR(Bct_);
|
||||
|
||||
@@ -14,9 +14,218 @@
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
|
||||
#include "bilininteg_hcurlhdiv_kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace
|
||||
{
|
||||
|
||||
void PAHcurlApplyCurl2D(const int c_dofs1D,
|
||||
const int o_dofs1D,
|
||||
const int NE,
|
||||
const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Gc_,
|
||||
const Vector &x_,
|
||||
Vector &y_)
|
||||
{
|
||||
auto Bo = Reshape(Bo_.Read(), o_dofs1D, o_dofs1D);
|
||||
auto Gc = Reshape(Gc_.Read(), o_dofs1D, c_dofs1D);
|
||||
auto X = Reshape(x_.Read(), 2 * c_dofs1D * o_dofs1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), o_dofs1D, o_dofs1D, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
for (int iy = 0; iy < c_dofs1D; ++iy)
|
||||
{
|
||||
for (int ix = 0; ix < o_dofs1D; ++ix)
|
||||
{
|
||||
const real_t xv = X(ix + iy * o_dofs1D, e);
|
||||
for (int oy = 0; oy < o_dofs1D; ++oy)
|
||||
{
|
||||
const real_t gy = Gc(oy, iy);
|
||||
for (int ox = 0; ox < o_dofs1D; ++ox)
|
||||
{
|
||||
Y(ox, oy, e) -= Bo(ox, ix) * gy * xv;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const int y_nd = c_dofs1D * o_dofs1D;
|
||||
for (int iy = 0; iy < o_dofs1D; ++iy)
|
||||
{
|
||||
for (int ix = 0; ix < c_dofs1D; ++ix)
|
||||
{
|
||||
const real_t xv = X(y_nd + ix + iy * c_dofs1D, e);
|
||||
for (int oy = 0; oy < o_dofs1D; ++oy)
|
||||
{
|
||||
const real_t by = Bo(oy, iy);
|
||||
for (int ox = 0; ox < o_dofs1D; ++ox)
|
||||
{
|
||||
Y(ox, oy, e) += Gc(ox, ix) * by * xv;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void PAHcurlApplyCurlTranspose2D(const int c_dofs1D,
|
||||
const int o_dofs1D,
|
||||
const int NE,
|
||||
const Array<real_t> &Bo_,
|
||||
const Array<real_t> &Gc_,
|
||||
const Vector &x_,
|
||||
Vector &y_)
|
||||
{
|
||||
auto Bo = Reshape(Bo_.Read(), o_dofs1D, o_dofs1D);
|
||||
auto Gc = Reshape(Gc_.Read(), o_dofs1D, c_dofs1D);
|
||||
auto X = Reshape(x_.Read(), o_dofs1D, o_dofs1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), 2 * c_dofs1D * o_dofs1D, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
for (int dy = 0; dy < c_dofs1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < o_dofs1D; ++dx)
|
||||
{
|
||||
real_t sum = 0.0;
|
||||
for (int oy = 0; oy < o_dofs1D; ++oy)
|
||||
{
|
||||
const real_t gy = Gc(oy, dy);
|
||||
for (int ox = 0; ox < o_dofs1D; ++ox)
|
||||
{
|
||||
sum -= Bo(ox, dx) * gy * X(ox, oy, e);
|
||||
}
|
||||
}
|
||||
Y(dx + dy * o_dofs1D, e) += sum;
|
||||
}
|
||||
}
|
||||
|
||||
const int y_nd = c_dofs1D * o_dofs1D;
|
||||
for (int dy = 0; dy < o_dofs1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < c_dofs1D; ++dx)
|
||||
{
|
||||
real_t sum = 0.0;
|
||||
for (int oy = 0; oy < o_dofs1D; ++oy)
|
||||
{
|
||||
const real_t by = Bo(oy, dy);
|
||||
for (int ox = 0; ox < o_dofs1D; ++ox)
|
||||
{
|
||||
sum += Gc(ox, dx) * by * X(ox, oy, e);
|
||||
}
|
||||
}
|
||||
Y(y_nd + dx + dy * c_dofs1D, e) += sum;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void PAHdivApplyCurl2D(const int c_dofs1D,
|
||||
const int o_dofs1D,
|
||||
const int NE,
|
||||
const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Gc_,
|
||||
const Vector &x_,
|
||||
Vector &y_)
|
||||
{
|
||||
auto Bc = Reshape(Bc_.Read(), c_dofs1D, c_dofs1D);
|
||||
auto Gc = Reshape(Gc_.Read(), o_dofs1D, c_dofs1D);
|
||||
auto X = Reshape(x_.Read(), c_dofs1D, c_dofs1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), 2 * c_dofs1D * o_dofs1D, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
for (int iy = 0; iy < c_dofs1D; ++iy)
|
||||
{
|
||||
for (int ix = 0; ix < c_dofs1D; ++ix)
|
||||
{
|
||||
const real_t xv = X(ix, iy, e);
|
||||
for (int oy = 0; oy < o_dofs1D; ++oy)
|
||||
{
|
||||
const real_t gy = Gc(oy, iy);
|
||||
for (int ox = 0; ox < c_dofs1D; ++ox)
|
||||
{
|
||||
Y(ox + oy * c_dofs1D, e) += Bc(ox, ix) * gy * xv;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const int y_nd = c_dofs1D * o_dofs1D;
|
||||
for (int iy = 0; iy < c_dofs1D; ++iy)
|
||||
{
|
||||
for (int ix = 0; ix < c_dofs1D; ++ix)
|
||||
{
|
||||
const real_t xv = X(ix, iy, e);
|
||||
for (int oy = 0; oy < c_dofs1D; ++oy)
|
||||
{
|
||||
const real_t by = Bc(oy, iy);
|
||||
for (int ox = 0; ox < o_dofs1D; ++ox)
|
||||
{
|
||||
Y(y_nd + ox + oy * o_dofs1D, e) -= Gc(ox, ix) * by * xv;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void PAHdivApplyCurlTranspose2D(const int c_dofs1D,
|
||||
const int o_dofs1D,
|
||||
const int NE,
|
||||
const Array<real_t> &Bc_,
|
||||
const Array<real_t> &Gc_,
|
||||
const Vector &x_,
|
||||
Vector &y_)
|
||||
{
|
||||
auto Bc = Reshape(Bc_.Read(), c_dofs1D, c_dofs1D);
|
||||
auto Gc = Reshape(Gc_.Read(), o_dofs1D, c_dofs1D);
|
||||
auto X = Reshape(x_.Read(), 2 * c_dofs1D * o_dofs1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), c_dofs1D, c_dofs1D, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
for (int dy = 0; dy < o_dofs1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < c_dofs1D; ++dx)
|
||||
{
|
||||
const real_t xv = X(dx + dy * c_dofs1D, e);
|
||||
for (int iy = 0; iy < c_dofs1D; ++iy)
|
||||
{
|
||||
const real_t gy = Gc(dy, iy);
|
||||
for (int ix = 0; ix < c_dofs1D; ++ix)
|
||||
{
|
||||
Y(ix, iy, e) += Bc(dx, ix) * gy * xv;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const int y_nd = c_dofs1D * o_dofs1D;
|
||||
for (int dy = 0; dy < c_dofs1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < o_dofs1D; ++dx)
|
||||
{
|
||||
const real_t xv = X(y_nd + dx + dy * o_dofs1D, e);
|
||||
for (int iy = 0; iy < c_dofs1D; ++iy)
|
||||
{
|
||||
const real_t by = Bc(dy, iy);
|
||||
for (int ix = 0; ix < c_dofs1D; ++ix)
|
||||
{
|
||||
Y(ix, iy, e) -= Gc(dx, ix) * by * xv;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
// Apply to x corresponding to DOFs in H^1 (domain) the (topological) gradient
|
||||
// to get a dof in H(curl) (range). You can think of the range as the "test" space
|
||||
// and the domain as the "trial" space, but there's no integration.
|
||||
@@ -1950,4 +2159,266 @@ void IdentityInterpolator::AddMultTransposePA(const Vector &x, Vector &y) const
|
||||
}
|
||||
}
|
||||
|
||||
void CurlInterpolator::AssemblePA(const FiniteElementSpace &dom_fes,
|
||||
const FiniteElementSpace &ran_fes)
|
||||
{
|
||||
Mesh *mesh = dom_fes.GetMesh();
|
||||
dim = mesh->Dimension();
|
||||
ne = dom_fes.GetNE();
|
||||
pa_mode_2d = 0;
|
||||
MFEM_VERIFY(ne == ran_fes.GetNE(),
|
||||
"Different meshes for domain and range spaces");
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
pa_data.SetSize(0);
|
||||
const FiniteElement *dom_fel = dom_fes.GetTypicalFE();
|
||||
const FiniteElement *ran_fel = ran_fes.GetTypicalFE();
|
||||
const bool hcurl_to_scalar =
|
||||
dynamic_cast<const VectorTensorFiniteElement*>(dom_fel) != NULL &&
|
||||
dom_fel->GetDerivType() == FiniteElement::CURL &&
|
||||
dynamic_cast<const TensorBasisElement*>(ran_fel) != NULL &&
|
||||
ran_fel->GetRangeType() == FiniteElement::SCALAR;
|
||||
const bool scalar_to_hdiv =
|
||||
dynamic_cast<const TensorBasisElement*>(dom_fel) != NULL &&
|
||||
dom_fel->GetRangeType() == FiniteElement::SCALAR &&
|
||||
dynamic_cast<const VectorTensorFiniteElement*>(ran_fel) != NULL &&
|
||||
ran_fel->GetDerivType() == FiniteElement::DIV;
|
||||
|
||||
MFEM_VERIFY(hcurl_to_scalar || scalar_to_hdiv,
|
||||
"2D CurlInterpolator PA supports H(curl)->scalar and scalar->H(div) only.");
|
||||
|
||||
int closed_basis_type = -1;
|
||||
int open_basis_type = -1;
|
||||
if (hcurl_to_scalar)
|
||||
{
|
||||
const auto *trial_fec = dynamic_cast<const ND_FECollection*>(dom_fes.FEColl());
|
||||
const auto *range_fec = dynamic_cast<const L2_FECollection*>(ran_fes.FEColl());
|
||||
MFEM_VERIFY(trial_fec != NULL, "H(curl) domain must use ND_FECollection.");
|
||||
MFEM_VERIFY(range_fec != NULL, "Scalar range must use L2_FECollection.");
|
||||
MFEM_VERIFY(ran_fel->GetMapType() == FiniteElement::INTEGRAL,
|
||||
"2D H(curl)->scalar CurlInterpolator PA supports integral-map scalar range spaces only.");
|
||||
closed_basis_type = trial_fec->GetClosedBasisType();
|
||||
open_basis_type = trial_fec->GetOpenBasisType();
|
||||
MFEM_VERIFY(range_fec->GetBasisType() == open_basis_type,
|
||||
"Domain/range open basis types do not match.");
|
||||
pa_mode_2d = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto *trial_fec = dynamic_cast<const H1_FECollection*>(dom_fes.FEColl());
|
||||
const auto *range_fec = dynamic_cast<const RT_FECollection*>(ran_fes.FEColl());
|
||||
MFEM_VERIFY(trial_fec != NULL, "Scalar domain must use H1_FECollection.");
|
||||
MFEM_VERIFY(range_fec != NULL, "H(div) range must use RT_FECollection.");
|
||||
closed_basis_type = trial_fec->GetBasisType();
|
||||
open_basis_type = range_fec->GetOpenBasisType();
|
||||
MFEM_VERIFY(range_fec->GetClosedBasisType() == closed_basis_type,
|
||||
"Domain/range closed basis types do not match.");
|
||||
pa_mode_2d = 2;
|
||||
}
|
||||
|
||||
const int order = hcurl_to_scalar
|
||||
? dynamic_cast<const VectorTensorFiniteElement*>(dom_fel)->GetOrder()
|
||||
: dynamic_cast<const NodalTensorFiniteElement*>(dom_fel)->GetOrder();
|
||||
c_dofs1D = order + 1;
|
||||
o_dofs1D = order;
|
||||
|
||||
closed_dofquad_fe.reset(new H1_SegmentElement(order, closed_basis_type));
|
||||
open_dofquad_fe.reset(new L2_SegmentElement(order - 1, open_basis_type));
|
||||
|
||||
mfem::QuadratureFunctions1D qf1d;
|
||||
mfem::IntegrationRule closed_ir;
|
||||
closed_ir.SetSize(c_dofs1D);
|
||||
qf1d.GaussLobatto(c_dofs1D, &closed_ir);
|
||||
|
||||
mfem::IntegrationRule open_ir;
|
||||
open_ir.SetSize(o_dofs1D);
|
||||
qf1d.GaussLegendre(o_dofs1D, &open_ir);
|
||||
|
||||
maps_C_C = &closed_dofquad_fe->GetDofToQuad(closed_ir, DofToQuad::TENSOR);
|
||||
maps_O_C = &closed_dofquad_fe->GetDofToQuad(open_ir, DofToQuad::TENSOR);
|
||||
maps_O_O = &open_dofquad_fe->GetDofToQuad(open_ir, DofToQuad::TENSOR);
|
||||
|
||||
MFEM_VERIFY(maps_C_C->ndof == c_dofs1D && maps_C_C->nqpt == c_dofs1D, "");
|
||||
MFEM_VERIFY(maps_O_C->ndof == c_dofs1D && maps_O_C->nqpt == o_dofs1D, "");
|
||||
MFEM_VERIFY(maps_O_O->ndof == o_dofs1D && maps_O_O->nqpt == o_dofs1D, "");
|
||||
return;
|
||||
}
|
||||
|
||||
closed_dofquad_fe.reset();
|
||||
open_dofquad_fe.reset();
|
||||
maps_C_C = nullptr;
|
||||
maps_O_C = nullptr;
|
||||
maps_O_O = nullptr;
|
||||
|
||||
const VectorTensorFiniteElement *dom_el =
|
||||
dynamic_cast<const VectorTensorFiniteElement *>(dom_fes.GetTypicalFE());
|
||||
const VectorTensorFiniteElement *ran_el =
|
||||
dynamic_cast<const VectorTensorFiniteElement *>(ran_fes.GetTypicalFE());
|
||||
MFEM_VERIFY(dom_el != NULL, "Only VectorTensorFiniteElement is supported!");
|
||||
MFEM_VERIFY(ran_el != NULL, "Only VectorTensorFiniteElement is supported!");
|
||||
MFEM_VERIFY(dom_el->GetDerivType() == FiniteElement::CURL,
|
||||
"Domain space must be H(curl)");
|
||||
MFEM_VERIFY(ran_el->GetDerivType() == FiniteElement::DIV,
|
||||
"Range space must be H(div)");
|
||||
|
||||
const int dims = dom_el->GetDim();
|
||||
MFEM_VERIFY(dims == 3, "");
|
||||
|
||||
ndof_o = dom_el->GetOrder();
|
||||
int ndof_c = ndof_o + 1;
|
||||
nquad_o = ran_el->GetOrder();
|
||||
int nquad_c = nquad_o + 1;
|
||||
|
||||
// extract the tensor product range dof locations
|
||||
std::vector<real_t> qc(nquad_c);
|
||||
std::vector<real_t> qo(nquad_o);
|
||||
{
|
||||
const IntegrationRule &ran_nodes = ran_el->GetNodes();
|
||||
const Array<int> &quad_map = ran_el->GetDofMap();
|
||||
for (int i = 0; i < nquad_c; ++i)
|
||||
{
|
||||
int idx = UnsignIndex(quad_map[i]);
|
||||
qc[i] = ran_nodes.IntPoint(idx).x;
|
||||
}
|
||||
int offset = ndof_c * ndof_o * ndof_o;
|
||||
for (int i = 0; i < nquad_o; ++i)
|
||||
{
|
||||
int idx = UnsignIndex(quad_map[i + offset]);
|
||||
qo[i] = ran_nodes.IntPoint(idx).x;
|
||||
}
|
||||
}
|
||||
|
||||
// evaluate closed/open 1D basis (and their derivatives) at closed and
|
||||
// open quads
|
||||
// storage order: GCO, BCC, BOO
|
||||
pa_data.SetSize(ndof_c * nquad_o + ndof_c * nquad_c + ndof_o * nquad_o);
|
||||
auto ptr = pa_data.HostWrite();
|
||||
auto &cbasis1d = dom_el->GetBasis1D();
|
||||
auto &obasis1d = dom_el->GetOpenBasis1D();
|
||||
Vector b, g;
|
||||
b.SetSize(ndof_c);
|
||||
g.SetSize(ndof_c);
|
||||
for (int j = 0; j < nquad_o; ++j)
|
||||
{
|
||||
cbasis1d.Eval(qo[j], b, g);
|
||||
for (int i = 0; i < ndof_c; ++i)
|
||||
{
|
||||
ptr[j + i * nquad_o] = g[i];
|
||||
}
|
||||
}
|
||||
ptr += nquad_o * ndof_c;
|
||||
|
||||
for (int j = 0; j < nquad_c; ++j)
|
||||
{
|
||||
cbasis1d.Eval(qc[j], b);
|
||||
for (int i = 0; i < ndof_c; ++i)
|
||||
{
|
||||
ptr[j + i * nquad_c] = b[i];
|
||||
}
|
||||
}
|
||||
ptr += ndof_c * nquad_c;
|
||||
|
||||
b.SetSize(ndof_o);
|
||||
for (int j = 0; j < nquad_o; ++j)
|
||||
{
|
||||
obasis1d.Eval(qo[j], b);
|
||||
for (int i = 0; i < ndof_o; ++i)
|
||||
{
|
||||
ptr[j + i * nquad_o] = b[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
CurlInterpolator::Kernels::Kernels()
|
||||
{
|
||||
CurlInterpolator::AddSpecialization<3, 1, 1>();
|
||||
CurlInterpolator::AddSpecialization<3, 2, 2>();
|
||||
CurlInterpolator::AddSpecialization<3, 3, 3>();
|
||||
CurlInterpolator::AddSpecialization<3, 4, 4>();
|
||||
CurlInterpolator::AddSpecialization<3, 5, 5>();
|
||||
}
|
||||
|
||||
CurlInterpolator::CurlInterpolator() { static Kernels kernels{}; }
|
||||
|
||||
void CurlInterpolator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
MFEM_VERIFY(maps_C_C != nullptr && maps_O_C != nullptr,
|
||||
"2D CurlInterpolator PA data is not assembled.");
|
||||
if (pa_mode_2d == 1)
|
||||
{
|
||||
MFEM_VERIFY(maps_O_O != nullptr,
|
||||
"2D CurlInterpolator scalar curl map is not assembled.");
|
||||
PAHcurlApplyCurl2D(c_dofs1D, o_dofs1D, ne, maps_O_O->B, maps_O_C->G,
|
||||
x, y);
|
||||
}
|
||||
else if (pa_mode_2d == 2)
|
||||
{
|
||||
PAHdivApplyCurl2D(c_dofs1D, o_dofs1D, ne, maps_C_C->B, maps_O_C->G,
|
||||
x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unsupported 2D CurlInterpolator mode.");
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
ApplyPAKernels::Run(dim, ndof_o, nquad_o, ne, ndof_o, nquad_o, pa_data, x, y);
|
||||
}
|
||||
|
||||
void CurlInterpolator::AddMultTransposePA(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
MFEM_VERIFY(maps_C_C != nullptr && maps_O_C != nullptr,
|
||||
"2D CurlInterpolator PA data is not assembled.");
|
||||
if (pa_mode_2d == 1)
|
||||
{
|
||||
MFEM_VERIFY(maps_O_O != nullptr,
|
||||
"2D CurlInterpolator scalar curl map is not assembled.");
|
||||
PAHcurlApplyCurlTranspose2D(c_dofs1D, o_dofs1D, ne, maps_O_O->B,
|
||||
maps_O_C->G, x, y);
|
||||
}
|
||||
else if (pa_mode_2d == 2)
|
||||
{
|
||||
PAHdivApplyCurlTranspose2D(c_dofs1D, o_dofs1D, ne, maps_C_C->B,
|
||||
maps_O_C->G, x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unsupported 2D CurlInterpolator mode.");
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
ApplyTPAKernels::Run(dim, ndof_o, nquad_o, ne, ndof_o, nquad_o, pa_data, x, y);
|
||||
}
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
|
||||
CurlInterpolator::ApplyKernelType
|
||||
CurlInterpolator::ApplyPAKernels::Fallback(int DIM, int, int)
|
||||
{
|
||||
if (DIM == 3)
|
||||
{
|
||||
return internal::CurlInterpolatorApply3DSmem<0, 0>;
|
||||
}
|
||||
MFEM_ABORT("Bad dimension!");
|
||||
}
|
||||
|
||||
CurlInterpolator::ApplyKernelType
|
||||
CurlInterpolator::ApplyTPAKernels::Fallback(int DIM, int, int)
|
||||
{
|
||||
if (DIM == 3)
|
||||
{
|
||||
return internal::CurlInterpolatorTApply3DSmem<0, 0>;
|
||||
}
|
||||
MFEM_ABORT("Bad dimension!");
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -672,6 +672,8 @@ void MixedVectorGradientIntegrator::AssemblePA(const FiniteElementSpace
|
||||
const NodalTensorFiniteElement *trial_el =
|
||||
dynamic_cast<const NodalTensorFiniteElement*>(trial_fel);
|
||||
MFEM_VERIFY(trial_el != NULL, "Only NodalTensorFiniteElement is supported!");
|
||||
MFEM_VERIFY(trial_el->GetMapType() == FiniteElement::VALUE,
|
||||
"Only value map type is supported!");
|
||||
|
||||
const VectorTensorFiniteElement *test_el =
|
||||
dynamic_cast<const VectorTensorFiniteElement*>(test_fel);
|
||||
|
||||
@@ -22,6 +22,8 @@ void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const FiniteElement &el = *fes.GetTypicalFE();
|
||||
MFEM_VERIFY(el.GetMapType() == FiniteElement::VALUE,
|
||||
"Only value map type supported");
|
||||
ElementTransformation &Trans = *mesh->GetTypicalElementTransformation();
|
||||
const auto *ir = IntRule ? IntRule : &MassIntegrator::GetRule(el, el, Trans);
|
||||
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_BILININTEG_VECTORFEMASS_KERNELS_HPP
|
||||
#define MFEM_BILININTEG_VECTORFEMASS_KERNELS_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
|
||||
#include "bilininteg_diffusion_kernels.hpp"
|
||||
#include "bilininteg_hcurl_kernels.hpp"
|
||||
#include "bilininteg_hdiv_kernels.hpp"
|
||||
#include "bilininteg_hcurlhdiv_kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
namespace internal
|
||||
{
|
||||
namespace hcurlmass
|
||||
{
|
||||
constexpr int NBZ3D(int d1d, int q1d)
|
||||
{
|
||||
if (d1d <= 1 || q1d <= 0)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
// assume q1d >= d1d
|
||||
// z dimension is capped at 64 on nvidia and amd gpus
|
||||
int tmp = std::min((128 + q1d * q1d * q1d - 1) / (q1d * q1d * q1d), 64);
|
||||
int smem_req =
|
||||
sizeof(mfem::real_t) *
|
||||
(3 * ((d1d - 1) * d1d * d1d + 2 * q1d * q1d * q1d) * tmp +
|
||||
q1d * (d1d - 1) + q1d * d1d);
|
||||
// assume GPU has at least 48k shared memory
|
||||
return std::max(std::min(tmp, (48 * 1024 + smem_req - 1) / smem_req), 1);
|
||||
}
|
||||
} // namespace hcurlmass
|
||||
} // namespace internal
|
||||
|
||||
template <FiniteElement::DerivType TrialType, FiniteElement::DerivType TestType,
|
||||
int DIM, int TrialD1D, int TestD1D, int Q1D>
|
||||
VectorFEMassIntegrator::ApplyKernelType
|
||||
VectorFEMassIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
constexpr bool trial_curl = (TrialType == mfem::FiniteElement::CURL);
|
||||
constexpr bool trial_div = (TrialType == mfem::FiniteElement::DIV);
|
||||
constexpr bool test_curl = (TestType == mfem::FiniteElement::CURL);
|
||||
constexpr bool test_div = (TestType == mfem::FiniteElement::DIV);
|
||||
|
||||
if constexpr (DIM == 3)
|
||||
{
|
||||
if constexpr (trial_curl && test_curl)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
// assume TrialD1D == TestD1D
|
||||
return internal::SmemPAHcurlMassApply3D<
|
||||
TrialD1D, Q1D, internal::hcurlmass::NBZ3D(TrialD1D, Q1D)>;
|
||||
}
|
||||
else
|
||||
{
|
||||
return internal::PAHcurlMassApply3D;
|
||||
}
|
||||
}
|
||||
else if constexpr (trial_div && test_div)
|
||||
{
|
||||
// assumes TrialD1D == TestD1D
|
||||
return internal::SmemPAHdivMassApply3D<TrialD1D, Q1D>;
|
||||
}
|
||||
else if constexpr (trial_curl && test_div)
|
||||
{
|
||||
return internal::PAHdivHcurlMassApply3D;
|
||||
}
|
||||
else if constexpr (trial_div && test_curl)
|
||||
{
|
||||
return internal::PAHcurlHdivMassApply3D;
|
||||
}
|
||||
}
|
||||
else if constexpr (DIM == 2) // 2D
|
||||
{
|
||||
if constexpr (trial_curl && test_curl)
|
||||
{
|
||||
return internal::PAHcurlMassApply2D;
|
||||
}
|
||||
else if constexpr (trial_div && test_div)
|
||||
{
|
||||
// assumes TrialD1D == TestD1D
|
||||
return internal::SmemPAHdivMassApply2D<TrialD1D, Q1D>;
|
||||
}
|
||||
else if constexpr (trial_curl && test_div)
|
||||
{
|
||||
return internal::PAHdivHcurlMassApply2D;
|
||||
}
|
||||
else if constexpr (trial_div && test_curl)
|
||||
{
|
||||
return internal::PAHcurlHdivMassApply2D;
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -10,15 +10,123 @@
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
#include "bilininteg_diffusion_kernels.hpp"
|
||||
#include "bilininteg_hcurl_kernels.hpp"
|
||||
#include "bilininteg_hdiv_kernels.hpp"
|
||||
#include "bilininteg_hcurlhdiv_kernels.hpp"
|
||||
#include "bilininteg_vectorfemass_kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
VectorFEMassIntegrator::ApplyKernelType
|
||||
VectorFEMassIntegrator::ApplyPAKernels::Fallback(
|
||||
FiniteElement::DerivType TrialType, FiniteElement::DerivType TestType,
|
||||
int dim, int, int, int)
|
||||
{
|
||||
const bool trial_curl = (TrialType == mfem::FiniteElement::CURL);
|
||||
const bool trial_div = (TrialType == mfem::FiniteElement::DIV);
|
||||
const bool test_curl = (TestType == mfem::FiniteElement::CURL);
|
||||
const bool test_div = (TestType == mfem::FiniteElement::DIV);
|
||||
|
||||
if (dim == 3)
|
||||
{
|
||||
if (trial_curl && test_curl)
|
||||
{
|
||||
return internal::PAHcurlMassApply3D;
|
||||
}
|
||||
else if (trial_div && test_div)
|
||||
{
|
||||
return internal::PAHdivMassApply3D;
|
||||
}
|
||||
else if (trial_curl && test_div)
|
||||
{
|
||||
return internal::PAHdivHcurlMassApply3D;
|
||||
}
|
||||
else if (trial_div && test_curl)
|
||||
{
|
||||
return internal::PAHcurlHdivMassApply3D;
|
||||
}
|
||||
}
|
||||
else if (dim == 2) // 2D
|
||||
{
|
||||
if (trial_curl && test_curl)
|
||||
{
|
||||
return internal::PAHcurlMassApply2D;
|
||||
}
|
||||
else if (trial_div && test_div)
|
||||
{
|
||||
return internal::PAHdivMassApply2D;
|
||||
}
|
||||
else if (trial_curl && test_div)
|
||||
{
|
||||
return internal::PAHdivHcurlMassApply2D;
|
||||
}
|
||||
else if (trial_div && test_curl)
|
||||
{
|
||||
return internal::PAHcurlHdivMassApply2D;
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
VectorFEMassIntegrator::Kernels::Kernels()
|
||||
{
|
||||
// h(curl), h(curl)
|
||||
// Q = P + 1 (3D)
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 2, 2, 3>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 3, 3, 4>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 4, 4, 5>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 5, 5, 6>();
|
||||
// Q = P + 2 (3D)
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 2, 2, 4>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 3, 3, 5>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 4, 4, 6>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 5, 5, 7>();
|
||||
// Q = P + 4 (3D)
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 2, 2, 6>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 3, 3, 7>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 4, 4, 8>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
|
||||
FiniteElement::CURL, 3, 5, 5, 9>();
|
||||
// h(div), h(div)
|
||||
// Q = P (2D)
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
|
||||
FiniteElement::DIV, 2, 2, 2, 2>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
|
||||
FiniteElement::DIV, 2, 3, 3, 3>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
|
||||
FiniteElement::DIV, 2, 4, 4, 4>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
|
||||
FiniteElement::DIV, 2, 5, 5, 5>();
|
||||
|
||||
// Q = P + 1 (3D)
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
|
||||
FiniteElement::DIV, 3, 2, 2, 3>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
|
||||
FiniteElement::DIV, 3, 3, 3, 4>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
|
||||
FiniteElement::DIV, 3, 4, 4, 5>();
|
||||
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
|
||||
FiniteElement::DIV, 3, 5, 5, 6>();
|
||||
}
|
||||
|
||||
void VectorFEMassIntegrator::Init(Coefficient *q, DiagonalMatrixCoefficient *dq,
|
||||
MatrixCoefficient *mq)
|
||||
{
|
||||
static Kernels kernels{};
|
||||
Q = q;
|
||||
DQ = dq;
|
||||
MQ = mq;
|
||||
}
|
||||
|
||||
void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
@@ -67,8 +175,8 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &trial_fes,
|
||||
|
||||
MFEM_VERIFY(dofs1D == mapsO->ndof + 1 && quad1D == mapsO->nqpt, "");
|
||||
|
||||
trial_fetype = trial_el->GetDerivType();
|
||||
test_fetype = test_el->GetDerivType();
|
||||
trial_fetype = static_cast<FiniteElement::DerivType>(trial_el->GetDerivType());
|
||||
test_fetype = static_cast<FiniteElement::DerivType>(test_el->GetDerivType());
|
||||
|
||||
const bool trial_curl = (trial_fetype == mfem::FiniteElement::CURL);
|
||||
const bool trial_div = (trial_fetype == mfem::FiniteElement::DIV);
|
||||
@@ -215,225 +323,34 @@ void VectorFEMassIntegrator::AssembleDiagonalPA(Vector& diag)
|
||||
|
||||
void VectorFEMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
const bool trial_curl = (trial_fetype == mfem::FiniteElement::CURL);
|
||||
const bool trial_div = (trial_fetype == mfem::FiniteElement::DIV);
|
||||
const bool test_curl = (test_fetype == mfem::FiniteElement::CURL);
|
||||
const bool test_div = (test_fetype == mfem::FiniteElement::DIV);
|
||||
|
||||
if (dim == 3)
|
||||
{
|
||||
if (trial_curl && test_curl)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
const int ID = (dofs1D << 4) | quad1D;
|
||||
switch (ID)
|
||||
{
|
||||
case 0x23:
|
||||
return internal::SmemPAHcurlMassApply3D<2,3>(
|
||||
dofs1D, quad1D, ne, symmetric,
|
||||
mapsO->B, mapsC->B, mapsO->Bt,
|
||||
mapsC->Bt, pa_data, x, y);
|
||||
case 0x34:
|
||||
return internal::SmemPAHcurlMassApply3D<3,4>(
|
||||
dofs1D, quad1D, ne, symmetric,
|
||||
mapsO->B, mapsC->B, mapsO->Bt,
|
||||
mapsC->Bt, pa_data, x, y);
|
||||
case 0x45:
|
||||
return internal::SmemPAHcurlMassApply3D<4,5>(
|
||||
dofs1D, quad1D, ne, symmetric,
|
||||
mapsO->B, mapsC->B, mapsO->Bt,
|
||||
mapsC->Bt, pa_data, x, y);
|
||||
case 0x56:
|
||||
return internal::SmemPAHcurlMassApply3D<5,6>(
|
||||
dofs1D, quad1D, ne, symmetric,
|
||||
mapsO->B, mapsC->B, mapsO->Bt,
|
||||
mapsC->Bt, pa_data, x, y);
|
||||
default:
|
||||
return internal::SmemPAHcurlMassApply3D(
|
||||
dofs1D, quad1D, ne, symmetric,
|
||||
mapsO->B, mapsC->B, mapsO->Bt,
|
||||
mapsC->Bt, pa_data, x, y);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
internal::PAHcurlMassApply3D(dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
|
||||
mapsO->Bt, mapsC->Bt, pa_data, x, y);
|
||||
}
|
||||
}
|
||||
else if (trial_div && test_div)
|
||||
{
|
||||
internal::PAHdivMassApply(3, dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
|
||||
mapsO->Bt, mapsC->Bt, pa_data, x, y);
|
||||
}
|
||||
else if (trial_curl && test_div)
|
||||
{
|
||||
const bool scalarCoeff = !(DQ || MQ);
|
||||
internal::PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
|
||||
true, false, mapsO->B, mapsC->B, mapsOtest->Bt,
|
||||
mapsCtest->Bt, pa_data, x, y);
|
||||
}
|
||||
else if (trial_div && test_curl)
|
||||
{
|
||||
const bool scalarCoeff = !(DQ || MQ);
|
||||
internal::PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
|
||||
false, false, mapsO->B, mapsC->B, mapsOtest->Bt,
|
||||
mapsCtest->Bt, pa_data, x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
}
|
||||
else // 2D
|
||||
{
|
||||
if (trial_curl && test_curl)
|
||||
{
|
||||
internal::PAHcurlMassApply2D(dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
|
||||
mapsO->Bt, mapsC->Bt, pa_data, x, y);
|
||||
}
|
||||
else if (trial_div && test_div)
|
||||
{
|
||||
internal::PAHdivMassApply(2, dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
|
||||
mapsO->Bt,
|
||||
mapsC->Bt, pa_data, x, y);
|
||||
}
|
||||
else if ((trial_curl && test_div) || (trial_div && test_curl))
|
||||
{
|
||||
const bool scalarCoeff = !(DQ || MQ);
|
||||
internal::PAHcurlHdivMassApply2D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
|
||||
trial_curl, false, mapsO->B, mapsC->B,
|
||||
mapsOtest->Bt, mapsCtest->Bt, pa_data, x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
}
|
||||
const bool scalar_coeff = !(DQ || MQ);
|
||||
ApplyPAKernels::Run(trial_fetype, test_fetype, dim, dofs1D, dofs1Dtest,
|
||||
quad1D, ne, symmetric, scalar_coeff, mapsO->B, mapsC->B,
|
||||
mapsOtest->Bt, mapsCtest->Bt, pa_data, x, y, dofs1D,
|
||||
dofs1Dtest, quad1D);
|
||||
}
|
||||
|
||||
void VectorFEMassIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
const bool trial_curl = (trial_fetype == mfem::FiniteElement::CURL);
|
||||
const bool trial_div = (trial_fetype == mfem::FiniteElement::DIV);
|
||||
const bool test_curl = (test_fetype == mfem::FiniteElement::CURL);
|
||||
const bool test_div = (test_fetype == mfem::FiniteElement::DIV);
|
||||
const bool scalar_coeff = !(DQ || MQ);
|
||||
|
||||
Vector abs_pa_data(pa_data);
|
||||
abs_pa_data.Abs();
|
||||
|
||||
Array<real_t> absBo(mapsO->B);
|
||||
Array<real_t> absBc(mapsC->B);
|
||||
Array<real_t> absBto(mapsO->Bt);
|
||||
Array<real_t> absBtc(mapsC->Bt);
|
||||
Array<real_t> absBto_t(mapsOtest->Bt);
|
||||
Array<real_t> absBtc_t(mapsCtest->Bt);
|
||||
|
||||
absBo.Abs();
|
||||
absBc.Abs();
|
||||
absBto.Abs();
|
||||
absBtc.Abs();
|
||||
absBto_t.Abs();
|
||||
absBtc_t.Abs();
|
||||
|
||||
if (dim == 3)
|
||||
{
|
||||
if (trial_curl && test_curl)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
const int ID = (dofs1D << 4) | quad1D;
|
||||
switch (ID)
|
||||
{
|
||||
case 0x23:
|
||||
return internal::SmemPAHcurlMassApply3D<2,3>(
|
||||
dofs1D, quad1D, ne, symmetric,
|
||||
absBo, absBc, absBto, absBtc,
|
||||
abs_pa_data, x, y);
|
||||
case 0x34:
|
||||
return internal::SmemPAHcurlMassApply3D<3,4>(
|
||||
dofs1D, quad1D, ne, symmetric,
|
||||
absBo, absBc, absBto, absBtc,
|
||||
abs_pa_data, x, y);
|
||||
case 0x45:
|
||||
return internal::SmemPAHcurlMassApply3D<4,5>(
|
||||
dofs1D, quad1D, ne, symmetric,
|
||||
absBo, absBc, absBto, absBtc,
|
||||
abs_pa_data, x, y);
|
||||
case 0x56:
|
||||
return internal::SmemPAHcurlMassApply3D<5,6>(
|
||||
dofs1D, quad1D, ne, symmetric,
|
||||
absBo, absBc, absBto, absBtc,
|
||||
abs_pa_data, x, y);
|
||||
default:
|
||||
return internal::SmemPAHcurlMassApply3D(
|
||||
dofs1D, quad1D, ne, symmetric,
|
||||
absBo, absBc, absBto, absBtc,
|
||||
abs_pa_data, x, y);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
internal::PAHcurlMassApply3D(dofs1D, quad1D, ne, symmetric,
|
||||
absBo, absBc, absBto, absBtc,
|
||||
abs_pa_data, x, y);
|
||||
}
|
||||
}
|
||||
else if (trial_div && test_div)
|
||||
{
|
||||
internal::PAHdivMassApply(3, dofs1D, quad1D, ne, symmetric,
|
||||
absBo, absBc, absBto, absBtc,
|
||||
abs_pa_data, x, y);
|
||||
}
|
||||
else if (trial_curl && test_div)
|
||||
{
|
||||
const bool scalarCoeff = !(DQ || MQ);
|
||||
internal::PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne,
|
||||
scalarCoeff, true, false,
|
||||
absBo, absBc, absBto_t, absBtc_t,
|
||||
abs_pa_data, x, y);
|
||||
}
|
||||
else if (trial_div && test_curl)
|
||||
{
|
||||
const bool scalarCoeff = !(DQ || MQ);
|
||||
internal::PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne,
|
||||
scalarCoeff, false, false,
|
||||
absBo, absBc, absBto_t, absBtc_t,
|
||||
abs_pa_data, x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
}
|
||||
else // 2D
|
||||
{
|
||||
if (trial_curl && test_curl)
|
||||
{
|
||||
internal::PAHcurlMassApply2D(dofs1D, quad1D, ne, symmetric,
|
||||
absBo, absBc, absBto, absBtc,
|
||||
abs_pa_data, x, y);
|
||||
}
|
||||
else if (trial_div && test_div)
|
||||
{
|
||||
internal::PAHdivMassApply(2, dofs1D, quad1D, ne, symmetric,
|
||||
absBo, absBc, absBto, absBtc,
|
||||
abs_pa_data, x, y);
|
||||
}
|
||||
else if ((trial_curl && test_div) || (trial_div && test_curl))
|
||||
{
|
||||
const bool scalarCoeff = !(DQ || MQ);
|
||||
internal::PAHcurlHdivMassApply2D(dofs1D, dofs1Dtest, quad1D, ne,
|
||||
scalarCoeff, trial_curl, false,
|
||||
absBo, absBc, absBto_t, absBtc_t,
|
||||
abs_pa_data, x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
}
|
||||
ApplyPAKernels::Run(trial_fetype, test_fetype, dim, dofs1D, dofs1Dtest,
|
||||
quad1D, ne, symmetric, scalar_coeff, absBo, absBc,
|
||||
absBto_t, absBtc_t, abs_pa_data, x, y, dofs1D,
|
||||
dofs1Dtest, quad1D);
|
||||
}
|
||||
|
||||
void VectorFEMassIntegrator::AddMultTransposePA(const Vector &x,
|
||||
|
||||
+4
-8
@@ -542,7 +542,10 @@ void QuadratureFunctions1D::GaussJacobi(const int np, const real_t alpha,
|
||||
return;
|
||||
}
|
||||
|
||||
#ifndef MFEM_USE_MPFR
|
||||
#ifdef MFEM_USE_MPFR
|
||||
MFEM_WARNING("MPFR implementation of Gauss-Jacobi quadrature not implemented yet. Falling "
|
||||
"back to double precision implementation...");
|
||||
#endif
|
||||
|
||||
const int n = np;
|
||||
// common constants for Jacobi polynomials
|
||||
@@ -611,13 +614,6 @@ void QuadratureFunctions1D::GaussJacobi(const int np, const real_t alpha,
|
||||
ab + 1) / ((1.0 - xi*xi)*pp*pp) / pow(2, ab);
|
||||
// map nodes and weights to the interval [0,1]
|
||||
}
|
||||
|
||||
#else // MFEM_USE_MPFR is defined
|
||||
|
||||
MFEM_ABORT("MPFR implementation of Gauss-Jacobi quadrature not defined yet");
|
||||
|
||||
#endif // MFEM_USE_MPFR
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
+4
-4
@@ -94,10 +94,10 @@ void BatchedLOR_AMS::Form2DEdgeToVertex_RT(Array<int> &edge2vert)
|
||||
const int iv0 = ix + iy*op1;
|
||||
const int iv1 = ix1 + iy1*op1;
|
||||
|
||||
// Rotated gradient in 2D (-dy, dx), so flip the sign for the first
|
||||
// component (c == 0).
|
||||
e2v(0, iedge) = (c == 1) ? iv0 : iv1;
|
||||
e2v(1, iedge) = (c == 1) ? iv1 : iv0;
|
||||
// 2D curl (dy, -dx), so flip the sign for the second
|
||||
// component (c == 1).
|
||||
e2v(0, iedge) = (c == 0) ? iv0 : iv1;
|
||||
e2v(1, iedge) = (c == 0) ? iv1 : iv0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+12
-9
@@ -142,8 +142,6 @@ static MFEM_HOST_DEVICE int GetAndIncrementNnzIndex(const int i_L, int* I)
|
||||
|
||||
int BatchedLORAssembly::FillI(SparseMatrix &A) const
|
||||
{
|
||||
static constexpr int Max = 16;
|
||||
|
||||
const int nvdof = fes_ho.GetVSize();
|
||||
|
||||
const int ndof_per_el = fes_ho.GetTypicalFE()->GetDof();
|
||||
@@ -165,6 +163,8 @@ int BatchedLORAssembly::FillI(SparseMatrix &A) const
|
||||
const auto K = dof_glob2loc_offsets_.Read();
|
||||
const auto map = Reshape(sparse_mapping.Read(), nnz_per_row, ndof_per_el);
|
||||
|
||||
Array<int> ij_elts(dof_glob2loc_.Size() * 2);
|
||||
auto d_ij_elts = Reshape(ij_elts.Write(), dof_glob2loc_.Size(), 2);
|
||||
|
||||
auto I = A.WriteI();
|
||||
|
||||
@@ -176,10 +176,10 @@ int BatchedLORAssembly::FillI(SparseMatrix &A) const
|
||||
const int sii = el_dof_lex(ii_el, iel_ho);
|
||||
const int ii = (sii >= 0) ? sii : -1 -sii;
|
||||
// Get number and list of elements containing this DOF
|
||||
int i_elts[Max];
|
||||
const int i_offset = K[ii];
|
||||
const int i_next_offset = K[ii+1];
|
||||
const int i_ne = i_next_offset - i_offset;
|
||||
int *i_elts = &d_ij_elts(i_offset, 0);
|
||||
for (int e_i = 0; e_i < i_ne; ++e_i)
|
||||
{
|
||||
const int si_E = dof_glob2loc[i_offset+e_i]; // signed
|
||||
@@ -202,7 +202,7 @@ int BatchedLORAssembly::FillI(SparseMatrix &A) const
|
||||
}
|
||||
else // assembly required
|
||||
{
|
||||
int j_elts[Max];
|
||||
int *j_elts = &d_ij_elts(j_offset, 1);
|
||||
for (int e_j = 0; e_j < j_ne; ++e_j)
|
||||
{
|
||||
const int sj_E = dof_glob2loc[j_offset+e_j]; // signed
|
||||
@@ -269,7 +269,8 @@ void BatchedLORAssembly::FillJAndData(SparseMatrix &A) const
|
||||
mfem::forall(nvdof + 1, [=] MFEM_HOST_DEVICE (int i) { I[i] = I2[i]; });
|
||||
}
|
||||
|
||||
static constexpr int Max = 16;
|
||||
Array<int> ij_B_el(dof_glob2loc_.Size() * 4);
|
||||
auto d_ij_B_el = Reshape(ij_B_el.Write(), dof_glob2loc_.Size(), 4);
|
||||
|
||||
mfem::forall(ndof_per_el*nel_ho, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
@@ -279,11 +280,13 @@ void BatchedLORAssembly::FillJAndData(SparseMatrix &A) const
|
||||
const int sii = el_dof_lex(ii_el, iel_ho); // signed
|
||||
const int ii = (sii >= 0) ? sii : -1 - sii;
|
||||
// Get number and list of elements containing this DOF
|
||||
int i_elts[Max];
|
||||
int i_B[Max];
|
||||
const int i_offset = K[ii];
|
||||
const int i_next_offset = K[ii+1];
|
||||
const int i_ne = i_next_offset - i_offset;
|
||||
|
||||
int *i_elts = &d_ij_B_el(i_offset, 0);
|
||||
int *i_B = &d_ij_B_el(i_offset, 1);
|
||||
|
||||
for (int e_i = 0; e_i < i_ne; ++e_i)
|
||||
{
|
||||
const int si_E = dof_glob2loc[i_offset+e_i]; // signed
|
||||
@@ -312,8 +315,8 @@ void BatchedLORAssembly::FillJAndData(SparseMatrix &A) const
|
||||
}
|
||||
else // assembly required
|
||||
{
|
||||
int j_elts[Max];
|
||||
int j_B[Max];
|
||||
int *j_elts = &d_ij_B_el(j_offset, 2);
|
||||
int *j_B = &d_ij_B_el(j_offset, 3);
|
||||
for (int e_j = 0; e_j < j_ne; ++e_j)
|
||||
{
|
||||
const int sj_E = dof_glob2loc[j_offset+e_j]; // signed
|
||||
|
||||
@@ -268,8 +268,9 @@ static void Derivatives3D(const int NE,
|
||||
DeviceMatrix B(BG[0], D1D, Q1D);
|
||||
DeviceMatrix G(BG[1], D1D, Q1D);
|
||||
|
||||
MFEM_SHARED real_t sm0[3][MQ1*MQ1*MQ1];
|
||||
MFEM_SHARED real_t sm1[3][MQ1*MQ1*MQ1];
|
||||
constexpr int MDQ = MD1 > MQ1 ? MD1 : MQ1;
|
||||
MFEM_SHARED real_t sm0[3][MD1*MD1*MDQ];
|
||||
MFEM_SHARED real_t sm1[3][MD1*MQ1*MQ1];
|
||||
DeviceTensor<3> X(sm0[2], D1D, D1D, D1D);
|
||||
DeviceTensor<3> DDQ0(sm0[0], D1D, D1D, Q1D);
|
||||
DeviceTensor<3> DDQ1(sm0[1], D1D, D1D, Q1D);
|
||||
|
||||
+45
-7
@@ -14,7 +14,7 @@
|
||||
|
||||
#include "../config/config.hpp"
|
||||
|
||||
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
#include <cusparse.h>
|
||||
#include <library_types.h>
|
||||
#include <cuda_runtime.h>
|
||||
@@ -22,7 +22,7 @@
|
||||
#endif
|
||||
#include "cuda.hpp"
|
||||
|
||||
#if defined(MFEM_USE_HIP) && defined(__HIP__)
|
||||
#if defined(MFEM_USE_HIP)
|
||||
#include <hip/hip_runtime.h>
|
||||
#endif
|
||||
#include "hip.hpp"
|
||||
@@ -45,15 +45,17 @@
|
||||
#endif
|
||||
|
||||
#if !defined(MFEM_USE_CUDA_OR_HIP)
|
||||
constexpr bool mfem_use_gpu = false;
|
||||
#define MFEM_DEVICE
|
||||
#define MFEM_HOST
|
||||
#define MFEM_LAMBDA
|
||||
// #define MFEM_HOST_DEVICE // defined in config/config.hpp
|
||||
// MFEM_DEVICE_SYNC is made available for debugging purposes
|
||||
#define MFEM_DEVICE_SYNC
|
||||
// MFEM_STREAM_SYNC is used for UVM and MPI GPU-Aware kernels
|
||||
#define MFEM_STREAM_SYNC
|
||||
#endif
|
||||
|
||||
#if !defined(MFEM_USE_CUDA_OR_HIP_LANG)
|
||||
#define MFEM_DEVICE
|
||||
#define MFEM_HOST
|
||||
#define MFEM_LAMBDA
|
||||
// #define MFEM_HOST_DEVICE // defined in config/config.hpp
|
||||
#define MFEM_LAUNCH_BOUNDS(...)
|
||||
#endif
|
||||
|
||||
@@ -66,6 +68,23 @@ constexpr bool mfem_use_gpu = false;
|
||||
#define MFEM_THREAD_SIZE(k) 1
|
||||
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=0; i<N; i++)
|
||||
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) MFEM_FOREACH_THREAD(i,k,N)
|
||||
// Assigns a thread block shaped (SX,SY,SZ) contiguous in x.
|
||||
// Example (3,2,1) block:
|
||||
// 0 (0,0), 1 (1,0), 2 (2,0)
|
||||
// 3 (1,0), 4 (1,1), 5 (2,1)
|
||||
#define MFEM_FOREACH_THREAD_DIRECT_3D(ix, iy, iz, k, SX, SY, SZ) \
|
||||
for (int iz = 0; iz < SZ; ++iz) \
|
||||
for (int iy = 0; iy < SY; ++iy) \
|
||||
for (int ix = 0; ix < SX; ++ix)
|
||||
// Assigns a thread block shaped (OX,OY,OZ) to work on items (SX,SY,SZ),
|
||||
// contiguous in x. This intentionally offsets threads within the block to avoid
|
||||
// shared memory bank conflicts.
|
||||
// Example (3,2,1) block assigned to work on (2,2,1) items:
|
||||
// 0 (0,0), 1 (1,0), 2 (N/A)
|
||||
// 3 (1,0), 4 (1,1), 5 (N/A)
|
||||
#define MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(ix, iy, iz, k, SX, SY, SZ, OX, \
|
||||
OY, OZ) \
|
||||
MFEM_FOREACH_THREAD_DIRECT_3D(ix, iy, iz, k, SX, SY, SZ)
|
||||
#endif
|
||||
|
||||
// 'double' and 'float' atomicAdd implementation for previous versions of CUDA
|
||||
@@ -109,4 +128,23 @@ MFEM_HOST_DEVICE T AtomicAdd(T &add, const T val)
|
||||
#endif
|
||||
}
|
||||
|
||||
namespace mfem::internal
|
||||
{
|
||||
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP) && !defined(MFEM_USE_CUDA_OR_HIP_LANG)
|
||||
static constexpr bool can_compile_kernels = false;
|
||||
#else
|
||||
static constexpr bool can_compile_kernels = true;
|
||||
#endif
|
||||
|
||||
template <bool can_compile_kernels = can_compile_kernels>
|
||||
void RequireKernelCompilation()
|
||||
{
|
||||
static_assert(
|
||||
can_compile_kernels,
|
||||
"The calling function needs to be compiled with CUDA/HIP language!");
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
#endif // MFEM_BACKENDS_HPP
|
||||
|
||||
+30
-9
@@ -18,14 +18,8 @@
|
||||
// CUDA block size used by MFEM.
|
||||
#define MFEM_CUDA_BLOCKS 256
|
||||
|
||||
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
#define MFEM_USE_CUDA_OR_HIP
|
||||
constexpr bool mfem_use_gpu = true;
|
||||
#define MFEM_DEVICE __device__
|
||||
#define MFEM_HOST __host__
|
||||
#define MFEM_LAMBDA __host__
|
||||
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
|
||||
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
|
||||
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(cudaDeviceSynchronize())
|
||||
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(cudaStreamSynchronize(0))
|
||||
// Define a CUDA error check macro, MFEM_GPU_CHECK(x), where x returns/is of
|
||||
@@ -40,6 +34,15 @@ constexpr bool mfem_use_gpu = true;
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
// Macros defined only when compiling with CUDA language
|
||||
#if defined(__CUDACC__)
|
||||
#define MFEM_USE_CUDA_OR_HIP_LANG
|
||||
#define MFEM_DEVICE __device__
|
||||
#define MFEM_HOST __host__
|
||||
#define MFEM_LAMBDA __host__
|
||||
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
|
||||
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
|
||||
|
||||
// Define the MFEM inner threading macros
|
||||
#if defined(__CUDA_ARCH__)
|
||||
#define MFEM_SHARED __shared__
|
||||
@@ -49,13 +52,31 @@ constexpr bool mfem_use_gpu = true;
|
||||
#define MFEM_THREAD_SIZE(k) blockDim.k
|
||||
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=threadIdx.k; i<N; i+=blockDim.k)
|
||||
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) if(const int i=threadIdx.k; i<N)
|
||||
// Assigns a thread block shaped (SX,SY,SZ) contiguous in x.
|
||||
// Example (3,2,1) block:
|
||||
// 0 (0,0), 1 (1,0), 2 (2,0)
|
||||
// 3 (1,0), 4 (1,1), 5 (2,1)
|
||||
#define MFEM_FOREACH_THREAD_DIRECT_3D(ix, iy, iz, k, SX, SY, SZ) \
|
||||
if (int ix = threadIdx.k % (SX), iy = threadIdx.k / (SX), iz = iy / (SY); \
|
||||
(iy %= (SY)), (threadIdx.k < (SX) * (SY) * (SZ)))
|
||||
// Assigns a thread block shaped (OX,OY,OZ) to work on items (SX,SY,SZ),
|
||||
// contiguous in x. This intentionally offsets threads within the block to avoid
|
||||
// shared memory bank conflicts.
|
||||
// Example (3,2,1) block assigned to work on (2,2,1) items:
|
||||
// 0 (0,0), 1 (1,0), 2 (N/A)
|
||||
// 3 (1,0), 4 (1,1), 5 (N/A)
|
||||
#define MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(ix, iy, iz, k, SX, SY, SZ, OX, \
|
||||
OY, OZ) \
|
||||
if (int ix = threadIdx.k % (OX), iy = threadIdx.k / (OX), iz = iy / (OY); \
|
||||
(ix < (SX)) && ((iy %= (OY)) < (SY)) && (iz < (SZ)))
|
||||
#endif // defined(__CUDA_ARCH__)
|
||||
#endif // defined(MFEM_USE_CUDA) && defined(__CUDACC__)
|
||||
#endif // defined(__CUDACC__)
|
||||
#endif // defined(MFEM_USE_CUDA)
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
// Function used by the macro MFEM_GPU_CHECK.
|
||||
void mfem_cuda_error(cudaError_t err, const char *expr, const char *func,
|
||||
const char *file, int line);
|
||||
|
||||
+1
-1
@@ -171,7 +171,7 @@ void mfem_error(const char *msg)
|
||||
#ifdef MFEM_USE_EXCEPTIONS
|
||||
if (mfem_error_action == MFEM_ERROR_THROW)
|
||||
{
|
||||
throw ErrorException(msg);
|
||||
throw ErrorException(msg ? msg : "");
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
+2
-10
@@ -15,7 +15,7 @@
|
||||
#include "../config/config.hpp"
|
||||
#include <iomanip>
|
||||
#include <sstream>
|
||||
#ifdef MFEM_USE_HIP
|
||||
#if defined(MFEM_USE_HIP)
|
||||
#include <hip/hip_runtime.h>
|
||||
#endif
|
||||
|
||||
@@ -153,21 +153,13 @@ void mfem_warning(const char *msg = NULL);
|
||||
|
||||
|
||||
// Additional abort functions for HIP
|
||||
#if defined(MFEM_USE_HIP)
|
||||
#ifndef __HIP_DEVICE_COMPILE__
|
||||
template<typename T>
|
||||
__host__ void abort_msg(T & msg)
|
||||
{
|
||||
MFEM_ABORT(msg);
|
||||
}
|
||||
#else
|
||||
#if defined(__HIP_DEVICE_COMPILE__)
|
||||
template<typename T>
|
||||
__device__ void abort_msg(T & msg)
|
||||
{
|
||||
abort();
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// Abort inside a device kernel
|
||||
#if defined(__CUDA_ARCH__)
|
||||
|
||||
@@ -1044,6 +1044,8 @@ inline void ForallWrap(const bool use_dev, const int N,
|
||||
const int X=0, const int Y=0, const int Z=0,
|
||||
const int G=0)
|
||||
{
|
||||
internal::RequireKernelCompilation();
|
||||
|
||||
MFEM_CONTRACT_VAR(X);
|
||||
MFEM_CONTRACT_VAR(Y);
|
||||
MFEM_CONTRACT_VAR(Z);
|
||||
@@ -1276,6 +1278,9 @@ inline void hypre_forall_cpu(int N, lambda &&body)
|
||||
template<typename lambda>
|
||||
inline void hypre_forall_gpu(int N, lambda &&body)
|
||||
{
|
||||
internal::RequireKernelCompilation();
|
||||
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP_LANG)
|
||||
#if defined(HYPRE_USING_CUDA)
|
||||
CuWrap1D(N, body);
|
||||
#elif defined(HYPRE_USING_HIP)
|
||||
@@ -1283,6 +1288,7 @@ inline void hypre_forall_gpu(int N, lambda &&body)
|
||||
#else
|
||||
#error Unknown HYPRE GPU backend!
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
+31
-8
@@ -18,14 +18,8 @@
|
||||
// HIP block size used by MFEM.
|
||||
#define MFEM_HIP_BLOCKS 256
|
||||
|
||||
#if defined(MFEM_USE_HIP) && defined(__HIP__)
|
||||
#if defined(MFEM_USE_HIP)
|
||||
#define MFEM_USE_CUDA_OR_HIP
|
||||
constexpr bool mfem_use_gpu = true;
|
||||
#define MFEM_DEVICE __device__
|
||||
#define MFEM_HOST __host__
|
||||
#define MFEM_LAMBDA __host__ __device__
|
||||
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
|
||||
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
|
||||
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(hipDeviceSynchronize())
|
||||
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(hipStreamSynchronize(0))
|
||||
// Define a HIP error check macro, MFEM_GPU_CHECK(x), where x returns/is of
|
||||
@@ -40,6 +34,15 @@ constexpr bool mfem_use_gpu = true;
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
// Macros defined only when compiling with HIP language
|
||||
#if defined(__HIP__)
|
||||
#define MFEM_USE_CUDA_OR_HIP_LANG
|
||||
#define MFEM_DEVICE __device__
|
||||
#define MFEM_HOST __host__
|
||||
#define MFEM_LAMBDA __host__ __device__
|
||||
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
|
||||
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
|
||||
|
||||
// Define the MFEM inner threading macros
|
||||
#if defined(__HIP_DEVICE_COMPILE__)
|
||||
#define MFEM_SHARED __shared__
|
||||
@@ -51,8 +54,28 @@ constexpr bool mfem_use_gpu = true;
|
||||
for(int i=hipThreadIdx_ ##k; i<N; i+=hipBlockDim_ ##k)
|
||||
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) \
|
||||
if(const int i=hipThreadIdx_ ##k; i<N)
|
||||
// Assigns a thread block shaped (SX,SY,SZ) contiguous in x.
|
||||
// Example (3,2,1) block:
|
||||
// 0 (0,0), 1 (1,0), 2 (2,0)
|
||||
// 3 (1,0), 4 (1,1), 5 (2,1)
|
||||
#define MFEM_FOREACH_THREAD_DIRECT_3D(ix, iy, iz, k, SX, SY, SZ) \
|
||||
if (int ix = hipThreadIdx_##k % (SX), iy = hipThreadIdx_##k / (SX), \
|
||||
iz = iy / (SY); \
|
||||
(iy %= (SY)), (hipThreadIdx_##k < (SX) * (SY) * (SZ)))
|
||||
// Assigns a thread block shaped (OX,OY,OZ) to work on items (SX,SY,SZ),
|
||||
// contiguous in x. This intentionally offsets threads within the block to avoid
|
||||
// shared memory bank conflicts.
|
||||
// Example (3,2,1) block assigned to work on (2,2,1) items:
|
||||
// 0 (0,0), 1 (1,0), 2 (N/A)
|
||||
// 3 (1,0), 4 (1,1), 5 (N/A)
|
||||
#define MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(ix, iy, iz, k, SX, SY, SZ, OX, \
|
||||
OY, OZ) \
|
||||
if (int ix = hipThreadIdx_##k % (OX), iy = hipThreadIdx_##k / (OX), \
|
||||
iz = iy / (OY); \
|
||||
(ix < (SX)) && ((iy %= (OY)) < (SY)) && (iz < (SZ)))
|
||||
#endif // defined(__HIP_DEVICE_COMPILE__)
|
||||
#endif // defined(MFEM_USE_HIP) && defined(__HIP__)
|
||||
#endif // defined(__HIP__)
|
||||
#endif // defined(MFEM_USE_HIP)
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
@@ -550,10 +550,10 @@ void reduce(int N, T &res, B &&body, const R &reducer, bool use_dev,
|
||||
|
||||
int num_mp = Device::NumMultiprocessors(Device::GetId());
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
// good value of mp_sat found experimentally on Lassen
|
||||
// good value of mp_sat found experimentally on Lassen (V100)
|
||||
constexpr int mp_sat = 8;
|
||||
#elif defined(MFEM_USE_HIP)
|
||||
// good value of mp_sat found experimentally on Tuolumne
|
||||
// good value of mp_sat found experimentally on Tuolumne (MI300A)
|
||||
constexpr int mp_sat = 4;
|
||||
#else
|
||||
num_mp = 1;
|
||||
|
||||
+7
-1
@@ -15,6 +15,10 @@
|
||||
#include "backends.hpp"
|
||||
#include "forall.hpp"
|
||||
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP) && !defined(MFEM_USE_CUDA_OR_HIP_LANG)
|
||||
#error "This header requires compilation with CUDA/HIP language!"
|
||||
#else
|
||||
|
||||
#ifdef MFEM_USE_CUDA
|
||||
#include <cub/device/device_scan.cuh>
|
||||
#include <cub/device/device_select.cuh>
|
||||
@@ -406,4 +410,6 @@ void CopyUnique(bool use_dev, InputIt d_in, OutputIt d_out,
|
||||
|
||||
#undef MFEM_CUB_NAMESPACE
|
||||
|
||||
#endif
|
||||
#endif // defined(MFEM_USE_CUDA_OR_HIP) && !defined(MFEM_USE_CUDA_OR_HIP_LANG)
|
||||
|
||||
#endif // MFEM_SCAN_HPP
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
#include "native.hpp"
|
||||
#include "gpu_blas.hpp"
|
||||
#include "magma.hpp"
|
||||
#include "../../general/reducers.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -119,4 +120,16 @@ void BatchedLinAlgBase::MultTranspose(const DenseTensor &A, const Vector &x,
|
||||
AddMult(A, x, y, 1.0, 0.0, Op::T);
|
||||
}
|
||||
|
||||
void VerifyBatchedLUInfo(const Array<int> &info_array, const char *message)
|
||||
{
|
||||
static Array<int> workspace;
|
||||
int status = 0;
|
||||
const int *d_info = info_array.Read();
|
||||
mfem::reduce(
|
||||
info_array.Size(), status,
|
||||
[=] MFEM_HOST_DEVICE (int i, int &r) { r |= d_info[i]; },
|
||||
BOrReducer<int> {}, true, workspace);
|
||||
MFEM_VERIFY(status == 0, message);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -141,6 +141,9 @@ public:
|
||||
virtual ~BatchedLinAlgBase() { }
|
||||
};
|
||||
|
||||
/// Check that all batched LU info values are zero.
|
||||
void VerifyBatchedLUInfo(const Array<int> &info_array, const char *message);
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
|
||||
@@ -126,7 +126,8 @@ void GPUBlasBatchedLinAlg::LUFactor(DenseTensor &A, Array<int> &P) const
|
||||
const blasStatus_t status = MFEM_GPUBLAS_PREFIX(getrfBatched)(
|
||||
GPUBlas::Handle(), n, d_A_ptrs, n, P.Write(),
|
||||
info_array.Write(), n_mat);
|
||||
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "");
|
||||
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "GPU BLAS error.");
|
||||
VerifyBatchedLUInfo(info_array, "Batch LU factorization failed");
|
||||
}
|
||||
|
||||
void GPUBlasBatchedLinAlg::LUSolve(
|
||||
@@ -189,12 +190,14 @@ void GPUBlasBatchedLinAlg::Invert(DenseTensor &A) const
|
||||
status = MFEM_GPUBLAS_PREFIX(getrfBatched)(
|
||||
GPUBlas::Handle(), n, d_LU_ptrs, n, P.Write(),
|
||||
info_array.Write(), n_mat);
|
||||
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "");
|
||||
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "GPU BLAS error.");
|
||||
VerifyBatchedLUInfo(info_array, "Batch LU factorization failed");
|
||||
|
||||
status = MFEM_GPUBLAS_PREFIX(getriBatched)(
|
||||
GPUBlas::Handle(), n, d_LU_ptrs, n, P.ReadWrite(), d_A_ptrs, n,
|
||||
info_array.Write(), n_mat);
|
||||
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "");
|
||||
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "GPU BLAS error.");
|
||||
VerifyBatchedLUInfo(info_array, "Batch matrix inversion failed");
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -99,7 +99,8 @@ void MagmaBatchedLinAlg::LUFactor(DenseTensor &A, Array<int> &P) const
|
||||
const magma_int_t status = MFEM_MAGMA_PREFIX(getrf_batched)(
|
||||
n, n, d_A_ptrs, n, d_P_ptrs,
|
||||
info_array.Write(), n_mat, Magma::Queue());
|
||||
MFEM_VERIFY(status == MAGMA_SUCCESS, "");
|
||||
MFEM_VERIFY(status == MAGMA_SUCCESS, "MAGMA error.");
|
||||
VerifyBatchedLUInfo(info_array, "Batch LU factorization failed");
|
||||
}
|
||||
|
||||
void MagmaBatchedLinAlg::LUSolve(
|
||||
@@ -169,12 +170,14 @@ void MagmaBatchedLinAlg::Invert(DenseTensor &A) const
|
||||
status = MFEM_MAGMA_PREFIX(getrf_batched)(
|
||||
n, n, d_LU_ptrs, n, d_P_ptrs, info_array.Write(), n_mat,
|
||||
Magma::Queue());
|
||||
MFEM_VERIFY(status == MAGMA_SUCCESS, "");
|
||||
MFEM_VERIFY(status == MAGMA_SUCCESS, "MAGMA error.");
|
||||
VerifyBatchedLUInfo(info_array, "Batch LU factorization failed");
|
||||
|
||||
status = MFEM_MAGMA_PREFIX(getri_outofplace_batched)(
|
||||
n, d_LU_ptrs, n, d_P_ptrs, d_A_ptrs, n, info_array.Write(),
|
||||
n_mat, Magma::Queue());
|
||||
MFEM_VERIFY(status == MAGMA_SUCCESS, "");
|
||||
MFEM_VERIFY(status == MAGMA_SUCCESS, "MAGMA error.");
|
||||
VerifyBatchedLUInfo(info_array, "Batch matrix inversion failed");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -246,6 +246,10 @@ SparseMatrix * ComplexSparseMatrix::GetSystemMatrix() const
|
||||
const int nrows_i = (A_i)?A_i->Height():0;
|
||||
const int nrows = std::max(nrows_r, nrows_i);
|
||||
|
||||
const int ncols_r = (A_r)?A_r->Width():0;
|
||||
const int ncols_i = (A_i)?A_i->Width():0;
|
||||
const int ncols = std::max(ncols_r, ncols_i);
|
||||
|
||||
const int *I_r = (A_r)?A_r->GetI():NULL;
|
||||
const int *I_i = (A_i)?A_i->GetI():NULL;
|
||||
|
||||
@@ -280,7 +284,7 @@ SparseMatrix * ComplexSparseMatrix::GetSystemMatrix() const
|
||||
J[I[i] + j] = J_r[I_r[i] + j];
|
||||
D[I[i] + j] = D_r[I_r[i] + j];
|
||||
|
||||
J[I[i+nrows] + off_i + j] = J_r[I_r[i] + j] + nrows;
|
||||
J[I[i+nrows] + off_i + j] = J_r[I_r[i] + j] + ncols;
|
||||
D[I[i+nrows] + off_i + j] = factor*D_r[I_r[i] + j];
|
||||
}
|
||||
}
|
||||
@@ -289,7 +293,7 @@ SparseMatrix * ComplexSparseMatrix::GetSystemMatrix() const
|
||||
const int off_r = (I_r)?(I_r[i+1] - I_r[i]):0;
|
||||
for (int j=0; j<I_i[i+1] - I_i[i]; j++)
|
||||
{
|
||||
J[I[i] + off_r + j] = J_i[I_i[i] + j] + nrows;
|
||||
J[I[i] + off_r + j] = J_i[I_i[i] + j] + ncols;
|
||||
D[I[i] + off_r + j] = -D_i[I_i[i] + j];
|
||||
|
||||
J[I[i+nrows] + j] = J_i[I_i[i] + j];
|
||||
@@ -892,12 +896,12 @@ ComplexHypreParMatrix::getColStartStop(const HypreParMatrix * A_r,
|
||||
HYPRE_BigInt loc_start_stop[2];
|
||||
offd_col_start_stop = new HYPRE_BigInt[2 * num_recv_procs];
|
||||
|
||||
const HYPRE_BigInt * row_part = (A_r) ? A_r->RowPart() :
|
||||
((A_i) ? A_i->RowPart() : NULL);
|
||||
const HYPRE_BigInt * col_part = (A_r) ? A_r->ColPart() :
|
||||
((A_i) ? A_i->ColPart() : NULL);
|
||||
|
||||
int row_part_ind = (HYPRE_AssumedPartitionCheck()) ? 0 : myid_;
|
||||
loc_start_stop[0] = row_part[row_part_ind];
|
||||
loc_start_stop[1] = row_part[row_part_ind+1];
|
||||
int col_part_ind = (HYPRE_AssumedPartitionCheck()) ? 0 : myid_;
|
||||
loc_start_stop[0] = col_part[col_part_ind];
|
||||
loc_start_stop[1] = col_part[col_part_ind+1];
|
||||
|
||||
MPI_Request * req = new MPI_Request[send_procs.size()+recv_procs.size()];
|
||||
MPI_Status * stat = new MPI_Status[send_procs.size()+recv_procs.size()];
|
||||
|
||||
+40
-20
@@ -87,6 +87,9 @@ CuDSSSolver::CuDSSSolver(MPI_Comm comm_) : mpi_comm(comm_)
|
||||
|
||||
CuDSSSolver::~CuDSSSolver()
|
||||
{
|
||||
// Sync the stream to make sure any pending asynchronous operations have
|
||||
// completed.
|
||||
MFEM_STREAM_SYNC;
|
||||
// Destroy the system Matrix, RHS vector and solution vector
|
||||
if (Ac)
|
||||
{
|
||||
@@ -99,7 +102,6 @@ CuDSSSolver::~CuDSSSolver()
|
||||
MFEM_CUDSS_CHECK(cudssDataDestroy(handle, solverData));
|
||||
MFEM_CUDSS_CHECK(cudssConfigDestroy(solverConfig));
|
||||
|
||||
|
||||
MFEM_CUDSS_CHECK(cudssDestroy(handle));
|
||||
handle = nullptr;
|
||||
|
||||
@@ -125,6 +127,9 @@ void CuDSSSolver::InitCuDSS()
|
||||
// Create the cuDSS handle
|
||||
MFEM_CUDSS_CHECK(cudssCreate(&handle));
|
||||
|
||||
// Set CuDSS to use MFEM's default stream of 0.
|
||||
MFEM_CUDSS_CHECK(cudssSetStream(handle, 0));
|
||||
|
||||
#ifdef MFEM_USE_OPENMP
|
||||
// NOTE: Set the threading layer library name to NULL so that cuDSS picks
|
||||
// it from the environment variable "CUDSS_THREADING_LIB"
|
||||
@@ -251,27 +256,42 @@ void CuDSSSolver::SetMatrixCuDSS(int *csr_offsets, int *csr_columns,
|
||||
Ac = std::make_unique<cudssMatrix_t>();
|
||||
// Create empty RHS and solution vectors
|
||||
SetNumRHS(1);
|
||||
// Allocate device memory for csr values
|
||||
CuMemAlloc(&csr_values_d, nnz * sizeof(real_t));
|
||||
}
|
||||
|
||||
if (cuDSSObjectInitialized && !reorder_reuse)
|
||||
{
|
||||
MFEM_STREAM_SYNC;
|
||||
MFEM_CUDSS_CHECK(cudssMatrixDestroy(*Ac));
|
||||
}
|
||||
|
||||
// Allocate device memory for csr values. Unless reuse is specified, the
|
||||
// nnz may be different, so we will free and reallocate.
|
||||
if (csr_values_d == NULL || !reorder_reuse)
|
||||
{
|
||||
if (csr_values_d != NULL) { CuMemFree(csr_values_d); }
|
||||
CuMemAlloc(&csr_values_d, nnz * sizeof(real_t));
|
||||
}
|
||||
CuMemcpyDtoD(csr_values_d, csr_values, nnz * sizeof(real_t));
|
||||
|
||||
// We copy and store the I and J arrays, since the CuDSS matrix object
|
||||
// technically needs these to be valid, so we protect against the caller
|
||||
// destroying the original matrix.
|
||||
if (!cuDSSObjectInitialized || !reorder_reuse)
|
||||
{
|
||||
if (csr_offsets_d != NULL) { CuMemFree(csr_offsets_d); }
|
||||
CuMemAlloc(&csr_offsets_d, (n_loc + 1) * sizeof(int));
|
||||
if (csr_columns_d != NULL) { CuMemFree(csr_columns_d); }
|
||||
CuMemAlloc(&csr_columns_d, nnz * sizeof(int));
|
||||
CuMemcpyDtoD(csr_offsets_d, csr_offsets, (n_loc + 1) * sizeof(int));
|
||||
CuMemcpyDtoD(csr_columns_d, csr_columns, nnz * sizeof(int));
|
||||
}
|
||||
|
||||
// New cuDSS CSR matrix object and analysis or reuse the one from a previous
|
||||
// matrix
|
||||
if (!cuDSSObjectInitialized || !reorder_reuse)
|
||||
{
|
||||
if (reorder_reuse) // !cuDSSObjectInitialized && reorder_reuse
|
||||
{
|
||||
// NOTE: For CuDSS solver to reuse the reordering (skipping analysis
|
||||
// phase), it needs to access the I and J arrays of the **initial**
|
||||
// matrix. Therefore, we need to copy and keep I and J in device memory.
|
||||
CuMemAlloc(&csr_offsets_d, (n_loc + 1) * sizeof(int));
|
||||
CuMemAlloc(&csr_columns_d, nnz * sizeof(int));
|
||||
|
||||
CuMemcpyDtoD(csr_offsets_d, csr_offsets, (n_loc + 1) * sizeof(int));
|
||||
CuMemcpyDtoD(csr_columns_d, csr_columns, nnz * sizeof(int));
|
||||
|
||||
#if CUDSS_VERSION >= 800
|
||||
MFEM_CUDSS_CHECK(
|
||||
cudssMatrixCreateCsr(
|
||||
@@ -288,21 +308,17 @@ void CuDSSSolver::SetMatrixCuDSS(int *csr_offsets, int *csr_columns,
|
||||
}
|
||||
else // !reorder_reuse
|
||||
{
|
||||
if (cuDSSObjectInitialized)
|
||||
{
|
||||
MFEM_CUDSS_CHECK(cudssMatrixDestroy(*Ac));
|
||||
}
|
||||
#if CUDSS_VERSION >= 800
|
||||
MFEM_CUDSS_CHECK(
|
||||
cudssMatrixCreateCsr(
|
||||
Ac.get(), n_global, n_global, nnz, csr_offsets, NULL,
|
||||
csr_columns, csr_values_d, CUDSS_INT_T, CUDSS_INT_T, CUDSS_REAL_T,
|
||||
Ac.get(), n_global, n_global, nnz, csr_offsets_d, NULL,
|
||||
csr_columns_d, csr_values_d, CUDSS_INT_T, CUDSS_INT_T, CUDSS_REAL_T,
|
||||
mat_type, mview, CUDSS_BASE_ZERO));
|
||||
#else
|
||||
MFEM_CUDSS_CHECK(
|
||||
cudssMatrixCreateCsr(
|
||||
Ac.get(), n_global, n_global, nnz, csr_offsets, NULL,
|
||||
csr_columns, csr_values_d, CUDSS_INT_T, CUDSS_REAL_T,
|
||||
Ac.get(), n_global, n_global, nnz, csr_offsets_d, NULL,
|
||||
csr_columns_d, csr_values_d, CUDSS_INT_T, CUDSS_REAL_T,
|
||||
mat_type, mview, CUDSS_BASE_ZERO));
|
||||
#endif
|
||||
}
|
||||
@@ -326,6 +342,9 @@ void CuDSSSolver::SetMatrixCuDSS(int *csr_offsets, int *csr_columns,
|
||||
// Factorization
|
||||
MFEM_CUDSS_CHECK(cudssExecute(handle, CUDSS_PHASE_FACTORIZATION, solverConfig,
|
||||
solverData, *Ac, yc, xc));
|
||||
|
||||
// In serial, the factorization can execute asynchronously.
|
||||
MFEM_STREAM_SYNC;
|
||||
}
|
||||
|
||||
void CuDSSSolver::SetOperator(const Operator &op)
|
||||
@@ -360,6 +379,7 @@ void CuDSSSolver::SetNumRHS(int nrhs_) const
|
||||
if (nrhs > 0)
|
||||
{
|
||||
// Destroy the previous RHS vector and solution vector
|
||||
MFEM_STREAM_SYNC;
|
||||
MFEM_CUDSS_CHECK(cudssMatrixDestroy(xc));
|
||||
MFEM_CUDSS_CHECK(cudssMatrixDestroy(yc));
|
||||
}
|
||||
|
||||
+1
-2
@@ -157,8 +157,7 @@ private:
|
||||
mutable int nrhs = 0; // the number of the RHSs
|
||||
int nnz = 0; // the number of non zeros
|
||||
|
||||
// copy and keep the I and J arrays in device memory when skipping analysis
|
||||
// phase
|
||||
// copy and keep the I and J arrays in device memory
|
||||
void *csr_offsets_d = NULL; // copy and keep I in device
|
||||
void *csr_columns_d = NULL; // copy and keep J in device
|
||||
void *csr_values_d = NULL; // copy and keep csr data in device
|
||||
|
||||
@@ -1136,6 +1136,17 @@ private:
|
||||
public:
|
||||
DenseTensor() : ni(0), nj(0), nk(0) { }
|
||||
|
||||
DenseTensor(const DenseTensor &other)
|
||||
: tdata(other.tdata), ni(other.ni), nj(other.nj), nk(other.nk) { }
|
||||
|
||||
DenseTensor(DenseTensor &&other)
|
||||
: tdata(std::move(other.tdata)), ni(other.ni), nj(other.nj), nk(other.nk)
|
||||
{
|
||||
// Reset other; other.tdata is reset in Array<T> move constructror.
|
||||
other.Mk.ClearExternalData();
|
||||
other.ni = other.nj = other.nk = 0;
|
||||
}
|
||||
|
||||
DenseTensor(int i, int j, int k) : tdata(i*j*k), ni(i), nj(j), nk(k) { }
|
||||
|
||||
DenseTensor(real_t *d, int i, int j, int k)
|
||||
@@ -1144,6 +1155,33 @@ public:
|
||||
DenseTensor(int i, int j, int k, MemoryType mt)
|
||||
: tdata(i*j*k, mt), ni(i), nj(j), nk(k) { }
|
||||
|
||||
DenseTensor &operator=(const DenseTensor &other)
|
||||
{
|
||||
if (this == &other) { return *this; }
|
||||
Mk.ClearExternalData();
|
||||
tdata = other.tdata;
|
||||
ni = other.ni;
|
||||
nj = other.nj;
|
||||
nk = other.nk;
|
||||
return *this;
|
||||
}
|
||||
|
||||
DenseTensor &operator=(DenseTensor &&other)
|
||||
{
|
||||
if (this == &other) { return *this; }
|
||||
Mk.ClearExternalData();
|
||||
tdata = std::move(other.tdata);
|
||||
ni = other.ni;
|
||||
nj = other.nj;
|
||||
nk = other.nk;
|
||||
|
||||
// Reset other; other.tdata is reset in Array<T> move assignment.
|
||||
other.Mk.ClearExternalData();
|
||||
other.ni = other.nj = other.nk = 0;
|
||||
|
||||
return *this;
|
||||
}
|
||||
|
||||
int SizeI() const { return ni; }
|
||||
int SizeJ() const { return nj; }
|
||||
int SizeK() const { return nk; }
|
||||
|
||||
@@ -5842,6 +5842,10 @@ void HypreAMS::MakeGradientAndInterpolation(
|
||||
{
|
||||
grad->AddTraceFaceInterpolator(new GradientInterpolator);
|
||||
}
|
||||
else if (dynamic_cast<const RT_FECollection *>(edge_fec))
|
||||
{
|
||||
grad->AddDomainInterpolator(new CurlInterpolator);
|
||||
}
|
||||
else
|
||||
{
|
||||
grad->AddDomainInterpolator(new GradientInterpolator);
|
||||
|
||||
@@ -810,6 +810,7 @@ MINIAPPS_SUBDIRS = dpg/util hooke/operators hooke/preconditioners \
|
||||
hooke/materials hooke/kernels
|
||||
FORMAT_FILES += $(foreach dir,$(TESTS_SUBDIRS),tests/$(dir)/*.?pp)
|
||||
FORMAT_FILES += $(foreach dir,$(UNIT_TESTS_SUBDIRS),tests/unit/$(dir)/*.?pp)
|
||||
FORMAT_FILES += tests/unit/fem/specializations/*.?pp
|
||||
FORMAT_FILES += $(foreach dir,$(MINIAPPS_SUBDIRS),miniapps/$(dir)/*.?pp)
|
||||
FORMAT_FILES += config/cmake/config.hpp.in config/config.hpp.in mfem*.hpp
|
||||
FORMAT_EXCLUDE = general/tinyxml2.cpp tests/unit/catch.hpp
|
||||
|
||||
+101
-5
@@ -667,9 +667,84 @@ void Mesh::GetEdgeTransformation(int EdgeNo,
|
||||
}
|
||||
EdTr->SetFE(edge_el);
|
||||
}
|
||||
else
|
||||
else // L2 Nodes (e.g., periodic mesh), go through the face containing the edge
|
||||
{
|
||||
MFEM_ABORT("Not implemented.");
|
||||
// Search for a face that contains this edge
|
||||
GetEdgeFaceTable();
|
||||
|
||||
Array<int> faces_e;
|
||||
edge_face->GetRow(EdgeNo, faces_e);
|
||||
|
||||
MFEM_VERIFY(faces_e.Size() > 0, "Edge not found in any face!");
|
||||
const int face_no = faces_e[0];
|
||||
|
||||
// Get edge local index and orientation
|
||||
Array<int> edges_f, oris_f;
|
||||
GetFaceEdges(face_no, edges_f, oris_f);
|
||||
const int local_idx = edges_f.Find(EdgeNo);
|
||||
MFEM_ASSERT(local_idx >= 0, "Edge not found on the face!");
|
||||
const int edge_ori = oris_f[local_idx] > 0 ? 0 : 1;
|
||||
|
||||
// Get face information
|
||||
const FaceInfo &face_info = faces_info[face_no];
|
||||
|
||||
// Get transformation from face to edge
|
||||
IntegrationPointTransformation LocEdge;
|
||||
int edge_info = EncodeFaceInfo(local_idx, edge_ori);
|
||||
Element::Type face_type = GetFaceElementType(face_no);
|
||||
|
||||
switch (face_type)
|
||||
{
|
||||
case Element::TRIANGLE:
|
||||
GetLocalSegToTriTransformation(LocEdge.Transf, edge_info);
|
||||
break;
|
||||
case Element::QUADRILATERAL:
|
||||
GetLocalSegToQuadTransformation(LocEdge.Transf, edge_info);
|
||||
break;
|
||||
default:
|
||||
MFEM_ABORT("Unsupported face type for edge transformation!");
|
||||
}
|
||||
|
||||
// Get edge element
|
||||
const int order = Nodes->FESpace()->GetElementOrder(face_info.Elem1No);
|
||||
const L2_FECollection *l2_fec = dynamic_cast<const L2_FECollection*>
|
||||
(Nodes->FESpace()->FEColl());
|
||||
if (l2_fec)
|
||||
{
|
||||
// L2 elements do not have a defined trace space
|
||||
if (!EdgeTransfElement || EdgeTransfElement->GetOrder() != order
|
||||
|| EdgeTransfElement->GetBasisType() != l2_fec->GetBasisType())
|
||||
{
|
||||
EdgeTransfElement = make_unique<L2_SegmentElement>(
|
||||
order, l2_fec->GetBasisType());
|
||||
}
|
||||
edge_el = EdgeTransfElement.get();
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unsupported finite element collection.");
|
||||
}
|
||||
|
||||
// Map edge nodes to face reference space
|
||||
IntegrationRule face_ir(edge_el->GetDof());
|
||||
LocEdge.Transform(edge_el->GetNodes(), face_ir);
|
||||
|
||||
// Then, map from face to element
|
||||
IntegrationPointTransformation Loc1;
|
||||
GetLocalFaceTransformation(face_type,
|
||||
GetElementType(face_info.Elem1No),
|
||||
Loc1.Transf, face_info.Elem1Inf);
|
||||
|
||||
IntegrationRule elem_ir(edge_el->GetDof());
|
||||
Loc1.Transf.ElementNo = face_info.Elem1No;
|
||||
Loc1.Transf.ElementType = ElementTransformation::ELEMENT;
|
||||
Loc1.Transf.mesh = this;
|
||||
Loc1.Transform(face_ir, elem_ir);
|
||||
|
||||
// Finally, get the physical coordinates
|
||||
Nodes->GetVectorValues(Loc1.Transf, elem_ir, pm);
|
||||
|
||||
EdTr->SetFE(edge_el);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1824,8 +1899,8 @@ void Mesh::Init()
|
||||
|
||||
void Mesh::InitTables()
|
||||
{
|
||||
el_to_edge =
|
||||
el_to_face = el_to_el = bel_to_edge = face_edge = edge_vertex = NULL;
|
||||
el_to_edge = el_to_face = el_to_el = bel_to_edge = NULL;
|
||||
face_edge = edge_face = edge_vertex = NULL;
|
||||
face_to_elem = NULL;
|
||||
}
|
||||
|
||||
@@ -1848,6 +1923,7 @@ void Mesh::DestroyTables()
|
||||
}
|
||||
|
||||
delete face_edge;
|
||||
delete edge_face;
|
||||
delete edge_vertex;
|
||||
|
||||
delete face_to_elem;
|
||||
@@ -1921,6 +1997,7 @@ void Mesh::ResetLazyData()
|
||||
{
|
||||
delete el_to_el; el_to_el = NULL;
|
||||
delete face_edge; face_edge = NULL;
|
||||
delete edge_face; edge_face = NULL;
|
||||
delete face_to_elem; face_to_elem = NULL;
|
||||
delete edge_vertex; edge_vertex = NULL;
|
||||
DeleteGeometricFactors();
|
||||
@@ -2845,6 +2922,7 @@ void Mesh::ReorderElements(const Array<int> &ordering, bool reorder_vertices)
|
||||
// boundary element ordering
|
||||
// - el_to_el - no need to rebuild
|
||||
// - face_edge - no need to rebuild
|
||||
// - edge_face - no need to rebuild
|
||||
// - edge_vertex - no need to rebuild
|
||||
// - geom_factors - no need to rebuild
|
||||
|
||||
@@ -4598,8 +4676,9 @@ Mesh::Mesh(const Mesh &mesh, bool copy_nodes)
|
||||
// Do NOT copy the element-to-element Table, el_to_el
|
||||
el_to_el = NULL;
|
||||
|
||||
// Do NOT copy the face-to-edge Table, face_edge
|
||||
// Do NOT copy the face-to-edge Table, face_edge and edge_face
|
||||
face_edge = NULL;
|
||||
edge_face = NULL;
|
||||
face_to_elem = NULL;
|
||||
|
||||
// Copy the edge-to-vertex Table, edge_vertex
|
||||
@@ -8094,6 +8173,22 @@ Table *Mesh::GetFaceEdgeTable() const
|
||||
return (face_edge);
|
||||
}
|
||||
|
||||
Table *Mesh::GetEdgeFaceTable() const
|
||||
{
|
||||
if (edge_face)
|
||||
{
|
||||
return edge_face;
|
||||
}
|
||||
|
||||
if (Dim != 3)
|
||||
{
|
||||
return NULL;
|
||||
}
|
||||
|
||||
edge_face = Transpose(*GetFaceEdgeTable());
|
||||
return edge_face;
|
||||
}
|
||||
|
||||
Table *Mesh::GetEdgeVertexTable() const
|
||||
{
|
||||
if (edge_vertex)
|
||||
@@ -11452,6 +11547,7 @@ void Mesh::Swap(Mesh& other, bool non_geometry)
|
||||
mfem::Swap(bel_to_edge, other.bel_to_edge);
|
||||
mfem::Swap(be_to_face, other.be_to_face);
|
||||
mfem::Swap(face_edge, other.face_edge);
|
||||
mfem::Swap(edge_face, other.edge_face);
|
||||
mfem::Swap(face_to_elem, other.face_to_elem);
|
||||
mfem::Swap(edge_vertex, other.edge_vertex);
|
||||
|
||||
|
||||
+8
-1
@@ -250,16 +250,18 @@ protected:
|
||||
Table *bel_to_edge; // for 3D only
|
||||
|
||||
// Note that the following tables are owned by this class and should not be
|
||||
// deleted by the caller. Of these three tables, only face_edge and
|
||||
// deleted by the caller. Of these four tables, only face_edge, edge_face and
|
||||
// edge_vertex are returned by access functions.
|
||||
mutable Table *face_to_elem; // Used by FindFaceNeighbors, not returned.
|
||||
mutable Table *face_edge; // Returned by GetFaceEdgeTable().
|
||||
mutable Table *edge_face; // Returned by GetEdgeFaceTable().
|
||||
mutable Table *edge_vertex; // Returned by GetEdgeVertexTable().
|
||||
|
||||
IsoparametricTransformation Transformation, Transformation2;
|
||||
IsoparametricTransformation BdrTransformation;
|
||||
IsoparametricTransformation FaceTransformation, EdgeTransformation;
|
||||
FaceElementTransformations FaceElemTr;
|
||||
mutable std::unique_ptr<L2_SegmentElement> EdgeTransfElement;
|
||||
|
||||
// refinement embeddings for forward compatibility with NCMesh
|
||||
mutable CoarseFineTransformations CoarseFineTr;
|
||||
@@ -1731,6 +1733,11 @@ public:
|
||||
/// @note The returned object should NOT be deleted by the caller.
|
||||
Table *GetFaceEdgeTable() const;
|
||||
|
||||
/// Returns the edge-to-face Table (3D)
|
||||
///
|
||||
/// @note The returned object should NOT be deleted by the caller.
|
||||
Table *GetEdgeFaceTable() const;
|
||||
|
||||
/// Returns the edge-to-vertex Table (3D)
|
||||
///
|
||||
/// @note The returned object should NOT be deleted by the caller.
|
||||
|
||||
@@ -4866,6 +4866,13 @@ void ParMesh::Print(std::ostream &os, const std::string &comments) const
|
||||
return;
|
||||
}
|
||||
|
||||
if (pncmesh && pncmesh->using_scaling)
|
||||
{
|
||||
// For nodes scaling, we write the file in the format MFEM NC mesh v1.1.
|
||||
Printer(os, "", comments);
|
||||
return;
|
||||
}
|
||||
|
||||
const Array<int>* s2l_face;
|
||||
if (!pncmesh)
|
||||
{
|
||||
|
||||
+138
-62
@@ -28,6 +28,48 @@ namespace mfem
|
||||
|
||||
using namespace bin_io;
|
||||
|
||||
static int GetHexEdgeSplit(const int* nodes, int v1, int v2);
|
||||
|
||||
static bool SameSplitScale(real_t a, real_t b)
|
||||
{
|
||||
#ifdef MFEM_USE_DOUBLE
|
||||
constexpr real_t rel_tol = 1.0e-8;
|
||||
#else
|
||||
constexpr real_t rel_tol = 1.0e-5;
|
||||
#endif
|
||||
return std::abs(a - b) <= rel_tol *
|
||||
std::max(real_t(1.0), std::max(std::abs(a), std::abs(b)));
|
||||
}
|
||||
|
||||
static real_t DirectedHexEdgeScale(const int* nodes, const Refinement &ref,
|
||||
int v0, int v1)
|
||||
{
|
||||
const int dir = GetHexEdgeSplit(nodes, v0, v1);
|
||||
static const int split_edges[3][4][2] =
|
||||
{
|
||||
{{0, 1}, {3, 2}, {4, 5}, {7, 6}},
|
||||
{{1, 2}, {0, 3}, {5, 6}, {4, 7}},
|
||||
{{0, 4}, {1, 5}, {2, 6}, {3, 7}}
|
||||
};
|
||||
|
||||
for (int i = 0; i < 4; i++)
|
||||
{
|
||||
const int a = nodes[split_edges[dir][i][0]];
|
||||
const int b = nodes[split_edges[dir][i][1]];
|
||||
if (a == v0 && b == v1)
|
||||
{
|
||||
return ref.s[dir];
|
||||
}
|
||||
if (a == v1 && b == v0)
|
||||
{
|
||||
return 1.0 - ref.s[dir];
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_ABORT("Shared face edge does not match the refinement direction.");
|
||||
return 0.0;
|
||||
}
|
||||
|
||||
ParNCMesh::ParNCMesh(MPI_Comm comm, const NCMesh &ncmesh,
|
||||
const int *partitioning)
|
||||
: NCMesh(ncmesh)
|
||||
@@ -1555,7 +1597,7 @@ bool ParNCMesh::AnisotropicConflict(const Array<Refinement> &refinements,
|
||||
ElementNeighborProcessors(elem, ranks);
|
||||
for (int j = 0; j < ranks.Size(); j++)
|
||||
{
|
||||
send_ref[ranks[j]].AddRefinement(elem, ref.GetType());
|
||||
send_ref[ranks[j]].AddRefinement(elem, ref);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1576,8 +1618,8 @@ bool ParNCMesh::AnisotropicConflict(const Array<Refinement> &refinements,
|
||||
for (int i = 0; i < refinements.Size(); i++)
|
||||
{
|
||||
const Refinement &ref = refinements[i];
|
||||
CheckRefinement(leaf_elements[ref.index], ref.GetType(), refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefinement(leaf_elements[ref.index], ref, refinements, elemToRef,
|
||||
conflicts);
|
||||
}
|
||||
|
||||
// Receive (ghost layer) refinements from all neighbors
|
||||
@@ -1593,7 +1635,9 @@ bool ParNCMesh::AnisotropicConflict(const Array<Refinement> &refinements,
|
||||
// check the ghost refinements
|
||||
for (int i = 0; i < msg.Size(); i++)
|
||||
{
|
||||
CheckRefinement(msg.elements[i], msg.values[i], refinements, elemToRef,
|
||||
Refinement ghost_ref(msg.elements[i], msg.values[i].ref_type);
|
||||
ghost_ref.SetScaleForType(msg.values[i].scale);
|
||||
CheckRefinement(msg.elements[i], ghost_ref, refinements, elemToRef,
|
||||
conflicts);
|
||||
}
|
||||
}
|
||||
@@ -1749,7 +1793,7 @@ int FindHexFace(const int* no, int vn1, int vn2, int vn3, int vn4)
|
||||
|
||||
// Assumption: v1 and v2 are indices of hex vertices connected by an edge.
|
||||
// The return value is {0,1,2} denoting split {X,Y,Z}.
|
||||
int GetHexEdgeSplit(const int* nodes, int v1, int v2)
|
||||
static int GetHexEdgeSplit(const int* nodes, int v1, int v2)
|
||||
{
|
||||
Array<int> v(2);
|
||||
v[0] = v1;
|
||||
@@ -1780,7 +1824,8 @@ int GetHexEdgeSplit(const int* nodes, int v1, int v2)
|
||||
return edgeDir[edge];
|
||||
}
|
||||
|
||||
void ParNCMesh::CheckRefAnisoFace(int elem, int vn1, int vn2, int vn3, int vn4,
|
||||
void ParNCMesh::CheckRefAnisoFace(const Refinement &ref, int elem,
|
||||
int vn1, int vn2, int vn3, int vn4,
|
||||
const Array<Refinement> &refinements,
|
||||
const std::map<int, int> &elemToRef,
|
||||
std::set<int> &conflicts)
|
||||
@@ -1798,11 +1843,11 @@ void ParNCMesh::CheckRefAnisoFace(int elem, int vn1, int vn2, int vn3, int vn4,
|
||||
if (elemToRef.count(nghbIndex) > 0)
|
||||
{
|
||||
const int refIndex = elemToRef.at(nghbIndex);
|
||||
const Refinement& ref = refinements[refIndex];
|
||||
const Refinement& nghb_ref = refinements[refIndex];
|
||||
|
||||
bool refDir[3];
|
||||
for (int i=0; i<3; ++i)
|
||||
refDir[i] = ref.s[i] > real_t{0};
|
||||
refDir[i] = nghb_ref.s[i] > real_t{0};
|
||||
|
||||
const int localFace = FindHexFace(nghb.node, vn1, vn2, vn3, vn4);
|
||||
const int faceDir = GetHexFaceDir(localFace);
|
||||
@@ -1834,30 +1879,50 @@ void ParNCMesh::CheckRefAnisoFace(int elem, int vn1, int vn2, int vn3, int vn4,
|
||||
MFEM_ASSERT(cnt == 2 && hexSplitOnFace >= 0, "");
|
||||
|
||||
const int edgeSplit = GetHexEdgeSplit(nghb.node, vn1, vn2);
|
||||
if (edgeSplit != hexSplitOnFace) { conflicts.insert(refIndex); }
|
||||
if (edgeSplit != hexSplitOnFace)
|
||||
{
|
||||
conflicts.insert(refIndex);
|
||||
}
|
||||
else
|
||||
{
|
||||
const real_t elem_scale =
|
||||
DirectedHexEdgeScale(elements[elem].node, ref, vn1, vn2);
|
||||
const real_t nghb_scale =
|
||||
DirectedHexEdgeScale(nghb.node, nghb_ref, vn1, vn2);
|
||||
if (!SameSplitScale(elem_scale, nghb_scale))
|
||||
{
|
||||
conflicts.insert(refIndex);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// The else case is that the neighbor is not refined, so there is no need to
|
||||
// check for conflicts.
|
||||
}
|
||||
|
||||
void ParNCMesh::CheckRefIsoFace(int elem, int vn1, int vn2, int vn3, int vn4,
|
||||
void ParNCMesh::CheckRefIsoFace(const Refinement &ref, int elem,
|
||||
int vn1, int vn2, int vn3, int vn4,
|
||||
int en1, int en2, int en3, int en4,
|
||||
const Array<Refinement> &refinements,
|
||||
const std::map<int, int> &elemToRef,
|
||||
std::set<int> &conflicts)
|
||||
{
|
||||
CheckRefAnisoFace(elem, vn1, vn2, en2, en4, refinements, elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, en4, en2, vn3, vn4, refinements, elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, vn4, vn1, en1, en3, refinements, elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, en3, en1, vn2, vn3, refinements, elemToRef, conflicts);
|
||||
CheckRefAnisoFace(ref, elem, vn1, vn2, en2, en4, refinements, elemToRef,
|
||||
conflicts);
|
||||
CheckRefAnisoFace(ref, elem, en4, en2, vn3, vn4, refinements, elemToRef,
|
||||
conflicts);
|
||||
CheckRefAnisoFace(ref, elem, vn4, vn1, en1, en3, refinements, elemToRef,
|
||||
conflicts);
|
||||
CheckRefAnisoFace(ref, elem, en3, en1, vn2, vn3, refinements, elemToRef,
|
||||
conflicts);
|
||||
}
|
||||
|
||||
void ParNCMesh::CheckRefinement(int elem, char ref_type,
|
||||
void ParNCMesh::CheckRefinement(int elem, const Refinement &ref,
|
||||
const Array<Refinement> &refinements,
|
||||
const std::map<int, int> &elemToRef,
|
||||
std::set<int> &conflicts)
|
||||
{
|
||||
const char ref_type = ref.GetType();
|
||||
const Element &el = elements[elem];
|
||||
MFEM_ASSERT(el.geom == Geometry::CUBE && el.ref_type == 0,
|
||||
"Element must be an unrefined hexahedron");
|
||||
@@ -1868,46 +1933,46 @@ void ParNCMesh::CheckRefinement(int elem, char ref_type,
|
||||
// This follows the logic of NCMesh::RefineElement().
|
||||
if (ref_type == Refinement::X) // split along X axis
|
||||
{
|
||||
CheckRefAnisoFace(elem, no[0], no[1], no[5], no[4], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[0], no[1], no[5], no[4], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[2], no[3], no[7], no[6], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[2], no[3], no[7], no[6], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[4], no[5], no[6], no[7], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[4], no[5], no[6], no[7], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[3], no[2], no[1], no[0], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[3], no[2], no[1], no[0], refinements,
|
||||
elemToRef, conflicts);
|
||||
}
|
||||
else if (ref_type == Refinement::Y) // split along Y axis
|
||||
{
|
||||
CheckRefAnisoFace(elem, no[1], no[2], no[6], no[5], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[1], no[2], no[6], no[5], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[3], no[0], no[4], no[7], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[3], no[0], no[4], no[7], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[5], no[6], no[7], no[4], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[5], no[6], no[7], no[4], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[0], no[3], no[2], no[1], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[0], no[3], no[2], no[1], refinements,
|
||||
elemToRef, conflicts);
|
||||
}
|
||||
else if (ref_type == Refinement::Z) // split along Z axis
|
||||
{
|
||||
CheckRefAnisoFace(elem, no[4], no[0], no[1], no[5], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[4], no[0], no[1], no[5], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[5], no[1], no[2], no[6], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[5], no[1], no[2], no[6], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[6], no[2], no[3], no[7], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[6], no[2], no[3], no[7], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[7], no[3], no[0], no[4], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[7], no[3], no[0], no[4], refinements,
|
||||
elemToRef, conflicts);
|
||||
}
|
||||
else if (ref_type == Refinement::XY) // XY split
|
||||
{
|
||||
CheckRefAnisoFace(elem, no[0], no[1], no[5], no[4], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[0], no[1], no[5], no[4], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[1], no[2], no[6], no[5], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[1], no[2], no[6], no[5], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[2], no[3], no[7], no[6], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[2], no[3], no[7], no[6], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[3], no[0], no[4], no[7], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[3], no[0], no[4], no[7], refinements,
|
||||
elemToRef, conflicts);
|
||||
|
||||
const int mid01 = GetMidEdgeNode(no[0], no[1]);
|
||||
@@ -1920,20 +1985,20 @@ void ParNCMesh::CheckRefinement(int elem, char ref_type,
|
||||
const int mid67 = GetMidEdgeNode(no[6], no[7]);
|
||||
const int mid74 = GetMidEdgeNode(no[7], no[4]);
|
||||
|
||||
CheckRefIsoFace(elem, no[3], no[2], no[1], no[0], mid23, mid12, mid01,
|
||||
CheckRefIsoFace(ref, elem, no[3], no[2], no[1], no[0], mid23, mid12, mid01,
|
||||
mid30, refinements, elemToRef, conflicts);
|
||||
CheckRefIsoFace(elem, no[4], no[5], no[6], no[7], mid45, mid56, mid67,
|
||||
CheckRefIsoFace(ref, elem, no[4], no[5], no[6], no[7], mid45, mid56, mid67,
|
||||
mid74, refinements, elemToRef, conflicts);
|
||||
}
|
||||
else if (ref_type == Refinement::XZ) // XZ split
|
||||
{
|
||||
CheckRefAnisoFace(elem, no[3], no[2], no[1], no[0], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[3], no[2], no[1], no[0], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[2], no[6], no[5], no[1], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[2], no[6], no[5], no[1], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[6], no[7], no[4], no[5], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[6], no[7], no[4], no[5], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[7], no[3], no[0], no[4], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[7], no[3], no[0], no[4], refinements,
|
||||
elemToRef, conflicts);
|
||||
|
||||
const int mid01 = GetMidEdgeNode(no[0], no[1]);
|
||||
@@ -1946,9 +2011,9 @@ void ParNCMesh::CheckRefinement(int elem, char ref_type,
|
||||
const int mid26 = GetMidEdgeNode(no[2], no[6]);
|
||||
const int mid37 = GetMidEdgeNode(no[3], no[7]);
|
||||
|
||||
CheckRefIsoFace(elem, no[0], no[1], no[5], no[4], mid01, mid15, mid45,
|
||||
CheckRefIsoFace(ref, elem, no[0], no[1], no[5], no[4], mid01, mid15, mid45,
|
||||
mid04, refinements, elemToRef, conflicts);
|
||||
CheckRefIsoFace(elem, no[2], no[3], no[7], no[6], mid23, mid37, mid67,
|
||||
CheckRefIsoFace(ref, elem, no[2], no[3], no[7], no[6], mid23, mid37, mid67,
|
||||
mid26, refinements, elemToRef, conflicts);
|
||||
}
|
||||
else if (ref_type == Refinement::YZ) // YZ split
|
||||
@@ -1963,18 +2028,18 @@ void ParNCMesh::CheckRefinement(int elem, char ref_type,
|
||||
const int mid26 = GetMidEdgeNode(no[2], no[6]);
|
||||
const int mid37 = GetMidEdgeNode(no[3], no[7]);
|
||||
|
||||
CheckRefAnisoFace(elem, no[4], no[0], no[1], no[5], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[4], no[0], no[1], no[5], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[0], no[3], no[2], no[1], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[0], no[3], no[2], no[1], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[3], no[7], no[6], no[2], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[3], no[7], no[6], no[2], refinements,
|
||||
elemToRef, conflicts);
|
||||
CheckRefAnisoFace(elem, no[7], no[4], no[5], no[6], refinements,
|
||||
CheckRefAnisoFace(ref, elem, no[7], no[4], no[5], no[6], refinements,
|
||||
elemToRef, conflicts);
|
||||
|
||||
CheckRefIsoFace(elem, no[1], no[2], no[6], no[5], mid12, mid26, mid56,
|
||||
CheckRefIsoFace(ref, elem, no[1], no[2], no[6], no[5], mid12, mid26, mid56,
|
||||
mid15, refinements, elemToRef, conflicts);
|
||||
CheckRefIsoFace(elem, no[3], no[0], no[4], no[7], mid30, mid04, mid74,
|
||||
CheckRefIsoFace(ref, elem, no[3], no[0], no[4], no[7], mid30, mid04, mid74,
|
||||
mid37, refinements, elemToRef, conflicts);
|
||||
}
|
||||
else if (ref_type == Refinement::XYZ) // XYZ split
|
||||
@@ -1994,17 +2059,17 @@ void ParNCMesh::CheckRefinement(int elem, char ref_type,
|
||||
const int mid26 = GetMidEdgeNode(no[2], no[6]);
|
||||
const int mid37 = GetMidEdgeNode(no[3], no[7]);
|
||||
|
||||
CheckRefIsoFace(elem, no[3], no[2], no[1], no[0], mid23, mid12, mid01,
|
||||
CheckRefIsoFace(ref, elem, no[3], no[2], no[1], no[0], mid23, mid12, mid01,
|
||||
mid30, refinements, elemToRef, conflicts);
|
||||
CheckRefIsoFace(elem, no[0], no[1], no[5], no[4], mid01, mid15, mid45,
|
||||
CheckRefIsoFace(ref, elem, no[0], no[1], no[5], no[4], mid01, mid15, mid45,
|
||||
mid04, refinements, elemToRef, conflicts);
|
||||
CheckRefIsoFace(elem, no[1], no[2], no[6], no[5], mid12, mid26, mid56,
|
||||
CheckRefIsoFace(ref, elem, no[1], no[2], no[6], no[5], mid12, mid26, mid56,
|
||||
mid15, refinements, elemToRef, conflicts);
|
||||
CheckRefIsoFace(elem, no[2], no[3], no[7], no[6], mid23, mid37, mid67,
|
||||
CheckRefIsoFace(ref, elem, no[2], no[3], no[7], no[6], mid23, mid37, mid67,
|
||||
mid26, refinements, elemToRef, conflicts);
|
||||
CheckRefIsoFace(elem, no[3], no[0], no[4], no[7], mid30, mid04, mid74,
|
||||
CheckRefIsoFace(ref, elem, no[3], no[0], no[4], no[7], mid30, mid04, mid74,
|
||||
mid37, refinements, elemToRef, conflicts);
|
||||
CheckRefIsoFace(elem, no[4], no[5], no[6], no[7], mid45, mid56, mid67,
|
||||
CheckRefIsoFace(ref, elem, no[4], no[5], no[6], no[7], mid45, mid56, mid67,
|
||||
mid74, refinements, elemToRef, conflicts);
|
||||
}
|
||||
else
|
||||
@@ -2053,7 +2118,7 @@ void ParNCMesh::Refine(const Array<Refinement> &refinements)
|
||||
ElementNeighborProcessors(elem, ranks);
|
||||
for (int j = 0; j < ranks.Size(); j++)
|
||||
{
|
||||
send_ref[ranks[j]].AddRefinement(elem, ref.GetType());
|
||||
send_ref[ranks[j]].AddRefinement(elem, ref);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2063,8 +2128,9 @@ void ParNCMesh::Refine(const Array<Refinement> &refinements)
|
||||
// do local refinements
|
||||
for (int i = 0; i < refinements.Size(); i++)
|
||||
{
|
||||
const Refinement &ref = refinements[i];
|
||||
NCMesh::RefineElement(leaf_elements[ref.index], ref.GetType());
|
||||
Refinement ref_i = refinements[i];
|
||||
ref_i.index = leaf_elements[refinements[i].index];
|
||||
NCMesh::RefineElement(ref_i);
|
||||
}
|
||||
|
||||
// receive (ghost layer) refinements from all neighbors
|
||||
@@ -2080,7 +2146,9 @@ void ParNCMesh::Refine(const Array<Refinement> &refinements)
|
||||
// do the ghost refinements
|
||||
for (int i = 0; i < msg.Size(); i++)
|
||||
{
|
||||
NCMesh::RefineElement(msg.elements[i], msg.values[i]);
|
||||
Refinement ghost_ref(msg.elements[i], msg.values[i].ref_type);
|
||||
ghost_ref.SetScaleForType(msg.values[i].scale);
|
||||
NCMesh::RefineElement(ghost_ref);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2556,7 +2624,8 @@ void ParNCMesh::RedistributeElements(Array<int> &new_ranks, int target_elements,
|
||||
for (int i = 0; i < rank_neighbors.Size(); i++)
|
||||
{
|
||||
int elem = rank_neighbors[i];
|
||||
msg.AddElementRank(elem, new_ranks[elements[elem].index]);
|
||||
const Element &el = elements[elem];
|
||||
msg.AddElement(elem, new_ranks[el.index], el.attribute);
|
||||
}
|
||||
|
||||
msg.Isend(rank, MyComm);
|
||||
@@ -2579,7 +2648,9 @@ void ParNCMesh::RedistributeElements(Array<int> &new_ranks, int target_elements,
|
||||
{
|
||||
int ghost_index = elements[msg.elements[i]].index;
|
||||
MFEM_ASSERT(element_type[ghost_index] == 2, "");
|
||||
new_ranks[ghost_index] = msg.values[i];
|
||||
const ElementRankAndAttribute &value = msg.values[i];
|
||||
new_ranks[ghost_index] = value.rank;
|
||||
elements[msg.elements[i]].attribute = value.attribute;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2650,7 +2721,7 @@ void ParNCMesh::RedistributeElements(Array<int> &new_ranks, int target_elements,
|
||||
|
||||
if ((element_type[el.index] & 1) || el.rank != rank)
|
||||
{
|
||||
msg.AddElementRank(elem, el.rank);
|
||||
msg.AddElement(elem, el.rank, el.attribute);
|
||||
}
|
||||
// NOTE: we skip 'ghosts' that are of the receiver's rank because
|
||||
// they are not really ghosts and would get sent multiple times,
|
||||
@@ -2702,10 +2773,12 @@ void ParNCMesh::RedistributeElements(Array<int> &new_ranks, int target_elements,
|
||||
|
||||
for (int i = 0; i < msg.Size(); i++)
|
||||
{
|
||||
int elem_rank = msg.values[i];
|
||||
elements[msg.elements[i]].rank = elem_rank;
|
||||
const ElementRankAndAttribute &value = msg.values[i];
|
||||
Element &el = elements[msg.elements[i]];
|
||||
el.rank = value.rank;
|
||||
el.attribute = value.attribute;
|
||||
|
||||
if (elem_rank == MyRank) { received_elements++; }
|
||||
if (value.rank == MyRank) { received_elements++; }
|
||||
}
|
||||
|
||||
// save the ranks we received from, for later use in RecvRebalanceDofs
|
||||
@@ -2741,7 +2814,10 @@ void ParNCMesh::RedistributeElements(Array<int> &new_ranks, int target_elements,
|
||||
|
||||
for (int i = 0; i < msg.Size(); i++)
|
||||
{
|
||||
elements[msg.elements[i]].rank = msg.values[i];
|
||||
const ElementRankAndAttribute &value = msg.values[i];
|
||||
Element &el = elements[msg.elements[i]];
|
||||
el.rank = value.rank;
|
||||
el.attribute = value.attribute;
|
||||
}
|
||||
|
||||
// save the ranks we received from, for later use in RecvRebalanceDofs
|
||||
|
||||
+42
-15
@@ -497,11 +497,27 @@ protected: // implementation
|
||||
/** Used by ParNCMesh::Refine() to inform neighbors about refinements at
|
||||
* the processor boundary. This keeps their ghost layers synchronized.
|
||||
*/
|
||||
class NeighborRefinementMessage : public ElementValueMessage<char, false,
|
||||
VarMessageTag::NEIGHBOR_REFINEMENT_VM>
|
||||
struct NeighborRefinement
|
||||
{
|
||||
char ref_type;
|
||||
real_t scale[3];
|
||||
};
|
||||
|
||||
class NeighborRefinementMessage
|
||||
: public ElementValueMessage<NeighborRefinement, false,
|
||||
VarMessageTag::NEIGHBOR_REFINEMENT_VM>
|
||||
{
|
||||
public:
|
||||
void AddRefinement(int elem, char ref_type) { Add(elem, ref_type); }
|
||||
void AddRefinement(int elem, const Refinement &ref)
|
||||
{
|
||||
NeighborRefinement data{};
|
||||
data.ref_type = ref.GetType();
|
||||
for (int i = 0; i < 3; i++)
|
||||
{
|
||||
data.scale[i] = ref.s[i];
|
||||
}
|
||||
Add(elem, data);
|
||||
}
|
||||
typedef std::map<int, NeighborRefinementMessage> Map;
|
||||
};
|
||||
|
||||
@@ -515,26 +531,36 @@ protected: // implementation
|
||||
typedef std::map<int, NeighborDerefinementMessage> Map;
|
||||
};
|
||||
|
||||
/** Used in Step 2 of Rebalance() to synchronize new rank assignments in
|
||||
* the ghost layer.
|
||||
struct ElementRankAndAttribute
|
||||
{
|
||||
int rank;
|
||||
int attribute;
|
||||
};
|
||||
|
||||
/** Used in RedistributeElements() to synchronize new rank assignments and
|
||||
* element attributes in the ghost layer.
|
||||
*/
|
||||
class NeighborElementRankMessage : public ElementValueMessage<int, false,
|
||||
class NeighborElementRankMessage :
|
||||
public ElementValueMessage<ElementRankAndAttribute, false,
|
||||
VarMessageTag::NEIGHBOR_ELEMENT_RANK_VM>
|
||||
{
|
||||
public:
|
||||
void AddElementRank(int elem, int rank) { Add(elem, rank); }
|
||||
void AddElement(int elem, int rank, int attribute)
|
||||
{ Add(elem, {rank, attribute}); }
|
||||
typedef std::map<int, NeighborElementRankMessage> Map;
|
||||
};
|
||||
|
||||
/** Used by Rebalance() to send elements and their ranks. Note that
|
||||
/** Used by Rebalance() to send elements, ranks, and attributes. Note that
|
||||
* RefTypes == true which means the refinement hierarchy will be recreated
|
||||
* on the receiving side.
|
||||
*/
|
||||
class RebalanceMessage : public ElementValueMessage<int, true,
|
||||
class RebalanceMessage :
|
||||
public ElementValueMessage<ElementRankAndAttribute, true,
|
||||
VarMessageTag::REBALANCE_VM>
|
||||
{
|
||||
public:
|
||||
void AddElementRank(int elem, int rank) { Add(elem, rank); }
|
||||
void AddElement(int elem, int rank, int attribute)
|
||||
{ Add(elem, {rank, attribute}); }
|
||||
typedef std::map<int, RebalanceMessage> Map;
|
||||
};
|
||||
|
||||
@@ -602,7 +628,8 @@ protected: // implementation
|
||||
/** For the face with ordered vertices vn* and neighboring element @a elem,
|
||||
check whether the other neighboring element (if it exists) is marked for
|
||||
a horizontal refinement conflicting with a vertical split. */
|
||||
void CheckRefAnisoFace(int elem, int vn1, int vn2, int vn3, int vn4,
|
||||
void CheckRefAnisoFace(const Refinement &ref, int elem,
|
||||
int vn1, int vn2, int vn3, int vn4,
|
||||
const Array<Refinement> &refinements,
|
||||
const std::map<int, int> &elemToRef,
|
||||
std::set<int> &conflicts);
|
||||
@@ -611,7 +638,8 @@ protected: // implementation
|
||||
neighboring element @a elem, check whether the other neighboring element
|
||||
(if it exists) is marked for a refinement conflicting with an isotropic
|
||||
refinement of the face. */
|
||||
void CheckRefIsoFace(int elem, int vn1, int vn2, int vn3, int vn4,
|
||||
void CheckRefIsoFace(const Refinement &ref, int elem,
|
||||
int vn1, int vn2, int vn3, int vn4,
|
||||
int en1, int en2, int en3, int en4,
|
||||
const Array<Refinement> &refinements,
|
||||
const std::map<int, int> &elemToRef,
|
||||
@@ -622,9 +650,8 @@ protected: // implementation
|
||||
const std::map<int, int> &elemToRef,
|
||||
std::set<int> &conflicts);
|
||||
|
||||
/** Check whether the refinement of the element with index @a elem and type
|
||||
@a ref_type would cause a conflict. */
|
||||
void CheckRefinement(int elem, char ref_type,
|
||||
/// Check whether the input refinement would cause a conflict.
|
||||
void CheckRefinement(int elem, const Refinement &ref,
|
||||
const Array<Refinement> &refinements,
|
||||
const std::map<int, int> &elemToRef,
|
||||
std::set<int> &conflicts);
|
||||
|
||||
@@ -52,6 +52,8 @@ endif
|
||||
.SUFFIXES:
|
||||
.SUFFIXES: .o .cpp .mk
|
||||
.PHONY: all lib-common clean clean-build clean-exec
|
||||
# Keeping the *.o files fixes an issue with the MacOS version of 'make'.
|
||||
.PRECIOUS: %.o
|
||||
|
||||
# Remove built-in rules
|
||||
%: %.cpp
|
||||
|
||||
@@ -151,6 +151,10 @@ if (MFEM_USE_MPI)
|
||||
MAIN phpref.cpp
|
||||
LIBRARIES mfem)
|
||||
|
||||
add_mfem_miniapp(pref321
|
||||
MAIN pref321.cpp
|
||||
LIBRARIES mfem)
|
||||
|
||||
# Add parallel tests.
|
||||
if (MFEM_ENABLE_TESTING)
|
||||
set(PARALLEL_TESTS
|
||||
@@ -160,6 +164,7 @@ if (MFEM_USE_MPI)
|
||||
fit-node-position
|
||||
pminimal-surface
|
||||
phpref
|
||||
pref321
|
||||
)
|
||||
# Meshing miniapps that return MFEM_SKIP_RETURN_VALUE in some cases:
|
||||
set(SKIP_TESTS)
|
||||
|
||||
@@ -24,7 +24,7 @@ SEQ_MINIAPPS = mobius-strip klein-bottle toroid trimmer twist mesh-explorer\
|
||||
shaper extruder mesh-optimizer minimal-surface polar-nc reflector\
|
||||
ref321 mesh-quality hpref
|
||||
PAR_MINIAPPS = pmesh-optimizer pminimal-surface pmesh-fitting fit-node-position\
|
||||
phpref mesh-bounding-boxes
|
||||
phpref pref321 mesh-bounding-boxes
|
||||
ifeq ($(MFEM_USE_MPI),NO)
|
||||
MINIAPPS = $(SEQ_MINIAPPS)
|
||||
else
|
||||
@@ -99,6 +99,8 @@ hpref-test-seq: hpref
|
||||
@$(call mfem-test,$<,, Serial hp-refinement)
|
||||
phpref-test-par: phpref
|
||||
@$(call mfem-test,$<, $(RUN_MPI), Parallel hp-refinement)
|
||||
pref321-test-par: pref321
|
||||
@$(call mfem-test,$<, $(RUN_MPI), Parallel 3:1 refinement)
|
||||
mesh-bounding-boxes-test-par: mesh-bounding-boxes
|
||||
@$(call mfem-test,$<, $(RUN_MPI), Parallel bounding boxes)
|
||||
ref321-test-seq: ref321
|
||||
|
||||
@@ -0,0 +1,336 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
//
|
||||
// -----------------------------------------------------------------
|
||||
// 3:1 Refinement Miniapp: Parallel 3:1 anisotropic mesh refinements
|
||||
// -----------------------------------------------------------------
|
||||
//
|
||||
// This miniapp performs random 3:1 refinements of a quadrilateral or hexahedral
|
||||
// mesh. A diffusion equation is solved in an H1 finite element space defined on
|
||||
// the refined mesh, and its continuity is verified across local and shared
|
||||
// faces.
|
||||
//
|
||||
// Compile with: make pref321
|
||||
//
|
||||
// Sample runs: mpirun -np 4 pref321 -mm -dim 2 -o 2 -r 100
|
||||
// mpirun -np 4 pref321 -mm -dim 3 -o 2 -r 100
|
||||
// mpirun -np 4 pref321 -m ../../data/star.mesh -o 2 -r 100
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
real_t CheckH1Continuity(ParGridFunction &x);
|
||||
|
||||
// Find the two children of parent element `elem` after its refinement in one
|
||||
// direction.
|
||||
void FindChildren(const Mesh &mesh, int elem, Array<int> &children)
|
||||
{
|
||||
const CoarseFineTransformations &cf = mesh.ncmesh->GetRefinementTransforms();
|
||||
MFEM_ASSERT(mesh.GetNE() == cf.embeddings.Size(), "");
|
||||
|
||||
// Note that row `elem` of the table constructed by cf.MakeCoarseToFineTable
|
||||
// is an alternative to this global loop, but constructing the table is also
|
||||
// a global operation with global storage.
|
||||
for (int i = 0; i < mesh.GetNE(); i++)
|
||||
{
|
||||
const int p = cf.embeddings[i].parent;
|
||||
if (p == elem)
|
||||
{
|
||||
children.Append(i);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Refine 3:1 via 2 refinements with scalings 2/3 and 1/2.
|
||||
void Refine31(Mesh &mesh, int elem, char type)
|
||||
{
|
||||
Array<Refinement> refs; // Refinement is defined in ncmesh.hpp
|
||||
refs.Append(Refinement(elem, type, 2.0 / 3.0));
|
||||
mesh.GeneralRefinement(refs);
|
||||
|
||||
// Find the elements with parent `elem`
|
||||
Array<int> children;
|
||||
FindChildren(mesh, elem, children);
|
||||
MFEM_ASSERT(children.Size() == 2, "");
|
||||
|
||||
const int elem1 = children[0];
|
||||
|
||||
refs.SetSize(0);
|
||||
refs.Append(Refinement(elem1, type)); // Default scaling of 0.5
|
||||
mesh.GeneralRefinement(refs);
|
||||
}
|
||||
|
||||
// Randomly select elements for 3:1 refinements in random directions.
|
||||
void TestAnisoRefRandom(int num_refs, int dim, ParMesh &mesh, int myid,
|
||||
int seed = 0)
|
||||
{
|
||||
std::mt19937 gen(seed);
|
||||
for (int i = 0; i < num_refs; i++)
|
||||
{
|
||||
const int elem = gen() % mesh.GetNE();
|
||||
const int t = gen() % dim;
|
||||
auto type = t == 0 ? Refinement::X :
|
||||
(t == 1 ? Refinement::Y : Refinement::Z);
|
||||
|
||||
// In 3D, check for conflicts in the parallel refinements.
|
||||
if (dim == 3)
|
||||
{
|
||||
std::set<int> conflicts; // Indices in refs of conflicting elements
|
||||
Array<Refinement> refs;
|
||||
refs.Append(Refinement(elem, type));
|
||||
const bool conflict = mesh.AnisotropicConflict(refs, conflicts);
|
||||
if (conflict)
|
||||
{
|
||||
if (myid == 0)
|
||||
cout << "Anisotropic conflict on iteration " << i
|
||||
<< ", retrying\n";
|
||||
i--;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
Refine31(mesh, elem, type);
|
||||
}
|
||||
|
||||
mesh.EnsureNodes();
|
||||
mesh.SetScaledNCMesh();
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
Mpi::Init(argc, argv);
|
||||
Hypre::Init();
|
||||
|
||||
const int num_procs = Mpi::WorldSize();
|
||||
const int myid = Mpi::WorldRank();
|
||||
|
||||
// 1. Parse command-line options.
|
||||
const char *mesh_file = "../../data/star.mesh";
|
||||
int order = 1;
|
||||
bool visualization = true;
|
||||
bool makeMesh = false;
|
||||
int num_refs = 1;
|
||||
int tdim = 2; // Mesh dimension for Cartesian meshes.
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
"Mesh file to use.");
|
||||
args.AddOption(&order, "-o", "--order",
|
||||
"Finite element order (polynomial degree).");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.AddOption(&makeMesh, "-mm", "--make-mesh", "-no-mm",
|
||||
"--no-make-mesh", "Create Cartesian mesh");
|
||||
args.AddOption(&tdim, "-dim", "--dimension", "Dimension for Cartesian mesh");
|
||||
args.AddOption(&num_refs, "-r", "--refs", "Number of 3:1 refinements");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
// 2. Create or read the serial mesh on all ranks, then apply the same
|
||||
// deterministic 3:1 refinement sequence before partitioning it.
|
||||
Mesh mesh;
|
||||
if (makeMesh)
|
||||
{
|
||||
mesh = tdim == 3 ? Mesh::MakeCartesian3D(2, 2, 2, Element::HEXAHEDRON) :
|
||||
Mesh::MakeCartesian2D(2, 2, Element::QUADRILATERAL);
|
||||
}
|
||||
else
|
||||
{
|
||||
mesh = Mesh::LoadFromFile(mesh_file, 1, 1);
|
||||
}
|
||||
|
||||
const int dim = mesh.Dimension();
|
||||
|
||||
mesh.EnsureNCMesh();
|
||||
mesh.SetScaledNCMesh();
|
||||
|
||||
// 3. Partition the refined serial mesh.
|
||||
ParMesh pmesh(MPI_COMM_WORLD, mesh);
|
||||
mesh.Clear();
|
||||
|
||||
TestAnisoRefRandom(num_refs, dim, pmesh, myid, myid);
|
||||
|
||||
// 4. Define a parallel H1 finite element space and report its global size.
|
||||
H1_FECollection fec(order, dim);
|
||||
ParFiniteElementSpace fespace(&pmesh, &fec);
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Number of finite element unknowns: "
|
||||
<< fespace.GlobalTrueVSize() << endl;
|
||||
}
|
||||
|
||||
// 5. Assemble and solve the Poisson problem, following ex1p.
|
||||
ParGridFunction x(&fespace);
|
||||
x = 0.0;
|
||||
|
||||
ParLinearForm b(&fespace);
|
||||
ConstantCoefficient one(1.0);
|
||||
b.AddDomainIntegrator(new DomainLFIntegrator(one));
|
||||
b.Assemble();
|
||||
|
||||
ParBilinearForm a(&fespace);
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator());
|
||||
a.Assemble();
|
||||
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
Array<int> ess_tdof_list;
|
||||
if (pmesh.bdr_attributes.Size())
|
||||
{
|
||||
Array<int> ess_bdr(pmesh.bdr_attributes.Max());
|
||||
ess_bdr = 0;
|
||||
pmesh.MarkExternalBoundaries(ess_bdr);
|
||||
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
}
|
||||
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
|
||||
HypreBoomerAMG M;
|
||||
CGSolver cg(MPI_COMM_WORLD);
|
||||
cg.SetPreconditioner(M);
|
||||
cg.SetOperator(*A);
|
||||
cg.SetRelTol(1e-12);
|
||||
cg.SetMaxIter(2000);
|
||||
cg.SetPrintLevel(1);
|
||||
cg.Mult(B, X);
|
||||
|
||||
a.RecoverFEMSolution(X, b, x);
|
||||
|
||||
// 6. Verify the continuity of the solution in H1 over local and shared
|
||||
// faces and compute the global maximum jump.
|
||||
const real_t h1err = CheckH1Continuity(x);
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Error of H1 continuity: " << h1err << endl;
|
||||
}
|
||||
MFEM_VERIFY(h1err < 1.0e-7, "H1 discontinuity found");
|
||||
|
||||
// 7. Save the refined mesh and the solution in parallel. This output can
|
||||
// be viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
|
||||
{
|
||||
ostringstream mesh_name, sol_name;
|
||||
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
|
||||
sol_name << "sol." << setfill('0') << setw(6) << myid;
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
pmesh.Print(mesh_ofs);
|
||||
|
||||
ofstream sol_ofs(sol_name.str().c_str());
|
||||
sol_ofs.precision(8);
|
||||
x.Save(sol_ofs);
|
||||
}
|
||||
|
||||
// 8. Send the parallel solution to GLVis.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << pmesh << x << flush;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
real_t CheckH1Continuity(ParGridFunction &x)
|
||||
{
|
||||
const ParFiniteElementSpace *pfes = x.ParFESpace();
|
||||
ParMesh *pmesh = pfes->GetParMesh();
|
||||
const int dim = pmesh->Dimension();
|
||||
|
||||
real_t errorMax = 0.0;
|
||||
|
||||
// Shared-face values require face-neighbor data.
|
||||
x.ExchangeFaceNbrData();
|
||||
|
||||
// First handle faces for which both elements are local to this rank.
|
||||
for (int f = 0; f < pmesh->GetNumFaces(); f++)
|
||||
{
|
||||
const auto info = pmesh->GetFaceInformation(f);
|
||||
if (!info.IsLocal())
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
FaceElementTransformations *FT = pmesh->GetFaceElementTransformations(f);
|
||||
const int faceOrder = dim == 3 ? pfes->GetFaceOrder(f) :
|
||||
pfes->GetEdgeOrder(f);
|
||||
const IntegrationRule &ir = IntRules.Get(FT->FaceGeom, 2 * faceOrder);
|
||||
|
||||
for (int i = 0; i < ir.GetNPoints(); i++)
|
||||
{
|
||||
const IntegrationPoint &fip = ir.IntPoint(i);
|
||||
IntegrationPoint ip1, ip2;
|
||||
|
||||
FT->Loc1.Transform(fip, ip1);
|
||||
FT->Loc2.Transform(fip, ip2);
|
||||
|
||||
const real_t v1 = x.GetValue(*FT->Elem1, ip1);
|
||||
const real_t v2 = x.GetValue(*FT->Elem2, ip2);
|
||||
errorMax = std::max(errorMax, std::abs(v1 - v2));
|
||||
}
|
||||
}
|
||||
|
||||
// Then check partition interfaces. Conforming shared faces are handled on
|
||||
// the lower-rank side, while shared slave nonconforming faces are handled
|
||||
// only on the slave side and therefore do not need additional filtering.
|
||||
for (int sf = 0; sf < pmesh->GetNSharedFaces(); sf++)
|
||||
{
|
||||
const int f = pmesh->GetSharedFace(sf);
|
||||
const auto info = pmesh->GetFaceInformation(f);
|
||||
if (!info.IsShared())
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
FaceElementTransformations *FT = pmesh->GetSharedFaceTransformations(sf);
|
||||
const int faceOrder = dim == 3 ? pfes->GetFaceOrder(f) :
|
||||
pfes->GetEdgeOrder(f);
|
||||
const IntegrationRule &ir = IntRules.Get(FT->FaceGeom, 2 * faceOrder);
|
||||
|
||||
for (int i = 0; i < ir.GetNPoints(); i++)
|
||||
{
|
||||
const IntegrationPoint &fip = ir.IntPoint(i);
|
||||
IntegrationPoint ip1, ip2;
|
||||
|
||||
FT->Loc1.Transform(fip, ip1);
|
||||
FT->Loc2.Transform(fip, ip2);
|
||||
|
||||
const real_t v1 = x.GetValue(*FT->Elem1, ip1);
|
||||
const real_t v2 = x.GetValue(*FT->Elem2, ip2);
|
||||
errorMax = std::max(errorMax, std::abs(v1 - v2));
|
||||
}
|
||||
}
|
||||
|
||||
MPI_Allreduce(MPI_IN_PLACE, &errorMax, 1, MFEM_MPI_REAL_T, MPI_MAX,
|
||||
pmesh->GetComm());
|
||||
|
||||
return errorMax;
|
||||
}
|
||||
@@ -71,22 +71,14 @@ void Refine31(Mesh & mesh, int elem, char type)
|
||||
mesh.GeneralRefinement(refs);
|
||||
}
|
||||
|
||||
// Deterministic, somewhat random integer generator
|
||||
int MyRand(int & s)
|
||||
{
|
||||
s++;
|
||||
const double a = 1000 * sin(s * 1.1234 * M_PI);
|
||||
return int(std::abs(a));
|
||||
}
|
||||
|
||||
// Randomly select elements for 3:1 refinements in random directions.
|
||||
void TestAnisoRefRandom(int iter, int dim, Mesh & mesh)
|
||||
void TestAnisoRefRandom(int num_refs, int dim, Mesh & mesh)
|
||||
{
|
||||
int seed = 0;
|
||||
for (int i = 0; i < iter; i++)
|
||||
std::mt19937 gen(1);
|
||||
for (int i = 0; i < num_refs; i++)
|
||||
{
|
||||
const int elem = MyRand(seed) % mesh.GetNE();
|
||||
const int t = MyRand(seed) % dim;
|
||||
const auto elem = gen() % mesh.GetNE();
|
||||
const auto t = gen() % dim;
|
||||
auto type = t == 0 ? Refinement::X :
|
||||
(t == 1 ? Refinement::Y : Refinement::Z);
|
||||
Refine31(mesh, elem, type);
|
||||
|
||||
@@ -68,7 +68,7 @@ multidomain-test-par: multidomain
|
||||
multidomain_nd-test-par: multidomain_nd
|
||||
@$(call mfem-test,$<, $(RUN_MPI), Multidomain ND miniapp,-tf 0.001)
|
||||
multidomain_rt-test-par: multidomain_rt
|
||||
@$(call mfem-test,$<, $(RUN_MPI), Multidomain RT iniapp,-tf 0.001)
|
||||
@$(call mfem-test,$<, $(RUN_MPI), Multidomain RT miniapp,-tf 0.001)
|
||||
|
||||
# Generate an error message if the MFEM library is not built and exit
|
||||
$(MFEM_LIB_FILE):
|
||||
|
||||
@@ -32,7 +32,11 @@
|
||||
// Custom benchmark arguments generator
|
||||
static void CustomArguments(bm::Benchmark *b) noexcept
|
||||
{
|
||||
constexpr int MAX_NDOFS = 16 * 1024 * (mfem_use_gpu ? 1024 : 8);
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP_LANG)
|
||||
constexpr int MAX_NDOFS = 16 * 1024 * 1024;
|
||||
#else
|
||||
constexpr int MAX_NDOFS = 16 * 1024 * 8;
|
||||
#endif
|
||||
|
||||
const auto orders = { 7, 6, 5, 4, 3, 2, 1 };
|
||||
|
||||
|
||||
@@ -39,6 +39,7 @@ set(UNIT_TESTS_SRCS
|
||||
dfem/test_divergence.cpp
|
||||
dfem/test_lvector_interface.cpp
|
||||
dfem/test_mass.cpp
|
||||
dfem/test_tuple.cpp
|
||||
general/test_array.cpp
|
||||
general/test_scan.cpp
|
||||
general/test_arrays_by_name.cpp
|
||||
@@ -140,6 +141,7 @@ set(UNIT_TESTS_SRCS
|
||||
fem/test_lor_batched.cpp
|
||||
fem/test_lor_dg.cpp
|
||||
fem/test_lor.cpp
|
||||
fem/test_mixedsesqform.cpp
|
||||
fem/test_nonlinearform.cpp
|
||||
fem/test_operatorjacobismoother.cpp
|
||||
fem/test_oscillation.cpp
|
||||
@@ -168,6 +170,20 @@ set(UNIT_TESTS_SRCS
|
||||
fem/test_transfer.cpp
|
||||
fem/test_var_order.cpp
|
||||
fem/test_white_noise.cpp
|
||||
fem/specializations/test_diffusion_integ.cpp
|
||||
fem/specializations/test_mass_integ.cpp
|
||||
fem/specializations/test_convection_integ.cpp
|
||||
fem/specializations/test_vecmass_integ.cpp
|
||||
fem/specializations/test_curlcurl_integ.cpp
|
||||
fem/specializations/test_vecdiffusion_integ.cpp
|
||||
fem/specializations/test_dgtrace_integ.cpp
|
||||
fem/specializations/test_dgdiffusion_integ.cpp
|
||||
fem/specializations/test_dgmassinv.cpp
|
||||
fem/specializations/test_qinterp_det.cpp
|
||||
fem/specializations/test_qinterp_eval.cpp
|
||||
fem/specializations/test_qinterp_grad.cpp
|
||||
fem/specializations/test_qinterp_tensoreval.cpp
|
||||
fem/specializations/test_qinterp_eval_hdiv.cpp
|
||||
enzyme/compatibility.cpp
|
||||
# The following are tested separately (keep the comment as a reminder).
|
||||
# This list can be updated using (in bash):
|
||||
|
||||
@@ -0,0 +1,379 @@
|
||||
MFEM NC mesh v1.0
|
||||
|
||||
# NCMesh supported geometry types:
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
# PYRAMID = 7
|
||||
|
||||
dimension
|
||||
3
|
||||
|
||||
rank
|
||||
0
|
||||
|
||||
# rank attr geom ref_type nodes/children
|
||||
elements
|
||||
75
|
||||
0 1 5 0 0 1 5 4 16 17 21 20
|
||||
0 1 5 0 16 17 21 20 32 33 37 36
|
||||
-1 1 5 7 59 60 61 62 63 64 65 66
|
||||
0 1 5 0 1 2 6 5 17 18 22 21
|
||||
-1 1 5 7 27 28 29 30 31 32 33 34
|
||||
0 1 5 0 21 22 26 25 37 38 42 41
|
||||
-1 1 5 7 43 44 45 46 47 48 49 50
|
||||
0 1 5 0 4 5 9 8 20 21 25 24
|
||||
0 1 5 0 8 9 13 12 24 25 29 28
|
||||
0 1 5 0 24 25 29 28 40 41 45 44
|
||||
0 1 5 0 9 10 14 13 25 26 30 29
|
||||
-1 1 5 7 67 68 69 70 71 72 73 74
|
||||
0 1 5 0 41 42 46 45 57 58 62 61
|
||||
0 1 5 0 40 41 45 44 56 57 61 60
|
||||
0 1 5 0 36 37 41 40 52 53 57 56
|
||||
-1 1 5 7 35 36 37 38 39 40 41 42
|
||||
0 1 5 0 32 33 37 36 48 49 53 52
|
||||
0 1 5 0 33 34 38 37 49 50 54 53
|
||||
0 1 5 0 34 35 39 38 50 51 55 54
|
||||
0 1 5 0 38 39 43 42 54 55 59 58
|
||||
0 1 5 0 42 43 47 46 58 59 63 62
|
||||
0 1 5 0 26 27 31 30 42 43 47 46
|
||||
0 1 5 0 10 11 15 14 26 27 31 30
|
||||
0 1 5 0 6 7 11 10 22 23 27 26
|
||||
-1 1 5 7 51 52 53 54 55 56 57 58
|
||||
0 1 5 0 18 19 23 22 34 35 39 38
|
||||
0 1 5 0 2 3 7 6 18 19 23 22
|
||||
0 1 5 0 5 94 208 99 74 209 214 212
|
||||
0 1 5 0 94 6 97 208 209 96 210 214
|
||||
0 1 5 0 208 97 10 98 214 210 103 211
|
||||
0 1 5 0 99 208 98 9 212 214 211 104
|
||||
0 1 5 0 74 209 214 212 21 86 213 102
|
||||
0 1 5 0 209 96 210 214 86 22 100 213
|
||||
0 1 5 0 214 210 103 211 213 100 26 101
|
||||
0 1 5 0 212 214 211 104 102 213 101 25
|
||||
0 1 5 0 37 89 269 107 156 270 275 273
|
||||
0 1 5 0 89 38 105 269 270 159 271 275
|
||||
0 1 5 0 269 105 42 106 275 271 144 272
|
||||
0 1 5 0 107 269 106 41 273 275 272 143
|
||||
0 1 5 0 156 270 275 273 53 157 274 153
|
||||
0 1 5 0 270 159 271 275 157 54 158 274
|
||||
0 1 5 0 275 271 144 272 274 158 58 139
|
||||
0 1 5 0 273 275 272 143 153 274 139 57
|
||||
0 1 5 0 20 70 330 111 83 331 336 334
|
||||
0 1 5 0 70 21 102 330 331 82 332 336
|
||||
0 1 5 0 330 102 25 110 336 332 109 333
|
||||
0 1 5 0 111 330 110 24 334 336 333 114
|
||||
0 1 5 0 83 331 336 334 36 78 335 113
|
||||
0 1 5 0 331 82 332 336 78 37 107 335
|
||||
0 1 5 0 336 332 109 333 335 107 41 112
|
||||
0 1 5 0 334 336 333 114 113 335 112 40
|
||||
0 1 5 0 22 198 387 100 91 388 393 391
|
||||
0 1 5 0 198 23 199 387 388 201 389 393
|
||||
0 1 5 0 387 199 27 186 393 389 189 390
|
||||
0 1 5 0 100 387 186 26 391 393 390 108
|
||||
0 1 5 0 91 388 393 391 38 170 392 105
|
||||
0 1 5 0 388 201 389 393 170 39 176 392
|
||||
0 1 5 0 393 389 189 390 392 176 43 177
|
||||
0 1 5 0 391 393 390 108 105 392 177 42
|
||||
0 1 5 0 17 84 444 69 81 445 450 448
|
||||
0 1 5 0 84 18 85 444 445 90 446 450
|
||||
0 1 5 0 444 85 22 86 450 446 91 447
|
||||
0 1 5 0 69 444 86 21 448 450 447 82
|
||||
0 1 5 0 81 445 450 448 33 87 449 77
|
||||
0 1 5 0 445 90 446 450 87 34 88 449
|
||||
0 1 5 0 450 446 91 447 449 88 38 89
|
||||
0 1 5 0 448 450 447 82 77 449 89 37
|
||||
0 1 5 0 25 101 497 121 109 498 503 501
|
||||
0 1 5 0 101 26 133 497 498 108 499 503
|
||||
0 1 5 0 497 133 30 134 503 499 138 500
|
||||
0 1 5 0 121 497 134 29 501 503 500 129
|
||||
0 1 5 0 109 498 503 501 41 106 502 126
|
||||
0 1 5 0 498 108 499 503 106 42 136 502
|
||||
0 1 5 0 503 499 138 500 502 136 46 137
|
||||
0 1 5 0 501 503 500 129 126 502 137 45
|
||||
|
||||
# attr geom nodes
|
||||
boundary
|
||||
72
|
||||
1 3 4 5 1 0
|
||||
1 3 0 1 17 16
|
||||
1 3 4 0 16 20
|
||||
1 3 16 17 33 32
|
||||
1 3 20 16 32 36
|
||||
1 3 5 6 2 1
|
||||
1 3 1 2 18 17
|
||||
1 3 8 9 5 4
|
||||
1 3 8 4 20 24
|
||||
1 3 12 13 9 8
|
||||
1 3 13 12 28 29
|
||||
1 3 12 8 24 28
|
||||
1 3 29 28 44 45
|
||||
1 3 28 24 40 44
|
||||
1 3 13 14 10 9
|
||||
1 3 14 13 29 30
|
||||
1 3 46 45 61 62
|
||||
1 3 57 58 62 61
|
||||
1 3 45 44 60 61
|
||||
1 3 44 40 56 60
|
||||
1 3 56 57 61 60
|
||||
1 3 40 36 52 56
|
||||
1 3 52 53 57 56
|
||||
1 3 32 33 49 48
|
||||
1 3 36 32 48 52
|
||||
1 3 48 49 53 52
|
||||
1 3 33 34 50 49
|
||||
1 3 49 50 54 53
|
||||
1 3 34 35 51 50
|
||||
1 3 35 39 55 51
|
||||
1 3 50 51 55 54
|
||||
1 3 39 43 59 55
|
||||
1 3 54 55 59 58
|
||||
1 3 43 47 63 59
|
||||
1 3 47 46 62 63
|
||||
1 3 58 59 63 62
|
||||
1 3 27 31 47 43
|
||||
1 3 31 30 46 47
|
||||
1 3 14 15 11 10
|
||||
1 3 11 15 31 27
|
||||
1 3 15 14 30 31
|
||||
1 3 10 11 7 6
|
||||
1 3 7 11 27 23
|
||||
1 3 18 19 35 34
|
||||
1 3 19 23 39 35
|
||||
1 3 6 7 3 2
|
||||
1 3 2 3 19 18
|
||||
1 3 3 7 23 19
|
||||
2 3 99 208 94 5
|
||||
2 3 208 97 6 94
|
||||
2 3 98 10 97 208
|
||||
2 3 9 98 208 99
|
||||
2 3 53 157 274 153
|
||||
2 3 157 54 158 274
|
||||
2 3 274 158 58 139
|
||||
2 3 153 274 139 57
|
||||
2 3 111 20 83 334
|
||||
2 3 24 111 334 114
|
||||
2 3 334 83 36 113
|
||||
2 3 114 334 113 40
|
||||
2 3 23 199 389 201
|
||||
2 3 199 27 189 389
|
||||
2 3 201 389 176 39
|
||||
2 3 389 189 43 176
|
||||
2 3 17 84 445 81
|
||||
2 3 84 18 90 445
|
||||
2 3 81 445 87 33
|
||||
2 3 445 90 34 87
|
||||
2 3 30 134 500 138
|
||||
2 3 134 29 129 500
|
||||
2 3 138 500 137 46
|
||||
2 3 500 129 45 137
|
||||
|
||||
# vert_id p1 p2
|
||||
vertex_parents
|
||||
102
|
||||
69 17 21
|
||||
70 20 21
|
||||
74 5 21
|
||||
77 33 37
|
||||
78 36 37
|
||||
81 17 33
|
||||
82 21 37
|
||||
83 20 36
|
||||
84 17 18
|
||||
85 18 22
|
||||
86 21 22
|
||||
87 33 34
|
||||
88 34 38
|
||||
89 37 38
|
||||
90 18 34
|
||||
91 22 38
|
||||
94 5 6
|
||||
96 6 22
|
||||
97 6 10
|
||||
98 9 10
|
||||
99 5 9
|
||||
100 22 26
|
||||
101 25 26
|
||||
102 21 25
|
||||
103 10 26
|
||||
104 9 25
|
||||
105 38 42
|
||||
106 41 42
|
||||
107 37 41
|
||||
108 26 42
|
||||
109 25 41
|
||||
110 24 25
|
||||
111 20 24
|
||||
112 40 41
|
||||
113 36 40
|
||||
114 24 40
|
||||
121 25 29
|
||||
126 41 45
|
||||
129 29 45
|
||||
133 26 30
|
||||
134 29 30
|
||||
136 42 46
|
||||
137 45 46
|
||||
138 30 46
|
||||
139 57 58
|
||||
143 41 57
|
||||
144 42 58
|
||||
153 53 57
|
||||
156 37 53
|
||||
157 53 54
|
||||
158 54 58
|
||||
159 38 54
|
||||
170 38 39
|
||||
176 39 43
|
||||
177 42 43
|
||||
186 26 27
|
||||
189 27 43
|
||||
198 22 23
|
||||
199 23 27
|
||||
201 23 39
|
||||
208 97 99
|
||||
209 74 96
|
||||
210 96 103
|
||||
211 103 104
|
||||
212 74 104
|
||||
213 100 102
|
||||
214 209 211
|
||||
269 105 107
|
||||
270 156 159
|
||||
271 144 159
|
||||
272 143 144
|
||||
273 143 156
|
||||
274 153 158
|
||||
275 270 272
|
||||
330 102 111
|
||||
331 82 83
|
||||
332 82 109
|
||||
333 109 114
|
||||
334 83 114
|
||||
335 107 113
|
||||
336 331 333
|
||||
387 100 199
|
||||
388 91 201
|
||||
389 189 201
|
||||
390 108 189
|
||||
391 91 108
|
||||
392 105 176
|
||||
393 388 390
|
||||
444 69 85
|
||||
445 81 90
|
||||
446 90 91
|
||||
447 82 91
|
||||
448 81 82
|
||||
449 77 88
|
||||
450 445 447
|
||||
497 121 133
|
||||
498 108 109
|
||||
499 108 138
|
||||
500 129 138
|
||||
501 109 129
|
||||
502 126 136
|
||||
503 498 500
|
||||
|
||||
# root element orientation
|
||||
root_state
|
||||
27
|
||||
0
|
||||
1
|
||||
1
|
||||
15
|
||||
15
|
||||
6
|
||||
6
|
||||
22
|
||||
15
|
||||
8
|
||||
12
|
||||
10
|
||||
10
|
||||
18
|
||||
18
|
||||
13
|
||||
7
|
||||
22
|
||||
22
|
||||
15
|
||||
16
|
||||
16
|
||||
16
|
||||
7
|
||||
8
|
||||
6
|
||||
21
|
||||
|
||||
# top-level node coordinates
|
||||
coordinates
|
||||
64
|
||||
3
|
||||
0 0 0
|
||||
0.33333333 0 0
|
||||
0.66666667 0 0
|
||||
1 0 0
|
||||
0 0.33333333 0
|
||||
0.33333333 0.33333333 0
|
||||
0.66666667 0.33333333 0
|
||||
1 0.33333333 0
|
||||
0 0.66666667 0
|
||||
0.33333333 0.66666667 0
|
||||
0.66666667 0.66666667 0
|
||||
1 0.66666667 0
|
||||
0 1 0
|
||||
0.33333333 1 0
|
||||
0.66666667 1 0
|
||||
1 1 0
|
||||
0 0 0.33333333
|
||||
0.33333333 0 0.33333333
|
||||
0.66666667 0 0.33333333
|
||||
1 0 0.33333333
|
||||
0 0.33333333 0.33333333
|
||||
0.33333333 0.33333333 0.33333333
|
||||
0.66666667 0.33333333 0.33333333
|
||||
1 0.33333333 0.33333333
|
||||
0 0.66666667 0.33333333
|
||||
0.33333333 0.66666667 0.33333333
|
||||
0.66666667 0.66666667 0.33333333
|
||||
1 0.66666667 0.33333333
|
||||
0 1 0.33333333
|
||||
0.33333333 1 0.33333333
|
||||
0.66666667 1 0.33333333
|
||||
1 1 0.33333333
|
||||
0 0 0.66666667
|
||||
0.33333333 0 0.66666667
|
||||
0.66666667 0 0.66666667
|
||||
1 0 0.66666667
|
||||
0 0.33333333 0.66666667
|
||||
0.33333333 0.33333333 0.66666667
|
||||
0.66666667 0.33333333 0.66666667
|
||||
1 0.33333333 0.66666667
|
||||
0 0.66666667 0.66666667
|
||||
0.33333333 0.66666667 0.66666667
|
||||
0.66666667 0.66666667 0.66666667
|
||||
1 0.66666667 0.66666667
|
||||
0 1 0.66666667
|
||||
0.33333333 1 0.66666667
|
||||
0.66666667 1 0.66666667
|
||||
1 1 0.66666667
|
||||
0 0 1
|
||||
0.33333333 0 1
|
||||
0.66666667 0 1
|
||||
1 0 1
|
||||
0 0.33333333 1
|
||||
0.33333333 0.33333333 1
|
||||
0.66666667 0.33333333 1
|
||||
1 0.33333333 1
|
||||
0 0.66666667 1
|
||||
0.33333333 0.66666667 1
|
||||
0.66666667 0.66666667 1
|
||||
1 0.66666667 1
|
||||
0 1 1
|
||||
0.33333333 1 1
|
||||
0.66666667 1 1
|
||||
1 1 1
|
||||
|
||||
mfem_mesh_end
|
||||
@@ -0,0 +1,555 @@
|
||||
MFEM NC mesh v1.0
|
||||
|
||||
# NCMesh supported geometry types:
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
# PYRAMID = 7
|
||||
|
||||
dimension
|
||||
3
|
||||
|
||||
rank
|
||||
0
|
||||
|
||||
# rank attr geom ref_type nodes/children
|
||||
elements
|
||||
258
|
||||
0 1 4 0 21 0 5 1
|
||||
0 1 4 0 21 0 1 17
|
||||
0 1 4 0 21 0 17 16
|
||||
0 1 4 0 21 0 4 5
|
||||
0 1 4 0 21 0 20 4
|
||||
0 1 4 0 21 0 16 20
|
||||
0 1 4 0 22 1 6 2
|
||||
0 1 4 0 22 1 2 18
|
||||
0 1 4 0 22 1 18 17
|
||||
0 1 4 0 22 1 5 6
|
||||
0 1 4 0 22 1 21 5
|
||||
0 1 4 0 22 1 17 21
|
||||
0 1 4 0 23 2 7 3
|
||||
0 1 4 0 23 2 3 19
|
||||
0 1 4 0 23 2 19 18
|
||||
0 1 4 0 23 2 6 7
|
||||
0 1 4 0 23 2 22 6
|
||||
0 1 4 0 23 2 18 22
|
||||
0 1 4 0 25 4 9 5
|
||||
0 1 4 0 25 4 5 21
|
||||
0 1 4 0 25 4 21 20
|
||||
0 1 4 0 25 4 8 9
|
||||
0 1 4 0 25 4 24 8
|
||||
0 1 4 0 25 4 20 24
|
||||
-1 1 4 7 170 171 172 173 174 175 176 177
|
||||
0 1 4 0 26 5 6 22
|
||||
0 1 4 0 26 5 22 21
|
||||
-1 1 4 7 162 163 164 165 166 167 168 169
|
||||
0 1 4 0 26 5 25 9
|
||||
0 1 4 0 26 5 21 25
|
||||
0 1 4 0 27 6 11 7
|
||||
0 1 4 0 27 6 7 23
|
||||
0 1 4 0 27 6 23 22
|
||||
0 1 4 0 27 6 10 11
|
||||
0 1 4 0 27 6 26 10
|
||||
0 1 4 0 27 6 22 26
|
||||
0 1 4 0 29 8 13 9
|
||||
0 1 4 0 29 8 9 25
|
||||
0 1 4 0 29 8 25 24
|
||||
0 1 4 0 29 8 12 13
|
||||
0 1 4 0 29 8 28 12
|
||||
0 1 4 0 29 8 24 28
|
||||
0 1 4 0 30 9 14 10
|
||||
0 1 4 0 30 9 10 26
|
||||
0 1 4 0 30 9 26 25
|
||||
0 1 4 0 30 9 13 14
|
||||
0 1 4 0 30 9 29 13
|
||||
0 1 4 0 30 9 25 29
|
||||
0 1 4 0 31 10 15 11
|
||||
0 1 4 0 31 10 11 27
|
||||
0 1 4 0 31 10 27 26
|
||||
0 1 4 0 31 10 14 15
|
||||
0 1 4 0 31 10 30 14
|
||||
0 1 4 0 31 10 26 30
|
||||
0 1 4 0 37 16 21 17
|
||||
0 1 4 0 37 16 17 33
|
||||
0 1 4 0 37 16 33 32
|
||||
0 1 4 0 37 16 20 21
|
||||
0 1 4 0 37 16 36 20
|
||||
0 1 4 0 37 16 32 36
|
||||
0 1 4 0 38 17 22 18
|
||||
-1 1 4 7 226 227 228 229 230 231 232 233
|
||||
-1 1 4 7 234 235 236 237 238 239 240 241
|
||||
0 1 4 0 38 17 21 22
|
||||
0 1 4 0 38 17 37 21
|
||||
0 1 4 0 38 17 33 37
|
||||
0 1 4 0 39 18 23 19
|
||||
0 1 4 0 39 18 19 35
|
||||
0 1 4 0 39 18 35 34
|
||||
0 1 4 0 39 18 22 23
|
||||
0 1 4 0 39 18 38 22
|
||||
0 1 4 0 39 18 34 38
|
||||
0 1 4 0 41 20 25 21
|
||||
0 1 4 0 41 20 21 37
|
||||
0 1 4 0 41 20 37 36
|
||||
0 1 4 0 41 20 24 25
|
||||
-1 1 4 7 202 203 204 205 206 207 208 209
|
||||
-1 1 4 7 194 195 196 197 198 199 200 201
|
||||
0 1 4 0 42 21 26 22
|
||||
0 1 4 0 42 21 22 38
|
||||
0 1 4 0 42 21 38 37
|
||||
0 1 4 0 42 21 25 26
|
||||
0 1 4 0 42 21 41 25
|
||||
0 1 4 0 42 21 37 41
|
||||
-1 1 4 7 210 211 212 213 214 215 216 217
|
||||
-1 1 4 7 218 219 220 221 222 223 224 225
|
||||
0 1 4 0 43 22 39 38
|
||||
0 1 4 0 43 22 26 27
|
||||
0 1 4 0 43 22 42 26
|
||||
0 1 4 0 43 22 38 42
|
||||
0 1 4 0 45 24 29 25
|
||||
0 1 4 0 45 24 25 41
|
||||
0 1 4 0 45 24 41 40
|
||||
0 1 4 0 45 24 28 29
|
||||
0 1 4 0 45 24 44 28
|
||||
0 1 4 0 45 24 40 44
|
||||
0 1 4 0 46 25 30 26
|
||||
0 1 4 0 46 25 26 42
|
||||
0 1 4 0 46 25 42 41
|
||||
-1 1 4 7 250 251 252 253 254 255 256 257
|
||||
-1 1 4 7 242 243 244 245 246 247 248 249
|
||||
0 1 4 0 46 25 41 45
|
||||
0 1 4 0 47 26 31 27
|
||||
0 1 4 0 47 26 27 43
|
||||
0 1 4 0 47 26 43 42
|
||||
0 1 4 0 47 26 30 31
|
||||
0 1 4 0 47 26 46 30
|
||||
0 1 4 0 47 26 42 46
|
||||
0 1 4 0 53 32 37 33
|
||||
0 1 4 0 53 32 33 49
|
||||
0 1 4 0 53 32 49 48
|
||||
0 1 4 0 53 32 36 37
|
||||
0 1 4 0 53 32 52 36
|
||||
0 1 4 0 53 32 48 52
|
||||
0 1 4 0 54 33 38 34
|
||||
0 1 4 0 54 33 34 50
|
||||
0 1 4 0 54 33 50 49
|
||||
0 1 4 0 54 33 37 38
|
||||
0 1 4 0 54 33 53 37
|
||||
0 1 4 0 54 33 49 53
|
||||
0 1 4 0 55 34 39 35
|
||||
0 1 4 0 55 34 35 51
|
||||
0 1 4 0 55 34 51 50
|
||||
0 1 4 0 55 34 38 39
|
||||
0 1 4 0 55 34 54 38
|
||||
0 1 4 0 55 34 50 54
|
||||
0 1 4 0 57 36 41 37
|
||||
0 1 4 0 57 36 37 53
|
||||
0 1 4 0 57 36 53 52
|
||||
0 1 4 0 57 36 40 41
|
||||
0 1 4 0 57 36 56 40
|
||||
0 1 4 0 57 36 52 56
|
||||
0 1 4 0 58 37 42 38
|
||||
0 1 4 0 58 37 38 54
|
||||
-1 1 4 7 178 179 180 181 182 183 184 185
|
||||
0 1 4 0 58 37 41 42
|
||||
0 1 4 0 58 37 57 41
|
||||
-1 1 4 7 186 187 188 189 190 191 192 193
|
||||
0 1 4 0 59 38 43 39
|
||||
0 1 4 0 59 38 39 55
|
||||
0 1 4 0 59 38 55 54
|
||||
0 1 4 0 59 38 42 43
|
||||
0 1 4 0 59 38 58 42
|
||||
0 1 4 0 59 38 54 58
|
||||
0 1 4 0 61 40 45 41
|
||||
0 1 4 0 61 40 41 57
|
||||
0 1 4 0 61 40 57 56
|
||||
0 1 4 0 61 40 44 45
|
||||
0 1 4 0 61 40 60 44
|
||||
0 1 4 0 61 40 56 60
|
||||
0 1 4 0 62 41 46 42
|
||||
0 1 4 0 62 41 42 58
|
||||
0 1 4 0 62 41 58 57
|
||||
0 1 4 0 62 41 45 46
|
||||
0 1 4 0 62 41 61 45
|
||||
0 1 4 0 62 41 57 61
|
||||
0 1 4 0 63 42 47 43
|
||||
0 1 4 0 63 42 43 59
|
||||
0 1 4 0 63 42 59 58
|
||||
0 1 4 0 63 42 46 47
|
||||
0 1 4 0 63 42 62 46
|
||||
0 1 4 0 63 42 58 62
|
||||
0 1 4 0 26 125 132 126
|
||||
0 1 4 0 125 5 115 128
|
||||
0 1 4 0 132 115 9 133
|
||||
0 1 4 0 126 128 133 10
|
||||
0 1 4 0 125 133 132 126
|
||||
0 1 4 0 125 133 126 128
|
||||
0 1 4 0 125 133 128 115
|
||||
0 1 4 0 125 133 115 132
|
||||
0 1 4 0 26 125 126 127
|
||||
0 1 4 0 125 5 128 95
|
||||
0 1 4 0 126 128 10 129
|
||||
0 1 4 0 127 95 129 6
|
||||
0 1 4 0 125 129 126 127
|
||||
0 1 4 0 125 129 127 95
|
||||
0 1 4 0 125 129 95 128
|
||||
0 1 4 0 125 129 128 126
|
||||
0 1 4 0 58 305 308 309
|
||||
0 1 4 0 305 37 283 262
|
||||
0 1 4 0 308 283 54 284
|
||||
0 1 4 0 309 262 284 53
|
||||
0 1 4 0 305 284 308 309
|
||||
0 1 4 0 305 284 309 262
|
||||
0 1 4 0 305 284 262 283
|
||||
0 1 4 0 305 284 283 308
|
||||
0 1 4 0 58 305 309 311
|
||||
0 1 4 0 305 37 262 297
|
||||
0 1 4 0 309 262 53 298
|
||||
0 1 4 0 311 297 298 57
|
||||
0 1 4 0 305 298 309 311
|
||||
0 1 4 0 305 298 311 297
|
||||
0 1 4 0 305 298 297 262
|
||||
0 1 4 0 305 298 262 309
|
||||
0 1 4 0 41 213 217 219
|
||||
0 1 4 0 213 20 191 220
|
||||
0 1 4 0 217 191 36 222
|
||||
0 1 4 0 219 220 222 40
|
||||
0 1 4 0 213 222 217 219
|
||||
0 1 4 0 213 222 219 220
|
||||
0 1 4 0 213 222 220 191
|
||||
0 1 4 0 213 222 191 217
|
||||
0 1 4 0 41 213 219 218
|
||||
0 1 4 0 213 20 220 124
|
||||
0 1 4 0 219 220 40 221
|
||||
0 1 4 0 218 124 221 24
|
||||
0 1 4 0 213 221 219 218
|
||||
0 1 4 0 213 221 218 124
|
||||
0 1 4 0 213 221 124 220
|
||||
0 1 4 0 213 221 220 219
|
||||
0 1 4 0 43 230 231 232
|
||||
0 1 4 0 230 22 141 110
|
||||
0 1 4 0 231 141 27 140
|
||||
0 1 4 0 232 110 140 23
|
||||
0 1 4 0 230 140 231 232
|
||||
0 1 4 0 230 140 232 110
|
||||
0 1 4 0 230 140 110 141
|
||||
0 1 4 0 230 140 141 231
|
||||
0 1 4 0 43 230 232 233
|
||||
0 1 4 0 230 22 110 211
|
||||
0 1 4 0 232 110 23 204
|
||||
0 1 4 0 233 211 204 39
|
||||
0 1 4 0 230 204 232 233
|
||||
0 1 4 0 230 204 233 211
|
||||
0 1 4 0 230 204 211 110
|
||||
0 1 4 0 230 204 110 232
|
||||
0 1 4 0 38 193 195 196
|
||||
0 1 4 0 193 17 93 197
|
||||
0 1 4 0 195 93 18 198
|
||||
0 1 4 0 196 197 198 34
|
||||
0 1 4 0 193 198 195 196
|
||||
0 1 4 0 193 198 196 197
|
||||
0 1 4 0 193 198 197 93
|
||||
0 1 4 0 193 198 93 195
|
||||
0 1 4 0 38 193 196 199
|
||||
0 1 4 0 193 17 197 184
|
||||
0 1 4 0 196 197 34 200
|
||||
0 1 4 0 199 184 200 33
|
||||
0 1 4 0 193 200 196 199
|
||||
0 1 4 0 193 200 199 184
|
||||
0 1 4 0 193 200 184 197
|
||||
0 1 4 0 193 200 197 196
|
||||
0 1 4 0 46 247 253 252
|
||||
0 1 4 0 247 25 239 150
|
||||
0 1 4 0 253 239 45 238
|
||||
0 1 4 0 252 150 238 29
|
||||
0 1 4 0 247 238 253 252
|
||||
0 1 4 0 247 238 252 150
|
||||
0 1 4 0 247 238 150 239
|
||||
0 1 4 0 247 238 239 253
|
||||
0 1 4 0 46 247 252 248
|
||||
0 1 4 0 247 25 150 165
|
||||
0 1 4 0 252 150 29 168
|
||||
0 1 4 0 248 165 168 30
|
||||
0 1 4 0 247 168 252 248
|
||||
0 1 4 0 247 168 248 165
|
||||
0 1 4 0 247 168 165 150
|
||||
0 1 4 0 247 168 150 252
|
||||
|
||||
# attr geom nodes
|
||||
boundary
|
||||
144
|
||||
1 2 0 5 1
|
||||
1 2 0 1 17
|
||||
1 2 0 17 16
|
||||
1 2 0 4 5
|
||||
1 2 0 20 4
|
||||
1 2 0 16 20
|
||||
1 2 1 6 2
|
||||
1 2 1 2 18
|
||||
1 2 1 18 17
|
||||
1 2 1 5 6
|
||||
1 2 2 7 3
|
||||
1 2 23 3 7
|
||||
1 2 2 3 19
|
||||
1 2 23 19 3
|
||||
1 2 2 19 18
|
||||
1 2 2 6 7
|
||||
1 2 4 9 5
|
||||
1 2 4 8 9
|
||||
1 2 4 24 8
|
||||
1 2 4 20 24
|
||||
1 2 6 11 7
|
||||
1 2 27 7 11
|
||||
1 2 27 23 7
|
||||
1 2 6 10 11
|
||||
1 2 8 13 9
|
||||
1 2 8 12 13
|
||||
1 2 29 13 12
|
||||
1 2 8 28 12
|
||||
1 2 29 12 28
|
||||
1 2 8 24 28
|
||||
1 2 9 14 10
|
||||
1 2 9 13 14
|
||||
1 2 30 14 13
|
||||
1 2 30 13 29
|
||||
1 2 10 15 11
|
||||
1 2 31 11 15
|
||||
1 2 31 27 11
|
||||
1 2 10 14 15
|
||||
1 2 31 15 14
|
||||
1 2 31 14 30
|
||||
1 2 16 17 33
|
||||
1 2 16 33 32
|
||||
1 2 16 36 20
|
||||
1 2 16 32 36
|
||||
1 2 39 19 23
|
||||
1 2 18 19 35
|
||||
1 2 39 35 19
|
||||
1 2 18 35 34
|
||||
1 2 45 29 28
|
||||
1 2 24 44 28
|
||||
1 2 45 28 44
|
||||
1 2 24 40 44
|
||||
1 2 47 27 31
|
||||
1 2 47 43 27
|
||||
1 2 47 31 30
|
||||
1 2 47 30 46
|
||||
1 2 32 33 49
|
||||
1 2 32 49 48
|
||||
1 2 53 48 49
|
||||
1 2 32 52 36
|
||||
1 2 32 48 52
|
||||
1 2 53 52 48
|
||||
1 2 33 34 50
|
||||
1 2 33 50 49
|
||||
1 2 54 49 50
|
||||
1 2 54 53 49
|
||||
1 2 55 35 39
|
||||
1 2 34 35 51
|
||||
1 2 55 51 35
|
||||
1 2 34 51 50
|
||||
1 2 55 50 51
|
||||
1 2 55 54 50
|
||||
1 2 57 52 53
|
||||
1 2 36 56 40
|
||||
1 2 36 52 56
|
||||
1 2 57 56 52
|
||||
1 2 59 39 43
|
||||
1 2 59 55 39
|
||||
1 2 59 54 55
|
||||
1 2 59 58 54
|
||||
1 2 61 56 57
|
||||
1 2 61 45 44
|
||||
1 2 40 60 44
|
||||
1 2 61 44 60
|
||||
1 2 40 56 60
|
||||
1 2 61 60 56
|
||||
1 2 62 57 58
|
||||
1 2 62 46 45
|
||||
1 2 62 45 61
|
||||
1 2 62 61 57
|
||||
1 2 63 43 47
|
||||
1 2 63 59 43
|
||||
1 2 63 58 59
|
||||
1 2 63 47 46
|
||||
1 2 63 46 62
|
||||
1 2 63 62 58
|
||||
2 2 5 115 128
|
||||
2 2 115 9 133
|
||||
2 2 128 133 10
|
||||
2 2 133 128 115
|
||||
2 2 5 128 95
|
||||
2 2 128 10 129
|
||||
2 2 95 129 6
|
||||
2 2 129 95 128
|
||||
2 2 58 309 308
|
||||
2 2 308 284 54
|
||||
2 2 309 53 284
|
||||
2 2 284 308 309
|
||||
2 2 58 311 309
|
||||
2 2 309 298 53
|
||||
2 2 311 57 298
|
||||
2 2 298 309 311
|
||||
2 2 20 191 220
|
||||
2 2 191 36 222
|
||||
2 2 220 222 40
|
||||
2 2 222 220 191
|
||||
2 2 20 220 124
|
||||
2 2 220 40 221
|
||||
2 2 124 221 24
|
||||
2 2 221 124 220
|
||||
2 2 43 232 231
|
||||
2 2 231 140 27
|
||||
2 2 232 23 140
|
||||
2 2 140 231 232
|
||||
2 2 43 233 232
|
||||
2 2 232 204 23
|
||||
2 2 233 39 204
|
||||
2 2 204 232 233
|
||||
2 2 17 93 197
|
||||
2 2 93 18 198
|
||||
2 2 197 198 34
|
||||
2 2 198 197 93
|
||||
2 2 17 197 184
|
||||
2 2 197 34 200
|
||||
2 2 184 200 33
|
||||
2 2 200 184 197
|
||||
2 2 46 252 253
|
||||
2 2 253 238 45
|
||||
2 2 252 29 238
|
||||
2 2 238 253 252
|
||||
2 2 46 248 252
|
||||
2 2 252 168 29
|
||||
2 2 248 30 168
|
||||
2 2 168 252 248
|
||||
|
||||
# vert_id p1 p2
|
||||
vertex_parents
|
||||
54
|
||||
93 17 18
|
||||
95 5 6
|
||||
110 22 23
|
||||
115 5 9
|
||||
124 20 24
|
||||
125 5 26
|
||||
126 10 26
|
||||
127 6 26
|
||||
128 5 10
|
||||
129 6 10
|
||||
132 9 26
|
||||
133 9 10
|
||||
140 23 27
|
||||
141 22 27
|
||||
150 25 29
|
||||
165 25 30
|
||||
168 29 30
|
||||
184 17 33
|
||||
191 20 36
|
||||
193 17 38
|
||||
195 18 38
|
||||
196 34 38
|
||||
197 17 34
|
||||
198 18 34
|
||||
199 33 38
|
||||
200 33 34
|
||||
204 23 39
|
||||
211 22 39
|
||||
213 20 41
|
||||
217 36 41
|
||||
218 24 41
|
||||
219 40 41
|
||||
220 20 40
|
||||
221 24 40
|
||||
222 36 40
|
||||
230 22 43
|
||||
231 27 43
|
||||
232 23 43
|
||||
233 39 43
|
||||
238 29 45
|
||||
239 25 45
|
||||
247 25 46
|
||||
248 30 46
|
||||
252 29 46
|
||||
253 45 46
|
||||
262 37 53
|
||||
283 37 54
|
||||
284 53 54
|
||||
297 37 57
|
||||
298 53 57
|
||||
305 37 58
|
||||
308 54 58
|
||||
309 53 58
|
||||
311 57 58
|
||||
|
||||
# top-level node coordinates
|
||||
coordinates
|
||||
64
|
||||
3
|
||||
0 0 0
|
||||
0.33333333 0 0
|
||||
0.66666667 0 0
|
||||
1 0 0
|
||||
0 0.33333333 0
|
||||
0.33333333 0.33333333 0
|
||||
0.66666667 0.33333333 0
|
||||
1 0.33333333 0
|
||||
0 0.66666667 0
|
||||
0.33333333 0.66666667 0
|
||||
0.66666667 0.66666667 0
|
||||
1 0.66666667 0
|
||||
0 1 0
|
||||
0.33333333 1 0
|
||||
0.66666667 1 0
|
||||
1 1 0
|
||||
0 0 0.33333333
|
||||
0.33333333 0 0.33333333
|
||||
0.66666667 0 0.33333333
|
||||
1 0 0.33333333
|
||||
0 0.33333333 0.33333333
|
||||
0.33333333 0.33333333 0.33333333
|
||||
0.66666667 0.33333333 0.33333333
|
||||
1 0.33333333 0.33333333
|
||||
0 0.66666667 0.33333333
|
||||
0.33333333 0.66666667 0.33333333
|
||||
0.66666667 0.66666667 0.33333333
|
||||
1 0.66666667 0.33333333
|
||||
0 1 0.33333333
|
||||
0.33333333 1 0.33333333
|
||||
0.66666667 1 0.33333333
|
||||
1 1 0.33333333
|
||||
0 0 0.66666667
|
||||
0.33333333 0 0.66666667
|
||||
0.66666667 0 0.66666667
|
||||
1 0 0.66666667
|
||||
0 0.33333333 0.66666667
|
||||
0.33333333 0.33333333 0.66666667
|
||||
0.66666667 0.33333333 0.66666667
|
||||
1 0.33333333 0.66666667
|
||||
0 0.66666667 0.66666667
|
||||
0.33333333 0.66666667 0.66666667
|
||||
0.66666667 0.66666667 0.66666667
|
||||
1 0.66666667 0.66666667
|
||||
0 1 0.66666667
|
||||
0.33333333 1 0.66666667
|
||||
0.66666667 1 0.66666667
|
||||
1 1 0.66666667
|
||||
0 0 1
|
||||
0.33333333 0 1
|
||||
0.66666667 0 1
|
||||
1 0 1
|
||||
0 0.33333333 1
|
||||
0.33333333 0.33333333 1
|
||||
0.66666667 0.33333333 1
|
||||
1 0.33333333 1
|
||||
0 0.66666667 1
|
||||
0.33333333 0.66666667 1
|
||||
0.66666667 0.66666667 1
|
||||
1 0.66666667 1
|
||||
0 1 1
|
||||
0.33333333 1 1
|
||||
0.66666667 1 1
|
||||
1 1 1
|
||||
|
||||
mfem_mesh_end
|
||||
@@ -0,0 +1,274 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../unit_tests.hpp"
|
||||
#include "mfem.hpp"
|
||||
#ifndef MFEM_USE_MPI
|
||||
#include "../../../fem/dfem/tuple.hpp"
|
||||
#endif
|
||||
|
||||
using namespace mfem;
|
||||
using namespace mfem::future;
|
||||
|
||||
namespace tuple_test
|
||||
{
|
||||
|
||||
// A payload that is not a scalar, mimicking what dFEM kernels actually store.
|
||||
using vec3 = tensor<real_t, 3>;
|
||||
using tuple3 = tuple<real_t, int, vec3>;
|
||||
|
||||
// mfem::future::tuple is no longer an aggregate: it derives from tuple_leaf
|
||||
// bases so that it can be defined for an arbitrary number of elements. These
|
||||
// checks pin down the properties that the aggregate used to provide for free
|
||||
// and that device kernels (which capture tuples by value) depend on.
|
||||
static_assert(std::is_trivially_copyable<tuple3>::value,
|
||||
"tuple must be trivially copyable to be captured by value in device kernels");
|
||||
static_assert(std::is_trivially_destructible<tuple3>::value,
|
||||
"tuple must be trivially destructible");
|
||||
static_assert(std::is_trivially_default_constructible<tuple3>::value,
|
||||
"tuple must be trivially default constructible");
|
||||
static_assert(std::is_trivially_copy_assignable<tuple3>::value,
|
||||
"tuple must be trivially copy assignable");
|
||||
static_assert(sizeof(tuple3) == sizeof(real_t) + sizeof(int) + sizeof(vec3) +
|
||||
(alignof(real_t) - sizeof(int)),
|
||||
"tuple must not be larger than the sum of its (padded) members");
|
||||
|
||||
// Size and element types, both through mfem::future and through the std
|
||||
// specializations that drive structured bindings.
|
||||
static_assert(tuple_size<tuple3>::value == 3, "");
|
||||
static_assert(std::tuple_size<tuple3>::value == 3, "");
|
||||
static_assert(std::is_same<tuple_element<0, tuple3>::type, real_t>::value, "");
|
||||
static_assert(std::is_same<tuple_element<1, tuple3>::type, int>::value, "");
|
||||
static_assert(std::is_same<tuple_element<2, tuple3>::type, vec3>::value, "");
|
||||
static_assert(std::is_same<std::tuple_element_t<0, tuple3>, real_t>::value, "");
|
||||
static_assert(std::is_same<std::tuple_element_t<2, tuple3>, vec3>::value, "");
|
||||
|
||||
// get must preserve the value category and constness of its argument.
|
||||
static_assert(std::is_same<decltype(get<1>(std::declval<tuple3&>())),
|
||||
int&>::value, "get on an lvalue must return an lvalue reference");
|
||||
static_assert(std::is_same<decltype(get<1>(std::declval<const tuple3&>())),
|
||||
const int&>::value,
|
||||
"get on a const lvalue must return a const lvalue reference");
|
||||
static_assert(std::is_same<decltype(get<1>(std::declval<tuple3&&>())),
|
||||
int&&>::value, "get on an rvalue must return an rvalue reference");
|
||||
static_assert(std::is_same<decltype(get<1>(std::declval<const tuple3&&>())),
|
||||
const int&&>::value,
|
||||
"get on a const rvalue must return a const rvalue reference");
|
||||
|
||||
// += and -= must return a reference, not a copy of the whole tuple.
|
||||
using tuple2 = tuple<real_t, vec3>;
|
||||
static_assert(std::is_same<decltype(std::declval<tuple2&>() +=
|
||||
std::declval<const tuple2&>()), tuple2&>::value,
|
||||
"operator+= must return a reference");
|
||||
static_assert(std::is_same<decltype(std::declval<tuple2&>() -=
|
||||
std::declval<const tuple2&>()), tuple2&>::value,
|
||||
"operator-= must return a reference");
|
||||
|
||||
// The element-wise constructor must stay implicit, so that the
|
||||
// copy-list-initialization forms that worked with the aggregate keep working.
|
||||
static_assert(std::is_convertible<int, tuple<int>>::value,
|
||||
"tuple's element-wise constructor must not be explicit");
|
||||
|
||||
// Constructing from an incompatible type must SFINAE out rather than hard-error,
|
||||
// so that the constructor does not poison type traits.
|
||||
struct not_a_number { };
|
||||
static_assert(!std::is_constructible<tuple<int, int>, int, not_a_number>::value,
|
||||
"");
|
||||
static_assert(!std::is_constructible<tuple<int, int>, int>::value,
|
||||
"arity mismatch must not be constructible");
|
||||
|
||||
// Usable at compile time.
|
||||
constexpr tuple<int, real_t> const_tuple {2, 3.0};
|
||||
static_assert(get<0>(const_tuple) == 2, "");
|
||||
|
||||
// Copy-list-initialization in a return statement (broken by an explicit ctor).
|
||||
tuple<int, real_t> returns_braced_init_list() { return {7, 8.0}; }
|
||||
|
||||
} // namespace tuple_test
|
||||
|
||||
using namespace tuple_test;
|
||||
|
||||
TEST_CASE("dFEM tuple structured bindings", "[dFEM]")
|
||||
{
|
||||
tuple3 t {1.0, 2, vec3{{3.0, 4.0, 5.0}}};
|
||||
|
||||
SECTION("binding by reference writes through")
|
||||
{
|
||||
auto &[a, b, c] = t;
|
||||
a = 10.0;
|
||||
b = 20;
|
||||
c(0) = 30.0;
|
||||
REQUIRE(get<0>(t) == 10.0_r);
|
||||
REQUIRE(get<1>(t) == 20);
|
||||
REQUIRE(get<2>(t)(0) == 30.0_r);
|
||||
}
|
||||
|
||||
SECTION("binding by value copies")
|
||||
{
|
||||
auto [a, b, c] = t;
|
||||
a = 10.0;
|
||||
b = 20;
|
||||
c(0) = 30.0;
|
||||
REQUIRE(get<0>(t) == 1.0_r);
|
||||
REQUIRE(get<1>(t) == 2);
|
||||
REQUIRE(get<2>(t)(0) == 3.0_r);
|
||||
}
|
||||
|
||||
SECTION("binding to const")
|
||||
{
|
||||
const auto &[a, b, c] = t;
|
||||
REQUIRE(a == 1.0_r);
|
||||
REQUIRE(b == 2);
|
||||
REQUIRE(c(2) == 5.0_r);
|
||||
static_assert(std::is_same<decltype(a), const real_t>::value, "");
|
||||
static_assert(std::is_same<decltype(c), const vec3>::value, "");
|
||||
}
|
||||
|
||||
SECTION("the bindings alias the tuple storage")
|
||||
{
|
||||
auto &[a, b, c] = t;
|
||||
REQUIRE(&a == &get<0>(t));
|
||||
REQUIRE(&b == &get<1>(t));
|
||||
REQUIRE(&c == &get<2>(t));
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("dFEM tuple construction", "[dFEM]")
|
||||
{
|
||||
SECTION("copy-list-initialization")
|
||||
{
|
||||
tuple<int, real_t> a = {1, 2.0};
|
||||
REQUIRE(get<0>(a) == 1);
|
||||
REQUIRE(get<1>(a) == 2.0_r);
|
||||
|
||||
const auto b = returns_braced_init_list();
|
||||
REQUIRE(get<0>(b) == 7);
|
||||
REQUIRE(get<1>(b) == 8.0_r);
|
||||
}
|
||||
|
||||
SECTION("direct initialization and CTAD")
|
||||
{
|
||||
tuple c {1, 2.0_r, vec3{{1.0, 2.0, 3.0}}};
|
||||
static_assert(std::is_same<decltype(c), tuple<int, real_t, vec3>>::value,
|
||||
"CTAD must decay the arguments");
|
||||
REQUIRE(get<1>(c) == 2.0_r);
|
||||
}
|
||||
|
||||
SECTION("make_tuple")
|
||||
{
|
||||
const auto d = make_tuple(1, 2.0_r);
|
||||
static_assert(std::is_same<decltype(d), const tuple<int, real_t>>::value, "");
|
||||
REQUIRE(get<0>(d) == 1);
|
||||
}
|
||||
|
||||
SECTION("copy and move construction preserve values")
|
||||
{
|
||||
tuple3 t {1.0, 2, vec3{{3.0, 4.0, 5.0}}};
|
||||
tuple3 copy(t);
|
||||
tuple3 moved(std::move(t));
|
||||
REQUIRE(get<1>(copy) == 2);
|
||||
REQUIRE(get<2>(moved)(1) == 4.0_r);
|
||||
}
|
||||
|
||||
SECTION("value initialization zeroes trivial members")
|
||||
{
|
||||
tuple<int, real_t> z {};
|
||||
REQUIRE(get<0>(z) == 0);
|
||||
REQUIRE(get<1>(z) == 0.0_r);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("dFEM tuple arithmetic", "[dFEM]")
|
||||
{
|
||||
const tuple2 x {1.0, vec3{{1.0, 2.0, 3.0}}};
|
||||
const tuple2 y {2.0, vec3{{4.0, 5.0, 6.0}}};
|
||||
|
||||
SECTION("element-wise binary operators")
|
||||
{
|
||||
const auto sum = x + y;
|
||||
REQUIRE(get<0>(sum) == 3.0_r);
|
||||
REQUIRE(get<1>(sum)(2) == 9.0_r);
|
||||
|
||||
const auto diff = y - x;
|
||||
REQUIRE(get<0>(diff) == 1.0_r);
|
||||
REQUIRE(get<1>(diff)(0) == 3.0_r);
|
||||
}
|
||||
|
||||
SECTION("compound assignment mutates in place and returns a reference")
|
||||
{
|
||||
tuple2 z = x;
|
||||
auto &ref = (z += y);
|
||||
REQUIRE(&ref == &z);
|
||||
REQUIRE(get<0>(z) == 3.0_r);
|
||||
REQUIRE(get<1>(z)(1) == 7.0_r);
|
||||
|
||||
auto &ref2 = (z -= y);
|
||||
REQUIRE(&ref2 == &z);
|
||||
REQUIRE(get<0>(z) == 1.0_r);
|
||||
REQUIRE(get<1>(z)(1) == 2.0_r);
|
||||
}
|
||||
|
||||
SECTION("scalar operators and unary minus")
|
||||
{
|
||||
const auto scaled = 2.0_r * x;
|
||||
REQUIRE(get<0>(scaled) == 2.0_r);
|
||||
REQUIRE(get<1>(scaled)(2) == 6.0_r);
|
||||
|
||||
const auto halved = x / 2.0_r;
|
||||
REQUIRE(get<0>(halved) == 0.5_r);
|
||||
|
||||
const auto negated = -x;
|
||||
REQUIRE(get<0>(negated) == -1.0_r);
|
||||
REQUIRE(get<1>(negated)(0) == -1.0_r);
|
||||
}
|
||||
|
||||
SECTION("apply")
|
||||
{
|
||||
const auto s = apply([](const real_t &a, const vec3 &b) { return a + b(0); },
|
||||
x);
|
||||
REQUIRE(s == 2.0_r);
|
||||
}
|
||||
}
|
||||
|
||||
// The tuples are captured by value in device kernels, so exercise a round trip
|
||||
// through device memory: construct, mutate through structured bindings and read
|
||||
// back on the device.
|
||||
TEST_CASE("dFEM tuple on device", "[dFEM][GPU]")
|
||||
{
|
||||
Vector res(4);
|
||||
auto d_res = res.Write();
|
||||
|
||||
forall(1, [=] MFEM_HOST_DEVICE (int)
|
||||
{
|
||||
tuple3 t {1.0, 2, vec3{{3.0, 4.0, 5.0}}};
|
||||
auto &[a, b, c] = t;
|
||||
a += static_cast<real_t>(b);
|
||||
c(0) = a;
|
||||
|
||||
tuple2 u {get<0>(t), get<2>(t)};
|
||||
u += tuple2 {1.0, vec3{{1.0, 1.0, 1.0}}};
|
||||
|
||||
d_res[0] = get<0>(u);
|
||||
d_res[1] = get<1>(u)(0);
|
||||
d_res[2] = get<1>(u)(1);
|
||||
d_res[3] = static_cast<real_t>(get<1>(t));
|
||||
|
||||
tuple2 v1{0_r, vec3{0_r, 0_r, 0_r}};
|
||||
tuple2 v2{0_r, vec3{0_r, 0_r, 0_r}};
|
||||
[[maybe_unused]] auto v = v1 + v2;
|
||||
});
|
||||
|
||||
res.HostRead();
|
||||
REQUIRE(std::as_const(res)(0) == 4.0_r);
|
||||
REQUIRE(std::as_const(res)(1) == 4.0_r);
|
||||
REQUIRE(std::as_const(res)(2) == 5.0_r);
|
||||
REQUIRE(std::as_const(res)(3) == 2.0_r);
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/integ/bilininteg_convection_kernels.hpp"
|
||||
|
||||
TEST_CASE("Convection Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
ConvectionIntegrator::AddSpecialization<2, 2, 4>();
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/integ/bilininteg_hcurl_kernels.hpp"
|
||||
|
||||
TEST_CASE("CurlCurl Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
CurlCurlIntegrator::AddSpecialization<3, 2, 4>();
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/integ/bilininteg_dgdiffusion_kernels.hpp"
|
||||
|
||||
TEST_CASE("DGDiffusion Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
DGDiffusionIntegrator::AddSpecialization<2, 2, 4>();
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/dgmassinv_kernels.hpp"
|
||||
|
||||
TEST_CASE("DGMassInverse Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
DGMassInverse::CGKernels::Specialization<2, 1, 2>::Add();
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/integ/bilininteg_dgtrace_kernels.hpp"
|
||||
|
||||
TEST_CASE("DGTrace Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
DGTraceIntegrator::AddSpecialization<2, 2, 3>();
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/integ/bilininteg_diffusion_kernels.hpp"
|
||||
|
||||
TEST_CASE("Diffusion Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
DiffusionIntegrator::AddSpecialization<2, 1, 5>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2, 2, 3>();
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/integ/bilininteg_mass_kernels.hpp"
|
||||
|
||||
TEST_CASE("Mass Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
MassIntegrator::AddSpecialization<2, 1, 3>();
|
||||
MassIntegrator::AddSimplexSpecialization<2, 2, 4>();
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/qinterp/det.hpp"
|
||||
|
||||
TEST_CASE("QInterp Det Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
QuadratureInterpolator::AddDetSpecializations<2, 3, 2, 2>();
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/qinterp/eval.hpp"
|
||||
|
||||
TEST_CASE("QInterp Eval Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
QuadratureInterpolator::AddEvalSpecializations<2, 1, 1, 2>();
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/qinterp/eval_hdiv.hpp"
|
||||
|
||||
TEST_CASE("QInterp Eval HDiv Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
QuadratureInterpolator::TensorEvalHDivKernels::Specialization<
|
||||
2, QVectorLayout::byNODES, QuadratureInterpolator::PHYSICAL_VALUES, 2,
|
||||
4>::Add();
|
||||
QuadratureInterpolator::TensorEvalHDivKernels::Specialization<
|
||||
2, QVectorLayout::byNODES, QuadratureInterpolator::PHYSICAL_MAGNITUDES, 2,
|
||||
4>::Add();
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/qinterp/grad.hpp"
|
||||
|
||||
TEST_CASE("QInterp Grad Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
QuadratureInterpolator::AddGradSpecializations<
|
||||
2, QVectorLayout::byNODES, false, 1, 3, 3, 1>();
|
||||
QuadratureInterpolator::AddGradSpecializations<
|
||||
2, QVectorLayout::byNODES, true, 1, 3, 3, 1>();
|
||||
QuadratureInterpolator::AddCollocatedGradSpecializations<
|
||||
2, QVectorLayout::byNODES, false, 1, 2, 1>();
|
||||
QuadratureInterpolator::AddCollocatedGradSpecializations<
|
||||
2, QVectorLayout::byNODES, true, 1, 2, 1>();
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/qinterp/eval.hpp"
|
||||
|
||||
TEST_CASE("QInterp TensorEval Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
QuadratureInterpolator::AddTensorEvalSpecializations<
|
||||
2, QVectorLayout::byNODES, 1, 3, 3, 2>();
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/integ/bilininteg_vecdiffusion_pa.hpp"
|
||||
|
||||
TEST_CASE("VectorDiffusion Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
VectorDiffusionIntegrator::AddSpecialization<2, 2, 2, 3>();
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
/// Tests which make sure adding user-defined kernel specializations work.
|
||||
/// These tests are compile/link-only tests
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include "fem/integ/bilininteg_vecmass_pa.hpp"
|
||||
|
||||
TEST_CASE("Vector Mass Kernel Specializations", "[Specializations]")
|
||||
{
|
||||
using namespace mfem;
|
||||
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 2, 4>::Add();
|
||||
}
|
||||
@@ -3451,4 +3451,81 @@ TEST_CASE("2D Bilinear Scalar Weak Curl Cross Integrators",
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("2D Bilinear Scalar Curl Integrator PartialAssembly",
|
||||
"[MixedScalarCurlIntegrator]"
|
||||
"[BilinearFormIntegrator]"
|
||||
"[NonlinearFormIntegrator]"
|
||||
"[GPU]")
|
||||
{
|
||||
int order = 2, n = 1, dim = 2;
|
||||
double tol = 1e-9;
|
||||
|
||||
Mesh mesh = Mesh::MakeCartesian2D(n, n, Element::QUADRILATERAL, 1, 2.0, 3.0);
|
||||
|
||||
VectorFunctionCoefficient F2_coef(dim, F2);
|
||||
FunctionCoefficient q2_coef(q2);
|
||||
|
||||
SECTION("Operators on ND")
|
||||
{
|
||||
ND_FECollection fec_nd(order, dim);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
GridFunction f_nd(&fespace_nd); f_nd.ProjectCoefficient(F2_coef);
|
||||
|
||||
for (int map_type = (int)FiniteElement::VALUE;
|
||||
map_type <= (int)FiniteElement::INTEGRAL; map_type++)
|
||||
{
|
||||
SECTION("Mapping ND to L2 (" +
|
||||
MapTypeName((FiniteElement::MapType)map_type) + ")")
|
||||
{
|
||||
L2_FECollection fec_l2(order - 1, dim,
|
||||
BasisType::GaussLegendre,
|
||||
(FiniteElement::MapType)map_type);
|
||||
FiniteElementSpace fespace_l2(&mesh, &fec_l2);
|
||||
|
||||
Vector tmp_l2(fespace_l2.GetNDofs());
|
||||
Vector tmp_l2_pa(fespace_l2.GetNDofs());
|
||||
|
||||
SECTION("Without Coefficient")
|
||||
{
|
||||
MixedBilinearForm blf_fa(&fespace_nd, &fespace_l2);
|
||||
blf_fa.AddDomainIntegrator(new MixedScalarCurlIntegrator());
|
||||
blf_fa.Assemble();
|
||||
blf_fa.Finalize();
|
||||
|
||||
blf_fa.Mult(f_nd, tmp_l2);
|
||||
|
||||
MixedBilinearForm blf_pa(&fespace_nd, &fespace_l2);
|
||||
blf_pa.SetAssemblyLevel(mfem::AssemblyLevel::PARTIAL);
|
||||
blf_pa.AddDomainIntegrator(new MixedScalarCurlIntegrator());
|
||||
blf_pa.Assemble();
|
||||
|
||||
blf_pa.Mult(f_nd, tmp_l2_pa);
|
||||
tmp_l2_pa -= tmp_l2;
|
||||
REQUIRE(tmp_l2_pa.Normlinf() < tol);
|
||||
}
|
||||
SECTION("With Scalar Coefficient")
|
||||
{
|
||||
MixedBilinearForm blf_fa(&fespace_nd, &fespace_l2);
|
||||
blf_fa.AddDomainIntegrator(
|
||||
new MixedScalarCurlIntegrator(q2_coef));
|
||||
blf_fa.Assemble();
|
||||
blf_fa.Finalize();
|
||||
|
||||
blf_fa.Mult(f_nd, tmp_l2);
|
||||
|
||||
MixedBilinearForm blf_pa(&fespace_nd, &fespace_l2);
|
||||
blf_pa.SetAssemblyLevel(mfem::AssemblyLevel::PARTIAL);
|
||||
blf_pa.AddDomainIntegrator(new MixedScalarCurlIntegrator(q2_coef));
|
||||
blf_pa.Assemble();
|
||||
|
||||
blf_pa.Mult(f_nd, tmp_l2_pa);
|
||||
tmp_l2_pa -= tmp_l2;
|
||||
REQUIRE(tmp_l2_pa.Normlinf() < tol);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace bilininteg_2d
|
||||
|
||||
@@ -1069,4 +1069,238 @@ TEST_CASE("Exact Sequence Properties: d(df)=0",
|
||||
}
|
||||
}
|
||||
|
||||
template <class A, class B>
|
||||
static void TestCurl(FiniteElementSpace &dom_fes, FiniteElementSpace &ran_fes,
|
||||
A coeff, B dcoeff)
|
||||
{
|
||||
real_t tol = 1e-10;
|
||||
DiscreteLinearOperator CurlFA(&dom_fes, &ran_fes);
|
||||
CurlFA.AddDomainInterpolator(new CurlInterpolator());
|
||||
CurlFA.Assemble();
|
||||
CurlFA.Finalize();
|
||||
|
||||
SparseMatrix &Curl = CurlFA.SpMat();
|
||||
GridFunction x(&dom_fes), y_fa(&ran_fes), y(&ran_fes);
|
||||
x.ProjectCoefficient(coeff);
|
||||
y.ProjectCoefficient(dcoeff);
|
||||
REQUIRE(x.Size() == Curl.Width());
|
||||
REQUIRE(y_fa.Size() == Curl.Height());
|
||||
Curl.Mult(x, y_fa);
|
||||
y_fa -= y;
|
||||
REQUIRE(y_fa.Normlinf() < tol);
|
||||
}
|
||||
|
||||
template<class Coeff, class TCoeff>
|
||||
static void CompareCurlPA(FiniteElementSpace& dom_fes,
|
||||
FiniteElementSpace &ran_fes,
|
||||
Coeff coeff, TCoeff tcoeff)
|
||||
{
|
||||
real_t tol = 1e-10;
|
||||
DiscreteLinearOperator CurlFA(&dom_fes, &ran_fes);
|
||||
CurlFA.AddDomainInterpolator(new CurlInterpolator());
|
||||
CurlFA.Assemble();
|
||||
CurlFA.Finalize();
|
||||
DiscreteLinearOperator CurlPA(&dom_fes, &ran_fes);
|
||||
CurlPA.AddDomainInterpolator(new CurlInterpolator());
|
||||
CurlPA.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
CurlPA.Assemble();
|
||||
|
||||
SparseMatrix &Curl = CurlFA.SpMat();
|
||||
GridFunction x(&dom_fes), y_fa(&ran_fes), y_pa(&ran_fes);
|
||||
x.ProjectCoefficient(coeff);
|
||||
REQUIRE(x.Size() == Curl.Width());
|
||||
REQUIRE(y_fa.Size() == Curl.Height());
|
||||
REQUIRE(x.Size() == CurlPA.Width());
|
||||
REQUIRE(y_pa.Size() == CurlPA.Height());
|
||||
Curl.Mult(x, y_fa);
|
||||
CurlPA.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE(y_pa.Normlinf() < tol);
|
||||
// transpose
|
||||
y_fa.ProjectCoefficient(tcoeff);
|
||||
GridFunction x_fa(&dom_fes), x_pa(&dom_fes);
|
||||
Curl.MultTranspose(y_fa, x_fa);
|
||||
CurlPA.MultTranspose(y_fa, x_pa);
|
||||
x_pa -= x_fa;
|
||||
REQUIRE(x_pa.Normlinf() < tol);
|
||||
}
|
||||
|
||||
TEST_CASE("Partial Assemble Linear Interpolator",
|
||||
"[CurlInterpolator]"
|
||||
"[GPU]")
|
||||
{
|
||||
constexpr int maxOrder = 3;
|
||||
auto order = GENERATE_COPY(range(1, maxOrder + 1));
|
||||
CAPTURE(order);
|
||||
|
||||
auto dim = GENERATE(2, 3);
|
||||
CAPTURE(dim);
|
||||
|
||||
int n = 3;
|
||||
|
||||
Mesh mesh;
|
||||
|
||||
switch (dim)
|
||||
{
|
||||
case 2:
|
||||
mesh =
|
||||
Mesh::MakeCartesian2D(n, n, Element::QUADRILATERAL, true, 2.0, 3.0);
|
||||
break;
|
||||
case 3:
|
||||
mesh = Mesh::MakeCartesian3D(n, n, n, Element::HEXAHEDRON, 2.0, 3.0, 5.0);
|
||||
break;
|
||||
}
|
||||
|
||||
// domain spaces
|
||||
H1_FECollection fec_h1(order, dim);
|
||||
FiniteElementSpace fespace_h1(&mesh, &fec_h1);
|
||||
ND_FECollection fec_nd(order, dim);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
// range spaces
|
||||
RT_FECollection fec_rt(order - 1, dim);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
L2_FECollection fec_l2(order - 1, dim, BasisType::GaussLegendre,
|
||||
FiniteElement::INTEGRAL);
|
||||
FiniteElementSpace fespace_l2(&mesh, &fec_l2);
|
||||
|
||||
switch (dim)
|
||||
{
|
||||
case 2:
|
||||
{
|
||||
FunctionCoefficient coeff([](const Vector &x)
|
||||
{ return sin(2 * M_PI * x[1] / 3) - cos(2 * M_PI * x[0] / 2); });
|
||||
VectorFunctionCoefficient vcoeff(2, [](const Vector &x, Vector &y)
|
||||
{
|
||||
y.SetSize(2);
|
||||
y[0] = -cos(2 * M_PI * x[1] / 3);
|
||||
y[1] = sin(2 * M_PI * x[0] / 2);
|
||||
});
|
||||
// out of plane H1 -> in-plane RT
|
||||
SECTION("H1 to RT")
|
||||
{
|
||||
CompareCurlPA(fespace_h1, fespace_rt, coeff, vcoeff);
|
||||
}
|
||||
// in-plane ND -> out of plane L2
|
||||
SECTION("ND to L2")
|
||||
{
|
||||
CompareCurlPA(fespace_nd, fespace_l2, vcoeff, coeff);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 3:
|
||||
{
|
||||
VectorFunctionCoefficient coeff(3, [](const Vector &x, Vector &y)
|
||||
{
|
||||
y.SetSize(3);
|
||||
y[0] = sin(2 * M_PI * x[2] / 5) - cos(2 * M_PI * x[1] / 3);
|
||||
y[1] = sin(2 * M_PI * x[0] / 2) - cos(2 * M_PI * x[2] / 5);
|
||||
y[2] = sin(2 * M_PI * x[1] / 3) - cos(2 * M_PI * x[0] / 2);
|
||||
});
|
||||
CompareCurlPA(fespace_nd, fespace_rt, coeff, coeff);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("Curl Linear Interpolator",
|
||||
"[CurlInterpolator]"
|
||||
"[GPU]")
|
||||
{
|
||||
int order = 2;
|
||||
|
||||
auto type = (Element::Type)GENERATE(range((int)Element::TRIANGLE,
|
||||
(int)Element::PYRAMID + 1));
|
||||
CAPTURE(type);
|
||||
|
||||
int n = 3;
|
||||
|
||||
Mesh mesh;
|
||||
|
||||
int dim;
|
||||
|
||||
if (type < (int)Element::TETRAHEDRON)
|
||||
{
|
||||
dim = 2;
|
||||
mesh = Mesh::MakeCartesian2D(n, n, (Element::Type)type, 1, 2.0, 3.0);
|
||||
}
|
||||
else
|
||||
{
|
||||
dim = 3;
|
||||
mesh = Mesh::MakeCartesian3D(n, n, n, (Element::Type)type,
|
||||
2.0, 3.0, 5.0);
|
||||
}
|
||||
|
||||
// domain spaces
|
||||
H1_FECollection fec_h1(order, dim);
|
||||
FiniteElementSpace fespace_h1(&mesh, &fec_h1);
|
||||
ND_FECollection fec_nd(order, dim);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
// range spaces
|
||||
RT_FECollection fec_rt(order - 1, dim);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
L2_FECollection fec_l2(order - 1, dim, BasisType::GaussLegendre,
|
||||
FiniteElement::INTEGRAL);
|
||||
FiniteElementSpace fespace_l2(&mesh, &fec_l2);
|
||||
|
||||
switch (dim)
|
||||
{
|
||||
case 2:
|
||||
{
|
||||
// out of plane H1 -> in-plane RT
|
||||
SECTION("H1 to RT")
|
||||
{
|
||||
FunctionCoefficient coeff([](const Vector &x)
|
||||
{
|
||||
return 1 - 2 * x[0] + 3 * x[1];
|
||||
});
|
||||
VectorFunctionCoefficient dcoeff(2, [](const Vector &x, Vector &y)
|
||||
{
|
||||
y.SetSize(2);
|
||||
// d Ez/dy
|
||||
y[0] = 3;
|
||||
// -d Ez/dx
|
||||
y[1] = 2;
|
||||
});
|
||||
|
||||
TestCurl(fespace_h1, fespace_rt, coeff, dcoeff);
|
||||
}
|
||||
// in-plane ND -> out of plane L2
|
||||
SECTION("ND to L2")
|
||||
{
|
||||
VectorFunctionCoefficient coeff(2, [](const Vector &x, Vector &y)
|
||||
{
|
||||
y.SetSize(2);
|
||||
y[0] = 1 - 2 * x[0] + 3 * x[1];
|
||||
y[1] = 2 * (1 - 2 * x[0] + 3 * x[1]);
|
||||
});
|
||||
FunctionCoefficient dcoeff([](const Vector &x)
|
||||
{ return 2 * (-2) - 3; });
|
||||
TestCurl(fespace_nd, fespace_l2, coeff, dcoeff);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 3:
|
||||
{
|
||||
VectorFunctionCoefficient coeff(3, [](const Vector &x, Vector &y)
|
||||
{
|
||||
y.SetSize(3);
|
||||
y[0] = 1 + 2 * x[0] - 3 * x[1] + 4 * x[2];
|
||||
y[1] = 4 + 3 * x[0] - 2 * x[1] + 1 * x[2];
|
||||
y[2] = 2 - 1 * x[0] + 4 * x[1] - 3 * x[2];
|
||||
});
|
||||
VectorFunctionCoefficient dcoeff(3, [](const Vector &x, Vector &y)
|
||||
{
|
||||
y.SetSize(3);
|
||||
y[0] = 4 - 1;
|
||||
y[1] = 4 + 1;
|
||||
y[2] = 3 + 3;
|
||||
});
|
||||
TestCurl(fespace_nd, fespace_rt, coeff, dcoeff);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace lin_interp
|
||||
|
||||
@@ -214,7 +214,14 @@ TEST_CASE("LOR AMS", "[LOR][BatchedLOR][AMS][Parallel][GPU]")
|
||||
ParFiniteElementSpace vert_fespace(edge_fespace.GetParMesh(), &vert_fec);
|
||||
|
||||
ParDiscreteLinearOperator grad(&vert_fespace, &edge_fespace);
|
||||
grad.AddDomainInterpolator(new GradientInterpolator);
|
||||
if (space_type == RT)
|
||||
{
|
||||
grad.AddDomainInterpolator(new CurlInterpolator);
|
||||
}
|
||||
else
|
||||
{
|
||||
grad.AddDomainInterpolator(new GradientInterpolator);
|
||||
}
|
||||
grad.Assemble();
|
||||
grad.Finalize();
|
||||
std::unique_ptr<HypreParMatrix> G(grad.ParallelAssemble());
|
||||
|
||||
@@ -0,0 +1,323 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
|
||||
namespace
|
||||
{
|
||||
constexpr real_t a_coef = 1.0;
|
||||
constexpr real_t b_coef = 2.0;
|
||||
constexpr real_t c_coef = 3.0;
|
||||
constexpr real_t omega_val = 10.0;
|
||||
|
||||
real_t V_exact_fn(const Vector &x)
|
||||
{
|
||||
return a_coef*x[0] + b_coef*x[1] + c_coef*x[2];
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("Mixed Sesquilinear Form", "[MixedSesquilinearForm]")
|
||||
{
|
||||
const bool cross = GENERATE(false, true);
|
||||
const auto conv = GENERATE(ComplexOperator::HERMITIAN,
|
||||
ComplexOperator::BLOCK_SYMMETRIC);
|
||||
CAPTURE(cross, int(conv));
|
||||
|
||||
Mesh mesh = Mesh::MakeCartesian3D(10, 10, 1, Element::HEXAHEDRON);
|
||||
H1_FECollection fec_h1(1, mesh.Dimension());
|
||||
FiniteElementSpace fespace_h1(&mesh, &fec_h1);
|
||||
ND_FECollection fec_nd(1, mesh.Dimension());
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
Array<int> dbc_bdr(mesh.bdr_attributes.Max());
|
||||
dbc_bdr = 1;
|
||||
Array<int> ess_tdof_list_h1;
|
||||
Array<int> ess_tdof_list_nd;
|
||||
fespace_h1.GetEssentialTrueDofs(dbc_bdr, ess_tdof_list_h1);
|
||||
fespace_nd.GetEssentialTrueDofs(dbc_bdr, ess_tdof_list_nd);
|
||||
|
||||
ConstantCoefficient omega(omega_val);
|
||||
ConstantCoefficient neg_omega(-omega_val);
|
||||
ConstantCoefficient half(0.5);
|
||||
FunctionCoefficient V_exact_real(V_exact_fn);
|
||||
const real_t den = cross ? 2.0*omega_val : omega_val;
|
||||
Vector A_vec({a_coef/den, b_coef/den, c_coef/den});
|
||||
VectorConstantCoefficient A_exact_imag(A_vec);
|
||||
A_vec *= (cross ? -1.0 : 0.0);
|
||||
VectorConstantCoefficient A_exact_real(A_vec);
|
||||
|
||||
ComplexGridFunction V(&fespace_h1);
|
||||
ComplexGridFunction A(&fespace_nd);
|
||||
|
||||
V = 0.0;
|
||||
A = 0.0;
|
||||
V.real().ProjectBdrCoefficient(V_exact_real, dbc_bdr);
|
||||
A.real().ProjectBdrCoefficientTangent(A_exact_real, dbc_bdr);
|
||||
A.imag().ProjectBdrCoefficientTangent(A_exact_imag, dbc_bdr);
|
||||
|
||||
ComplexLinearForm b_h1(&fespace_h1);
|
||||
b_h1 = 0.0;
|
||||
b_h1.Assemble();
|
||||
ComplexLinearForm b_nd(&fespace_nd);
|
||||
b_nd = 0.0;
|
||||
b_nd.Assemble();
|
||||
|
||||
// Add integrators to the blocks
|
||||
MixedSesquilinearForm a_h1_nd(&fespace_h1, &fespace_nd, conv);
|
||||
a_h1_nd.AddDomainIntegrator(cross ? new MixedVectorGradientIntegrator(half)
|
||||
: new MixedVectorGradientIntegrator,
|
||||
cross ? new MixedVectorGradientIntegrator(half)
|
||||
: nullptr);
|
||||
a_h1_nd.Assemble();
|
||||
|
||||
MixedSesquilinearForm a_nd_h1(&fespace_nd, &fespace_h1, conv);
|
||||
a_nd_h1.AddDomainIntegrator(
|
||||
cross ? new MixedVectorWeakDivergenceIntegrator(neg_omega) : nullptr,
|
||||
new MixedVectorWeakDivergenceIntegrator(neg_omega));
|
||||
a_nd_h1.Assemble();
|
||||
|
||||
SesquilinearForm a_h1(&fespace_h1, conv);
|
||||
a_h1.AddDomainIntegrator(new DiffusionIntegrator, nullptr);
|
||||
a_h1.Assemble();
|
||||
|
||||
SesquilinearForm a_nd(&fespace_nd, conv);
|
||||
a_nd.AddDomainIntegrator(new CurlCurlIntegrator, nullptr);
|
||||
a_nd.AddDomainIntegrator(nullptr, new VectorFEMassIntegrator(omega));
|
||||
a_nd.Assemble();
|
||||
|
||||
// Set block offsets (doubled for real+imag)
|
||||
mfem::Array<int> bOffsets(3);
|
||||
bOffsets[0] = 0;
|
||||
bOffsets[1] = 2 * fespace_h1.GetTrueVSize();
|
||||
bOffsets[2] = 2 * fespace_nd.GetTrueVSize();
|
||||
bOffsets.PartialSum();
|
||||
|
||||
OperatorPtr A_h1, A_nd, A_h1_nd, A_nd_h1;
|
||||
BlockVector trueX(bOffsets), trueRHS(bOffsets);
|
||||
Vector B_h1, X_h1, B_nd, X_nd;
|
||||
|
||||
trueX = 0.0;
|
||||
trueRHS = 0.0;
|
||||
|
||||
// Form the diagonal entries
|
||||
a_h1.FormLinearSystem(ess_tdof_list_h1, V, b_h1, A_h1, X_h1, B_h1);
|
||||
a_nd.FormLinearSystem(ess_tdof_list_nd, A, b_nd, A_nd, X_nd, B_nd);
|
||||
|
||||
trueX.GetBlock(0) = X_h1;
|
||||
trueX.GetBlock(1) = X_nd;
|
||||
trueRHS.GetBlock(0) += B_h1;
|
||||
trueRHS.GetBlock(1) += B_nd;
|
||||
|
||||
// Form the off-diagonal entries
|
||||
a_h1_nd.FormRectangularLinearSystem(ess_tdof_list_h1, ess_tdof_list_nd, V, b_nd,
|
||||
A_h1_nd, X_h1, B_nd);
|
||||
a_nd_h1.FormRectangularLinearSystem(ess_tdof_list_nd, ess_tdof_list_h1, A, b_h1,
|
||||
A_nd_h1, X_nd, B_h1);
|
||||
|
||||
trueRHS.GetBlock(0) += B_h1;
|
||||
trueRHS.GetBlock(1) += B_nd;
|
||||
|
||||
auto *Ah1 = A_h1.As<ComplexSparseMatrix>();
|
||||
auto *And = A_nd.As<ComplexSparseMatrix>();
|
||||
auto *Ah1nd = A_h1_nd.As<ComplexSparseMatrix>();
|
||||
auto *Andh1 = A_nd_h1.As<ComplexSparseMatrix>();
|
||||
|
||||
BlockOperator blockOp(bOffsets);
|
||||
blockOp.SetBlock(0, 0, Ah1);
|
||||
blockOp.SetBlock(1, 1, And);
|
||||
blockOp.SetBlock(0, 1, Andh1);
|
||||
blockOp.SetBlock(1, 0, Ah1nd);
|
||||
|
||||
SparseMatrix *Sh1 = Ah1->GetSystemMatrix();
|
||||
SparseMatrix *Snd = And->GetSystemMatrix();
|
||||
GSSmoother smoothSh1(*Sh1), smoothSnd(*Snd);
|
||||
|
||||
BlockDiagonalPreconditioner P(bOffsets);
|
||||
P.SetDiagonalBlock(0, &smoothSh1);
|
||||
P.SetDiagonalBlock(1, &smoothSnd);
|
||||
|
||||
GMRESSolver gmres;
|
||||
gmres.SetOperator(blockOp);
|
||||
gmres.SetPreconditioner(P);
|
||||
gmres.SetAbsTol(1e-10);
|
||||
gmres.SetMaxIter(2000);
|
||||
gmres.SetKDim(200);
|
||||
gmres.SetPrintLevel(2);
|
||||
gmres.Mult(trueRHS, trueX);
|
||||
|
||||
delete Sh1;
|
||||
delete Snd;
|
||||
|
||||
V = trueX.GetBlock(0);
|
||||
A = trueX.GetBlock(1);
|
||||
|
||||
// Check solution
|
||||
ConstantCoefficient zero(0.0);
|
||||
|
||||
real_t err_V = V.ComputeL2Error(V_exact_real, zero);
|
||||
real_t err_A = A.ComputeL2Error(A_exact_real, A_exact_imag);
|
||||
|
||||
REQUIRE(err_V == MFEM_Approx(0.0, 1e-5));
|
||||
REQUIRE(err_A == MFEM_Approx(0.0, 1e-5));
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#ifdef MFEM_USE_SUPERLU
|
||||
|
||||
TEST_CASE("Parallel Mixed Sesquilinear Form",
|
||||
"[MixedSesquilinearForm][Parallel]")
|
||||
{
|
||||
// See the serial test above for the manufactured solution
|
||||
const bool cross = GENERATE(false, true);
|
||||
const auto conv = GENERATE(ComplexOperator::HERMITIAN,
|
||||
ComplexOperator::BLOCK_SYMMETRIC);
|
||||
CAPTURE(cross, int(conv));
|
||||
|
||||
Mesh mesh = Mesh::MakeCartesian3D(10, 10, 1, Element::HEXAHEDRON);
|
||||
ParMesh par_mesh(MPI_COMM_WORLD, mesh);
|
||||
H1_FECollection fec_h1(1, mesh.Dimension());
|
||||
ParFiniteElementSpace fespace_h1(&par_mesh, &fec_h1);
|
||||
ND_FECollection fec_nd(1, mesh.Dimension());
|
||||
ParFiniteElementSpace fespace_nd(&par_mesh, &fec_nd);
|
||||
|
||||
Array<int> dbc_bdr(par_mesh.bdr_attributes.Max());
|
||||
dbc_bdr = 1;
|
||||
Array<int> ess_tdof_list_h1;
|
||||
Array<int> ess_tdof_list_nd;
|
||||
fespace_h1.GetEssentialTrueDofs(dbc_bdr, ess_tdof_list_h1);
|
||||
fespace_nd.GetEssentialTrueDofs(dbc_bdr, ess_tdof_list_nd);
|
||||
|
||||
ConstantCoefficient omega(omega_val);
|
||||
ConstantCoefficient neg_omega(-omega_val);
|
||||
ConstantCoefficient half(0.5);
|
||||
FunctionCoefficient V_exact_real(V_exact_fn);
|
||||
const real_t den = cross ? 2.0*omega_val : omega_val;
|
||||
Vector A_vec({a_coef/den, b_coef/den, c_coef/den});
|
||||
VectorConstantCoefficient A_exact_imag(A_vec);
|
||||
A_vec *= (cross ? -1.0 : 0.0);
|
||||
VectorConstantCoefficient A_exact_real(A_vec);
|
||||
|
||||
ParComplexGridFunction V(&fespace_h1);
|
||||
ParComplexGridFunction A(&fespace_nd);
|
||||
|
||||
V = 0.0;
|
||||
A = 0.0;
|
||||
V.real().ProjectBdrCoefficient(V_exact_real, dbc_bdr);
|
||||
A.real().ProjectBdrCoefficientTangent(A_exact_real, dbc_bdr);
|
||||
A.imag().ProjectBdrCoefficientTangent(A_exact_imag, dbc_bdr);
|
||||
|
||||
ParComplexLinearForm b_h1(&fespace_h1);
|
||||
b_h1 = 0.0;
|
||||
b_h1.Assemble();
|
||||
ParComplexLinearForm b_nd(&fespace_nd);
|
||||
b_nd = 0.0;
|
||||
b_nd.Assemble();
|
||||
|
||||
// Add integrators to the blocks
|
||||
ParMixedSesquilinearForm a_h1_nd(&fespace_h1, &fespace_nd, conv);
|
||||
a_h1_nd.AddDomainIntegrator(cross ? new MixedVectorGradientIntegrator(half)
|
||||
: new MixedVectorGradientIntegrator,
|
||||
cross ? new MixedVectorGradientIntegrator(half)
|
||||
: nullptr);
|
||||
a_h1_nd.Assemble();
|
||||
|
||||
ParMixedSesquilinearForm a_nd_h1(&fespace_nd, &fespace_h1, conv);
|
||||
a_nd_h1.AddDomainIntegrator(
|
||||
cross ? new MixedVectorWeakDivergenceIntegrator(neg_omega) : nullptr,
|
||||
new MixedVectorWeakDivergenceIntegrator(neg_omega));
|
||||
a_nd_h1.Assemble();
|
||||
|
||||
ParSesquilinearForm a_h1(&fespace_h1, conv);
|
||||
a_h1.AddDomainIntegrator(new DiffusionIntegrator, nullptr);
|
||||
a_h1.Assemble();
|
||||
|
||||
ParSesquilinearForm a_nd(&fespace_nd, conv);
|
||||
a_nd.AddDomainIntegrator(new CurlCurlIntegrator, nullptr);
|
||||
a_nd.AddDomainIntegrator(nullptr, new VectorFEMassIntegrator(omega));
|
||||
a_nd.Assemble();
|
||||
|
||||
mfem::Array2D<const mfem::HypreParMatrix *> h_blocks;
|
||||
h_blocks.SetSize(2, 2);
|
||||
h_blocks = nullptr;
|
||||
|
||||
// Set block offsets
|
||||
mfem::Array<int> bOffsets(3);
|
||||
bOffsets[0] = 0;
|
||||
bOffsets[1] = 2 * fespace_h1.TrueVSize();
|
||||
bOffsets[2] = 2 * fespace_nd.TrueVSize();
|
||||
bOffsets.PartialSum();
|
||||
|
||||
OperatorPtr A_h1, A_nd, A_h1_nd, A_nd_h1;
|
||||
BlockVector trueX(bOffsets), trueRHS(bOffsets);
|
||||
Vector B_h1, X_h1, B_nd, X_nd;
|
||||
|
||||
trueX = 0.0;
|
||||
trueRHS = 0.0;
|
||||
|
||||
// Form the diagonal entries
|
||||
a_h1.FormLinearSystem(ess_tdof_list_h1, V, b_h1, A_h1, X_h1, B_h1);
|
||||
a_nd.FormLinearSystem(ess_tdof_list_nd, A, b_nd, A_nd, X_nd, B_nd);
|
||||
|
||||
trueX.GetBlock(0) = X_h1;
|
||||
trueX.GetBlock(1) = X_nd;
|
||||
trueRHS.GetBlock(0) += B_h1;
|
||||
trueRHS.GetBlock(1) += B_nd;
|
||||
|
||||
// Form the off-diagonal entries
|
||||
a_h1_nd.FormRectangularLinearSystem(ess_tdof_list_h1, ess_tdof_list_nd, V, b_nd,
|
||||
A_h1_nd, X_h1, B_nd);
|
||||
a_nd_h1.FormRectangularLinearSystem(ess_tdof_list_nd, ess_tdof_list_h1, A, b_h1,
|
||||
A_nd_h1, X_nd, B_h1);
|
||||
|
||||
trueRHS.GetBlock(0) += B_h1;
|
||||
trueRHS.GetBlock(1) += B_nd;
|
||||
|
||||
h_blocks(0,0) = A_h1.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
h_blocks(1,1) = A_nd.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
h_blocks(0,1) = A_nd_h1.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
h_blocks(1,0) = A_h1_nd.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
|
||||
OperatorHandle op(HypreParMatrixFromBlocks(h_blocks));
|
||||
|
||||
SuperLURowLocMatrix S_op(*op);
|
||||
SuperLUSolver superlu(MPI_COMM_WORLD);
|
||||
superlu.SetPrintStatistics(false);
|
||||
superlu.SetSymmetricPattern(false);
|
||||
superlu.SetOperator(S_op);
|
||||
superlu.Mult(trueRHS, trueX);
|
||||
|
||||
trueX.GetBlock(0).SyncAliasMemory(trueX);
|
||||
trueX.GetBlock(1).SyncAliasMemory(trueX);
|
||||
|
||||
V.Distribute(trueX.GetBlock(0));
|
||||
A.Distribute(trueX.GetBlock(1));
|
||||
|
||||
// Check solution
|
||||
ConstantCoefficient zero(0.0);
|
||||
|
||||
real_t err_Vr = V.real().ComputeL2Error(V_exact_real);
|
||||
real_t err_Vi = V.imag().ComputeL2Error(zero);
|
||||
real_t err_Ar = A.real().ComputeL2Error(A_exact_real);
|
||||
real_t err_Ai = A.imag().ComputeL2Error(A_exact_imag);
|
||||
|
||||
REQUIRE(err_Vr == MFEM_Approx(0.0, 1e-5));
|
||||
REQUIRE(err_Vi == MFEM_Approx(0.0, 1e-5));
|
||||
REQUIRE(err_Ar == MFEM_Approx(0.0, 1e-5));
|
||||
REQUIRE(err_Ai == MFEM_Approx(0.0, 1e-5));
|
||||
}
|
||||
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
||||
@@ -750,4 +750,420 @@ TEST_CASE("Hcurl/Hdiv Mixed PA Coefficient",
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("3D Bilinear VectorFE Integrators PartialAssembly",
|
||||
"[BilinearFormIntegrator]"
|
||||
"[PartialAssembly]"
|
||||
"[GPU]")
|
||||
{
|
||||
auto order = GENERATE(1, 2);
|
||||
CAPTURE(order);
|
||||
dimension = 3;
|
||||
|
||||
FunctionCoefficient q3_coeff(coeffFunction);
|
||||
VectorFunctionCoefficient F3_coeff(dimension, vectorCoeffFunction);
|
||||
MatrixFunctionCoefficient M3_coeff(dimension, asymmetricMatrixCoeffFunction);
|
||||
|
||||
auto mesh_fname =
|
||||
GENERATE("../../data/fichera-amr.mesh", "../../data/fichera-q2.mesh");
|
||||
CAPTURE(mesh_fname);
|
||||
Mesh mesh(mesh_fname);
|
||||
REQUIRE(mesh.Dimension() == dimension);
|
||||
REQUIRE(mesh.SpaceDimension() == dimension);
|
||||
|
||||
SECTION("RT to RT Scalar Coeff")
|
||||
{
|
||||
RT_FECollection fec_rt(order - 1, dimension);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
|
||||
BilinearForm bfa(&fespace_rt);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
BilinearForm bpa(&fespace_rt);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_rt), y_fa(&fespace_rt), y_pa(&fespace_rt);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("RT to RT Diagonal Matrix Coeff")
|
||||
{
|
||||
RT_FECollection fec_rt(order - 1, dimension);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
|
||||
BilinearForm bfa(&fespace_rt);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
BilinearForm bpa(&fespace_rt);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_rt), y_fa(&fespace_rt), y_pa(&fespace_rt);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("RT to RT Matrix Coeff")
|
||||
{
|
||||
RT_FECollection fec_rt(order - 1, dimension);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
|
||||
BilinearForm bfa(&fespace_rt);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
BilinearForm bpa(&fespace_rt);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_rt), y_fa(&fespace_rt), y_pa(&fespace_rt);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("RT to ND Scalar Coeff")
|
||||
{
|
||||
RT_FECollection fec_rt(order - 1, dimension);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
ND_FECollection fec_nd(order, dimension);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("RT to ND Diagonal Matrix Coeff")
|
||||
{
|
||||
RT_FECollection fec_rt(order - 1, dimension);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
ND_FECollection fec_nd(order, dimension);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("RT to ND Matrix Coeff")
|
||||
{
|
||||
RT_FECollection fec_rt(order - 1, dimension);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
ND_FECollection fec_nd(order, dimension);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("ND to RT Scalar Coeff")
|
||||
{
|
||||
RT_FECollection fec_rt(order - 1, dimension);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
ND_FECollection fec_nd(order, dimension);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
MixedBilinearForm bfa(&fespace_nd, &fespace_rt);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
MixedBilinearForm bpa(&fespace_nd, &fespace_rt);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_nd), y_fa(&fespace_rt), y_pa(&fespace_rt);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("ND to RT Diagonal Matrix Coeff")
|
||||
{
|
||||
RT_FECollection fec_rt(order - 1, dimension);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
ND_FECollection fec_nd(order, dimension);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
MixedBilinearForm bfa(&fespace_nd, &fespace_rt);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
MixedBilinearForm bpa(&fespace_nd, &fespace_rt);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_nd), y_fa(&fespace_rt), y_pa(&fespace_rt);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("ND to RT Matrix Coeff")
|
||||
{
|
||||
RT_FECollection fec_rt(order - 1, dimension);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
ND_FECollection fec_nd(order, dimension);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
MixedBilinearForm bfa(&fespace_nd, &fespace_rt);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
MixedBilinearForm bpa(&fespace_nd, &fespace_rt);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_nd), y_fa(&fespace_rt), y_pa(&fespace_rt);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("ND to ND Scalar Coeff")
|
||||
{
|
||||
ND_FECollection fec_nd(order, dimension);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
BilinearForm bfa(&fespace_nd);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
BilinearForm bpa(&fespace_nd);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_nd), y_fa(&fespace_nd), y_pa(&fespace_nd);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("ND to ND Diagonal Matrix Coeff")
|
||||
{
|
||||
ND_FECollection fec_nd(order, dimension);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
BilinearForm bfa(&fespace_nd);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
BilinearForm bpa(&fespace_nd);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_nd), y_fa(&fespace_nd), y_pa(&fespace_nd);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("ND to ND Matrix Coeff")
|
||||
{
|
||||
ND_FECollection fec_nd(order, dimension);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
|
||||
BilinearForm bfa(&fespace_nd);
|
||||
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
BilinearForm bpa(&fespace_nd);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_nd), y_fa(&fespace_nd), y_pa(&fespace_nd);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("3D Bilinear Weak Curl Integrators Partial Assembly",
|
||||
"[MixedVectorWeakCurlIntegrator]"
|
||||
"[BilinearFormIntegrator]"
|
||||
"[PartialAssembly]"
|
||||
"[GPU]")
|
||||
{
|
||||
auto order = GENERATE(1, 2);
|
||||
CAPTURE(order);
|
||||
int dim = 3;
|
||||
|
||||
FunctionCoefficient q3_coeff(coeffFunction);
|
||||
VectorFunctionCoefficient F3_coeff(dim, vectorCoeffFunction);
|
||||
|
||||
auto mesh_fname =
|
||||
GENERATE("../../data/fichera-amr.mesh", "../../data/ball-nurbs.mesh");
|
||||
CAPTURE(mesh_fname);
|
||||
Mesh mesh(mesh_fname);
|
||||
REQUIRE(mesh.Dimension() == dim);
|
||||
REQUIRE(mesh.SpaceDimension() == dim);
|
||||
|
||||
// convert nurbs into piecewise-quadratic curved mesh
|
||||
if (mesh.NURBSext)
|
||||
{
|
||||
mesh.UniformRefinement();
|
||||
mesh.SetCurvature(2);
|
||||
}
|
||||
|
||||
SECTION("RT to ND No Coeff")
|
||||
{
|
||||
ND_FECollection fec_nd(order, dim);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
RT_FECollection fec_rt(order - 1, dim);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
|
||||
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
|
||||
bfa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator);
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator);
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
|
||||
x.Randomize(1234);
|
||||
REQUIRE(bfa.Height() == y_fa.Size());
|
||||
REQUIRE(bfa.Width() == x.Size());
|
||||
REQUIRE(bpa.Height() == y_fa.Size());
|
||||
REQUIRE(bpa.Width() == x.Size());
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("RT to ND Scalar Coeff")
|
||||
{
|
||||
ND_FECollection fec_nd(order, dim);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
RT_FECollection fec_rt(order - 1, dim);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
|
||||
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
|
||||
bfa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator(q3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator(q3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
|
||||
SECTION("RT to ND Diagonal Matrix Coeff")
|
||||
{
|
||||
ND_FECollection fec_nd(order, dim);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
RT_FECollection fec_rt(order - 1, dim);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
|
||||
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
|
||||
bfa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator(F3_coeff));
|
||||
bfa.Assemble();
|
||||
bfa.Finalize();
|
||||
|
||||
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
|
||||
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
bpa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator(F3_coeff));
|
||||
bpa.Assemble();
|
||||
|
||||
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
|
||||
x.Randomize(1234);
|
||||
bfa.Mult(x, y_fa);
|
||||
bpa.Mult(x, y_pa);
|
||||
y_pa -= y_fa;
|
||||
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace pa_coeff
|
||||
|
||||
@@ -17,12 +17,63 @@ using namespace mfem;
|
||||
namespace project_bdr
|
||||
{
|
||||
|
||||
void Func_3D_lin(const Vector &x, Vector &v)
|
||||
TEST_CASE("3D ProjectBdrCoefficient",
|
||||
"[GridFunction]"
|
||||
"[NCMesh]")
|
||||
{
|
||||
v.SetSize(3);
|
||||
v[0] = 1.234 * x[0] - 2.357 * x[1] + 3.572 * x[2];
|
||||
v[1] = 2.537 * x[0] + 4.321 * x[1] - 1.234 * x[2];
|
||||
v[2] = -2.572 * x[0] + 1.321 * x[1] + 3.234 * x[2];
|
||||
const char *mesh_file = GENERATE("data/hex-nc-cross.mesh",
|
||||
"data/tet-nc-cross.mesh");
|
||||
CAPTURE(mesh_file);
|
||||
constexpr int order = 3;
|
||||
constexpr real_t freq = 5.0;
|
||||
|
||||
// Attributes
|
||||
Array<int> bdr_attr(2);
|
||||
bdr_attr = 0;
|
||||
bdr_attr[1] = 1;
|
||||
|
||||
// Coefficient
|
||||
FunctionCoefficient coeff([&](const Vector &x)
|
||||
{
|
||||
return cos(freq * M_PI * x[0])
|
||||
* cos(freq * M_PI * x[1])
|
||||
* cos(freq * M_PI * x[2]);
|
||||
});
|
||||
|
||||
// Vertex-based mesh
|
||||
Mesh mesh_v(mesh_file, 1, 1);
|
||||
H1_FECollection fec_v(order, mesh_v.Dimension());
|
||||
FiniteElementSpace fes_v(&mesh_v, &fec_v);
|
||||
GridFunction gf_v(&fes_v);
|
||||
gf_v = 0.0;
|
||||
gf_v.ProjectBdrCoefficient(coeff, bdr_attr);
|
||||
|
||||
// Nodal mesh
|
||||
Mesh mesh_n(mesh_file, 1, 1);
|
||||
mesh_n.SetCurvature(order, true);
|
||||
H1_FECollection fec_n(order, mesh_n.Dimension());
|
||||
FiniteElementSpace fes_n(&mesh_n, &fec_n);
|
||||
GridFunction gf_n(&fes_n);
|
||||
gf_n = 0.0;
|
||||
gf_n.ProjectBdrCoefficient(coeff, bdr_attr);
|
||||
|
||||
gf_n -= gf_v;
|
||||
|
||||
REQUIRE(gf_n.Norml2() == MFEM_Approx(0.0));
|
||||
}
|
||||
|
||||
void Func_lin(const Vector &x, Vector &v)
|
||||
{
|
||||
const int dim = x.Size();
|
||||
v.SetSize(dim);
|
||||
v[0] = 1.234 * x[0] - 2.357 * x[1];
|
||||
v[1] = 2.537 * x[0] + 4.321 * x[1];
|
||||
if (dim == 3)
|
||||
{
|
||||
v[0] += 3.572 * x[2];
|
||||
v[1] -= 1.234 * x[2];
|
||||
v[2] = -2.572 * x[0] + 1.321 * x[1] + 3.234 * x[2];
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("3D ProjectBdrCoefficientNormal Vector",
|
||||
@@ -41,7 +92,7 @@ TEST_CASE("3D ProjectBdrCoefficientNormal Vector",
|
||||
Mesh mesh = Mesh::MakeCartesian3D(
|
||||
n, n, n, (Element::Type)type, 2.0, 3.0, 5.0);
|
||||
|
||||
VectorFunctionCoefficient funcCoef(dim, Func_3D_lin);
|
||||
VectorFunctionCoefficient funcCoef(dim, Func_lin);
|
||||
|
||||
SECTION("3D GetVectorValue tests for element type " +
|
||||
std::to_string(type))
|
||||
@@ -133,7 +184,7 @@ TEST_CASE("3D ProjectBdrCoefficientNormal Scalar",
|
||||
Mesh mesh = Mesh::MakeCartesian3D(
|
||||
n, n, n, (Element::Type)type, 2.0, 3.0, 5.0);
|
||||
|
||||
VectorFunctionCoefficient funcCoef(dim, Func_3D_lin);
|
||||
VectorFunctionCoefficient funcCoef(dim, Func_lin);
|
||||
|
||||
SECTION("3D GetVectorValue tests for element type " +
|
||||
std::to_string(type))
|
||||
@@ -227,7 +278,7 @@ TEST_CASE("3D ProjectBdrCoefficientTangent",
|
||||
Mesh mesh = Mesh::MakeCartesian3D(
|
||||
n, n, n, (Element::Type)type, 2.0, 3.0, 5.0);
|
||||
|
||||
VectorFunctionCoefficient funcCoef(dim, Func_3D_lin);
|
||||
VectorFunctionCoefficient funcCoef(dim, Func_lin);
|
||||
|
||||
SECTION("3D GetVectorValue tests for element type " +
|
||||
std::to_string(type))
|
||||
@@ -305,4 +356,50 @@ TEST_CASE("3D ProjectBdrCoefficientTangent",
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("ProjectBdrCoefficientTangent with IntegratedGLL",
|
||||
"[GridFunction]"
|
||||
"[VectorGridFunctionCoefficient]")
|
||||
{
|
||||
const int dim = GENERATE(2, 3);
|
||||
CAPTURE(dim);
|
||||
Mesh mesh = (dim == 2) ?
|
||||
Mesh::MakeCartesian2D(1, 2, Element::QUADRILATERAL,
|
||||
true, 2.0, 5.0) :
|
||||
Mesh::MakeCartesian3D(1, 1, 2, Element::HEXAHEDRON,
|
||||
2.0, 3.0, 5.0);
|
||||
mesh.EnsureNodes();
|
||||
mesh.EnsureNCMesh(false);
|
||||
VectorFunctionCoefficient func_coef(dim, Func_lin);
|
||||
Array<int> all_bdr(mesh.bdr_attributes.Max());
|
||||
all_bdr = 1;
|
||||
|
||||
for (int order = 1; order <= 4; order++)
|
||||
{
|
||||
CAPTURE(order);
|
||||
ND_FECollection nd_fec(order, dim, BasisType::GaussLobatto,
|
||||
BasisType::IntegratedGLL);
|
||||
FiniteElementSpace nd_fespace(&mesh, &nd_fec);
|
||||
GridFunction volume_projection(&nd_fespace);
|
||||
GridFunction boundary_projection(&nd_fespace);
|
||||
|
||||
volume_projection.ProjectCoefficient(func_coef);
|
||||
boundary_projection = 0.0;
|
||||
boundary_projection.ProjectBdrCoefficientTangent(func_coef, all_bdr);
|
||||
|
||||
Array<int> ess_vdofs;
|
||||
nd_fespace.GetEssentialVDofs(all_bdr, ess_vdofs);
|
||||
real_t max_error = 0.0;
|
||||
for (int i = 0; i < ess_vdofs.Size(); i++)
|
||||
{
|
||||
if (ess_vdofs[i])
|
||||
{
|
||||
max_error = std::max(max_error, std::abs(
|
||||
boundary_projection[i] -
|
||||
volume_projection[i]));
|
||||
}
|
||||
}
|
||||
REQUIRE(max_error == MFEM_Approx(0.0));
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace project_bdr
|
||||
|
||||
@@ -164,12 +164,29 @@ TEST_CASE("ComplexHypreParMatrix GetSystemMatrix",
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(one),
|
||||
new VectorFEMassIntegrator(one));
|
||||
a.Assemble();
|
||||
|
||||
// 2. Test ParSesquilinearForm::FormSystemMatrix directly and verify that
|
||||
// essential entries on the imaginary diagonal are zero.
|
||||
OperatorPtr Ah;
|
||||
a.FormSystemMatrix(ess_tdof_list, Ah);
|
||||
ComplexHypreParMatrix *A_complex = Ah.Is<ComplexHypreParMatrix>();
|
||||
REQUIRE(A_complex != nullptr);
|
||||
Vector diag;
|
||||
A_complex->imag().GetDiag(diag);
|
||||
const Array<int> &ess_tdofs = ess_tdof_list;
|
||||
const Vector &diag_h = diag;
|
||||
ess_tdofs.HostRead();
|
||||
diag_h.HostRead();
|
||||
for (const int tdof : ess_tdofs)
|
||||
{
|
||||
REQUIRE(diag_h[tdof] == 0.0);
|
||||
}
|
||||
|
||||
// 3. Test the call to ComplexHypreParMatrix::GetSystemMatrix and destroying
|
||||
// the returned matrix.
|
||||
Vector B, X;
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
|
||||
|
||||
// 2. Test the call to ComplexHypreParMatrix::GetSystemMatrix and destroying
|
||||
// the returned matrix.
|
||||
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
delete A;
|
||||
}
|
||||
|
||||
@@ -494,6 +494,64 @@ TEST_CASE("Batched Linear Algebra",
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_EXCEPTIONS
|
||||
namespace
|
||||
{
|
||||
|
||||
DenseTensor MakeSingularBatchedMatrices()
|
||||
{
|
||||
const int n = 3;
|
||||
const int n_mat = 3;
|
||||
|
||||
DenseTensor A_batch(n, n, n_mat);
|
||||
for (int i = 0; i < n_mat; ++i)
|
||||
{
|
||||
DenseMatrix &A = A_batch(i);
|
||||
A = 0.0;
|
||||
for (int j = 0; j < n; ++j)
|
||||
{
|
||||
A(j, j) = 2.0 + i + j;
|
||||
}
|
||||
}
|
||||
|
||||
DenseMatrix &singular = A_batch(1);
|
||||
singular = 0.0;
|
||||
singular(0, 0) = 1.0;
|
||||
singular(2, 2) = 1.0;
|
||||
|
||||
return A_batch;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
TEST_CASE("Batched LU factorization failure handling",
|
||||
"[DenseMatrix][GPU]")
|
||||
{
|
||||
auto backend = GENERATE(BatchedLinAlg::NATIVE,
|
||||
BatchedLinAlg::GPU_BLAS,
|
||||
BatchedLinAlg::MAGMA);
|
||||
if (!BatchedLinAlg::IsAvailable(backend)) { return; }
|
||||
CAPTURE(backend);
|
||||
|
||||
SECTION("LUFactor")
|
||||
{
|
||||
DenseTensor A_batch = MakeSingularBatchedMatrices();
|
||||
Array<int> P;
|
||||
REQUIRE_THROWS_WITH(BatchedLinAlg::Get(backend).LUFactor(A_batch, P),
|
||||
Catch::Matchers::Contains(
|
||||
"Batch LU factorization failed"));
|
||||
}
|
||||
|
||||
SECTION("Invert")
|
||||
{
|
||||
DenseTensor A_batch = MakeSingularBatchedMatrices();
|
||||
REQUIRE_THROWS_WITH(BatchedLinAlg::Get(backend).Invert(A_batch),
|
||||
Catch::Matchers::Contains(
|
||||
"Batch LU factorization failed"));
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
TEST_CASE("DenseTensor copy", "[DenseMatrix][DenseTensor]")
|
||||
{
|
||||
DenseTensor t1(2,3,4);
|
||||
|
||||
@@ -152,7 +152,7 @@ TEST_CASE("GlobalBBoxTensorGridMap Parallel",
|
||||
std::map<int, std::vector<int>> pt_to_procs;
|
||||
map.MapPointsToProcs(centers, 1, pt_to_procs);
|
||||
|
||||
REQUIRE(pt_to_procs.size() == nel + 1);
|
||||
REQUIRE(pt_to_procs.size() == (unsigned)nel + 1);
|
||||
for (int i = 0; i < nel; i++)
|
||||
{
|
||||
std::vector<int> procs = pt_to_procs[i];
|
||||
|
||||
@@ -304,6 +304,66 @@ TEST_CASE("pNCMesh PA diagonal", "[Parallel], [NCMesh]")
|
||||
}
|
||||
} // test case
|
||||
|
||||
TEST_CASE("ParNCMesh Rebalance preserves element attributes",
|
||||
"[Parallel], [NCMesh]")
|
||||
{
|
||||
const int rank = Mpi::WorldRank();
|
||||
const int nranks = Mpi::WorldSize();
|
||||
if (nranks < 2) { return; }
|
||||
|
||||
auto mesh_fname = GENERATE("../../data/star.mesh",
|
||||
"../../data/fichera.mesh");
|
||||
CAPTURE(mesh_fname);
|
||||
|
||||
auto CheckRebalance = [rank, nranks, mesh_fname](bool refine,
|
||||
bool custom_partition)
|
||||
{
|
||||
Mesh mesh(mesh_fname);
|
||||
mesh.EnsureNCMesh();
|
||||
ParMesh pmesh(MPI_COMM_WORLD, mesh);
|
||||
|
||||
const int attribute = 1234 + (custom_partition ? rank : 0);
|
||||
for (int i = 0; i < pmesh.GetNE(); i++)
|
||||
{
|
||||
pmesh.SetAttribute(i, attribute);
|
||||
}
|
||||
pmesh.SetAttributes();
|
||||
|
||||
if (refine)
|
||||
{
|
||||
Array<int> refinements;
|
||||
if (pmesh.GetNE() && (custom_partition || rank == 0))
|
||||
{
|
||||
refinements.Append(0);
|
||||
}
|
||||
pmesh.GeneralRefinement(refinements);
|
||||
}
|
||||
|
||||
int expected_attribute = attribute;
|
||||
if (custom_partition)
|
||||
{
|
||||
// Move every element to the next rank, as in GitHub issue #4009.
|
||||
Array<int> partition(pmesh.GetNE());
|
||||
partition = (rank + 1) % nranks;
|
||||
pmesh.Rebalance(partition);
|
||||
expected_attribute = 1234 + (rank + nranks - 1) % nranks;
|
||||
}
|
||||
else
|
||||
{
|
||||
pmesh.Rebalance();
|
||||
}
|
||||
|
||||
for (int i = 0; i < pmesh.GetNE(); i++)
|
||||
{
|
||||
CHECK(pmesh.GetAttribute(i) == expected_attribute);
|
||||
}
|
||||
};
|
||||
|
||||
SECTION("Custom partition, unrefined") { CheckRebalance(false, true); }
|
||||
SECTION("Custom partition, refined") { CheckRebalance(true, true); }
|
||||
SECTION("Default partition, refined") { CheckRebalance(true, false); }
|
||||
}
|
||||
|
||||
TEST_CASE("EdgeFaceConstraint", "[Parallel], [NCMesh]")
|
||||
{
|
||||
auto exact_soln = [](const Vector& x)
|
||||
|
||||
Reference in New Issue
Block a user