Compare commits
196
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c01a9cb623 | ||
|
|
007183c70a | ||
|
|
0e63fc90e3 | ||
|
|
06cb917637 | ||
|
|
1861b80627 | ||
|
|
a8c0c856f6 | ||
|
|
dbb86c87c2 | ||
|
|
e866ede0c4 | ||
|
|
db97637c09 | ||
|
|
cbaf930388 | ||
|
|
c498d9c759 | ||
|
|
2896b2fb02 | ||
|
|
8efb9483bc | ||
|
|
5507772f37 | ||
|
|
1b20704b24 | ||
|
|
55bafba69a | ||
|
|
3a29fda4cd | ||
|
|
c53bce08f1 | ||
|
|
afadd435f5 | ||
|
|
36d7629b26 | ||
|
|
8f32e6620b | ||
|
|
766760659d | ||
|
|
2c0419a9a8 | ||
|
|
9185762478 | ||
|
|
2c41e483fe | ||
|
|
78a4a77ed7 | ||
|
|
ea4aceeffc | ||
|
|
1f8af4538c | ||
|
|
ca1bbaa7ed | ||
|
|
2ce98bec12 | ||
|
|
3db24b1b40 | ||
|
|
b53dd0fea1 | ||
|
|
0e61a94b5f | ||
|
|
194f2a56b7 | ||
|
|
efe05b9b1a | ||
|
|
acf167ed86 | ||
|
|
8a8ac07910 | ||
|
|
91600c12eb | ||
|
|
35aef0486f | ||
|
|
df6d58da30 | ||
|
|
500e952d5c | ||
|
|
1c56fe47c4 | ||
|
|
bbd33cfcc1 | ||
|
|
5083a29ccd | ||
|
|
6e063a5d23 | ||
|
|
87b11c227b | ||
|
|
db10fd292a | ||
|
|
072147289b | ||
|
|
cde2b05366 | ||
|
|
e8d1fc9b60 | ||
|
|
60771f2f27 | ||
|
|
dbe2c6862c | ||
|
|
35442a2004 | ||
|
|
1a4c7eb027 | ||
|
|
60a9893d52 | ||
|
|
8c2ffb9d26 | ||
|
|
08f41f6450 | ||
|
|
d184921e09 | ||
|
|
57e26f75b0 | ||
|
|
2eaf46c80d | ||
|
|
48a2648ec5 | ||
|
|
bfffb837d3 | ||
|
|
2737feaa2a | ||
|
|
988cc5b18d | ||
|
|
ae49f4be68 | ||
|
|
c51d05f1e1 | ||
|
|
c5ef67adcf | ||
|
|
c5f78ea58a | ||
|
|
d066b11e18 | ||
|
|
cb85a7b804 | ||
|
|
3dcba10659 | ||
|
|
4c6be0bc6f | ||
|
|
4d1cd791f3 | ||
|
|
cb2d4bde47 | ||
|
|
ca43ab0c61 | ||
|
|
1c60d5946b | ||
|
|
c9768e34bc | ||
|
|
2352f7be6b | ||
|
|
244db1b571 | ||
|
|
75795809ef | ||
|
|
825a5a7f2e | ||
|
|
4ed8f16325 | ||
|
|
ec896295d6 | ||
|
|
c2d5eed541 | ||
|
|
f8e71cf89c | ||
|
|
2375135f9a | ||
|
|
a4800f42dd | ||
|
|
1e48c7e6d0 | ||
|
|
88456472dc | ||
|
|
7040fe65da | ||
|
|
6df83bd190 | ||
|
|
10d975d0e7 | ||
|
|
bae772c6f1 | ||
|
|
79e0bc1ab2 | ||
|
|
847cf4a646 | ||
|
|
bdb3f39ffa | ||
|
|
803e2bd5c9 | ||
|
|
2f7ba402dd | ||
|
|
26374d9be8 | ||
|
|
7ae7690846 | ||
|
|
c21c6cb00b | ||
|
|
1bd9dd2e5d | ||
|
|
bc0ec2e717 | ||
|
|
31409edb7f | ||
|
|
ed66371ebd | ||
|
|
5727331966 | ||
|
|
40039a6897 | ||
|
|
a05d4e1852 | ||
|
|
29c2442e47 | ||
|
|
fcb78b81ec | ||
|
|
92ab53dec6 | ||
|
|
c2475e43fd | ||
|
|
232853214d | ||
|
|
a024fb10bc | ||
|
|
531e8d7ad9 | ||
|
|
5293b9694d | ||
|
|
9008d5a050 | ||
|
|
c18f3ba6f9 | ||
|
|
d48af86cdf | ||
|
|
bd4f07f6cb | ||
|
|
6f72e7f752 | ||
|
|
ddde1ff8d4 | ||
|
|
7c09989768 | ||
|
|
7794557c18 | ||
|
|
495cb138ee | ||
|
|
b177b2f0dc | ||
|
|
b494d821b1 | ||
|
|
f63b033c72 | ||
|
|
4d782b8fad | ||
|
|
fc1bd60e49 | ||
|
|
112a9871ee | ||
|
|
1c1ffa875e | ||
|
|
3f50a6f4ce | ||
|
|
1e61c5e366 | ||
|
|
ee0821d62f | ||
|
|
9cd037dfdd | ||
|
|
aaa828472f | ||
|
|
15443a32a1 | ||
|
|
9360abf011 | ||
|
|
7475a13e6a | ||
|
|
9d2df07b71 | ||
|
|
7f80725ddc | ||
|
|
bc9ba8c8da | ||
|
|
7718b37ecf | ||
|
|
10efeb79d1 | ||
|
|
bdd9db4892 | ||
|
|
c0f61c5cb4 | ||
|
|
85a4d88e2a | ||
|
|
fd7efca993 | ||
|
|
742db7c701 | ||
|
|
9a47796ea3 | ||
|
|
f724cf348a | ||
|
|
54cb56988b | ||
|
|
09dddd6f11 | ||
|
|
91590f39c3 | ||
|
|
161278cd30 | ||
|
|
fe3251bf02 | ||
|
|
da4f94e9ef | ||
|
|
628818b2f1 | ||
|
|
c9cf8d080d | ||
|
|
1e7f897efb | ||
|
|
46de4f5911 | ||
|
|
e51ea52ca4 | ||
|
|
003afb8a4c | ||
|
|
2a013af660 | ||
|
|
2bf7cff7b4 | ||
|
|
fa41baa1c8 | ||
|
|
6558294943 | ||
|
|
85b8bfb57d | ||
|
|
7714f8f42c | ||
|
|
1c1f622b7d | ||
|
|
16f1935531 | ||
|
|
a402b4e9b2 | ||
|
|
d269b25c17 | ||
|
|
7573b7c9fa | ||
|
|
56dfb0e67f | ||
|
|
3474158bf6 | ||
|
|
93a2f318e0 | ||
|
|
4a5ae97be5 | ||
|
|
186105c664 | ||
|
|
8df6973bc7 | ||
|
|
2cc76c588b | ||
|
|
8096c493e8 | ||
|
|
c0fe8c599b | ||
|
|
7cda1566e2 | ||
|
|
e83191f54e | ||
|
|
985dfe2749 | ||
|
|
e275737aa9 | ||
|
|
91b4825cca | ||
|
|
4c0a122240 | ||
|
|
db4060cf77 | ||
|
|
34addd59c5 | ||
|
|
8caddf1738 | ||
|
|
d9ca28ba43 | ||
|
|
a331a1951b | ||
|
|
336f1f8f35 |
@@ -369,6 +369,7 @@ miniapps/shifted/lsf_integral
|
||||
miniapps/tools/display-basis
|
||||
miniapps/tools/load-dc
|
||||
miniapps/tools/convert-dc
|
||||
miniapps/tools/compare-dc
|
||||
miniapps/tools/gridfunction-bounds
|
||||
miniapps/tools/lor-transfer
|
||||
miniapps/tools/plor-transfer
|
||||
|
||||
@@ -22,6 +22,11 @@ Meshing improvements
|
||||
- Improved support for 1D NURBS meshes with variable order, including using
|
||||
the patches construct for 1D NURBS meshes.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Electromagnetics/lorentz miniapp has been updated to leverage the ParticleSet
|
||||
capability.
|
||||
|
||||
|
||||
Version 4.9, released on Dec 11, 2025
|
||||
=====================================
|
||||
@@ -106,6 +111,23 @@ Linear and nonlinear solvers
|
||||
Filtering (AMGF), providing robust preconditioning for linear systems arising
|
||||
in constrained optimization problems such as frictionless contact.
|
||||
|
||||
Added 'GetResiduals' and 'GetFinalAbsResidualNorm' to 'HyprePCG',
|
||||
'HypreGMRES', and 'HypreFGMRES' to get 'r' and '|r|_p'. Note that the latter
|
||||
computes '|r|_p' from 'r' instead of returning a cached value like the
|
||||
relative 'GetFinalResidualNorm'. These require Hypre >= 2.15.0.
|
||||
|
||||
Changed the default solver parameters for 'HyprePCG' to 'tol=1e-6' and
|
||||
'max_iter=1000'. This matches the default parameters in Hypre 3.0.
|
||||
|
||||
Added various helper functions for querying/modifying Hypre solvers:
|
||||
'HypreSmoother::GetType', 'HypreSmoother::GetSOROptions',
|
||||
'HypreSmoother::GetPolyOptions', 'HypreSmoother::GetWindowParameters',
|
||||
'HypreSmoother::IsOperatorSymmetric', 'HyprePCG::GetTol',
|
||||
'HyprePCG::GetAbsTol', 'HyprePCG::GetMaxIter', 'HyprePCG::SetUseTwoNorm',
|
||||
'HypreGMRES::GetTol', 'HypreGMRES::GetAbsTol', 'HypreGMRES::GetMaxIter',
|
||||
'HypreGMRES::GetKDim', 'HypreFGMRES::GetTol', 'HypreFGMRES::GetMaxIter',
|
||||
'HypreFGMRES::GetKDim', and 'HypreBoomerAMG::GetMaxIter'.
|
||||
|
||||
GPU computing
|
||||
-------------
|
||||
- Added the 'gpu', 'raja-gpu', and 'ceed-gpu' backend aliases/shortcuts which
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
# Some choices below are based on the OS type:
|
||||
NOTMAC := $(subst Darwin,,$(shell uname -s))
|
||||
|
||||
ASTYLE_BIN = astyle
|
||||
ETAGS_BIN = $(shell command -v etags 2> /dev/null)
|
||||
EGREP_BIN = $(shell command -v egrep 2> /dev/null)
|
||||
|
||||
|
||||
@@ -119,8 +119,6 @@ namespace mfem {
|
||||
* - <a class="el" href="ex40p_8cpp_source.html">Example 40p</a>: parallel eikonal equation
|
||||
* - <a class="el" href="ex41_8cpp_source.html">Example 41</a>: DG/CG IMEX time-dependent advection-diffusion
|
||||
* - <a class="el" href="ex41p_8cpp_source.html">Example 41p</a>: parallel DG/CG IMEX time-dependent advection-diffusion
|
||||
* - <a class="el" href="ex43_8cpp_source.html">Example 43</a>: sliding boundary conditions in linear elasticity
|
||||
* - <a class="el" href="ex43p_8cpp_source.html">Example 43p</a>: parallel sliding boundary conditions in linear elasticity
|
||||
*
|
||||
* <H4>AmgX Examples</H4>
|
||||
* - Variants of Examples
|
||||
|
||||
@@ -47,7 +47,6 @@ list(APPEND ALL_EXE_SRCS
|
||||
ex39.cpp
|
||||
ex40.cpp
|
||||
ex41.cpp
|
||||
ex43.cpp
|
||||
)
|
||||
|
||||
if (MFEM_USE_MPI)
|
||||
@@ -92,7 +91,6 @@ if (MFEM_USE_MPI)
|
||||
ex39p.cpp
|
||||
ex40p.cpp
|
||||
ex41p.cpp
|
||||
ex43p.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
|
||||
+5
-1
@@ -9,6 +9,7 @@
|
||||
// ex4 -m ../data/beam-hex.mesh -o 2 -pa
|
||||
// ex4 -m ../data/escher.mesh
|
||||
// ex4 -m ../data/fichera.mesh -o 2 -hb
|
||||
// ex4 -m ../data/fichera.mesh -o 2 -hb -ea
|
||||
// ex4 -m ../data/fichera-q2.vtk
|
||||
// ex4 -m ../data/fichera-q3.mesh -o 2 -sc
|
||||
// ex4 -m ../data/square-disc-nurbs.mesh
|
||||
@@ -18,6 +19,7 @@
|
||||
// ex4 -m ../data/amr-quad.mesh
|
||||
// ex4 -m ../data/amr-hex.mesh
|
||||
// ex4 -m ../data/amr-hex.mesh -o 2 -hb
|
||||
// ex4 -m ../data/amr-hex.mesh -o 2 -hb -ea
|
||||
// ex4 -m ../data/fichera-amr.mesh -o 2 -sc
|
||||
// ex4 -m ../data/ref-prism.mesh -o 1
|
||||
// ex4 -m ../data/octahedron.mesh -o 1
|
||||
@@ -25,6 +27,8 @@
|
||||
//
|
||||
// Device sample runs:
|
||||
// ex4 -m ../data/star.mesh -pa -d cuda
|
||||
// ex4 -m ../data/star.mesh -hb -ea -d cuda
|
||||
// ex4 -m ../data/amr-quad.mesh -hb -ea -d cuda
|
||||
// ex4 -m ../data/star.mesh -pa -d raja-cuda
|
||||
// ex4 -m ../data/star.mesh -pa -d raja-omp
|
||||
// ex4 -m ../data/beam-hex.mesh -pa -d cuda
|
||||
@@ -193,7 +197,7 @@ int main(int argc, char *argv[])
|
||||
cout << "Size of linear system: " << A->Height() << endl;
|
||||
|
||||
// 11. Solve the linear system A X = B.
|
||||
if (!pa)
|
||||
if (!pa && (!ea || hybridization))
|
||||
{
|
||||
#ifndef MFEM_USE_SUITESPARSE
|
||||
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
|
||||
|
||||
@@ -1,278 +0,0 @@
|
||||
// MFEM Example 43
|
||||
//
|
||||
// Compile with: make ex43
|
||||
//
|
||||
// Sample runs: ex43 -m ../data/ball-nurbs.mesh -r 2
|
||||
// ex43 -m ../data/ref-cube.mesh -r 2
|
||||
// ex43 -m ../data/fichera.mesh
|
||||
// ex43 -m ../data/star.mesh
|
||||
//
|
||||
// Description: This example code solves a linear elasticity problem using
|
||||
// Nitsche's method to enforce sliding boundary conditions. In
|
||||
// particular, we consider a linear elastic body that is displaced
|
||||
// in the normal direction on the entire boundary, but is free to
|
||||
// slide in the tangential direction. This is achieved by imposing
|
||||
// homogeneous Dirichlet boundary conditions on the normal
|
||||
// component of the displacement, while applying homogeneous
|
||||
// Neumann boundary conditions on the tangential components of the
|
||||
// displacement. By enforcing a uniform, constant normal
|
||||
// displacement on the boundary, we can simulate the effect of
|
||||
// compressing or expanding the elastic body uniformly. These
|
||||
// boundary conditions are applied weakly using Nitsche's method,
|
||||
// allowing for more flexibility in handling complex geometries in
|
||||
// either 2D or 3D.
|
||||
//
|
||||
// The strong form is given by:
|
||||
//
|
||||
// −Div(σ(u)) = 0 in Ω
|
||||
// u ⋅ n = g on Γ
|
||||
// σ(u) ⊥ n on Γ
|
||||
//
|
||||
// where σ(u) = λ tr(ε(u)) I + 2μ ε(u) is the stress tensor, ε(u)
|
||||
// is the strain tensor, λ and μ are the Lamé parameters, and g is
|
||||
// the prescribed displacement on the boundary. Here, n is the
|
||||
// outward normal on the boundary Γ = ∂Ω.
|
||||
//
|
||||
// The weak form using Nitsche's method is:
|
||||
//
|
||||
// Find u ∈ V such that a(u,v) = b(v) for all v ∈ V
|
||||
//
|
||||
// where
|
||||
//
|
||||
// a(u,v) := ∫_Ω σ(u) : ε(v) dx
|
||||
// - ∫_Γ (σ(u) n ⋅ n) (v ⋅ n) dS
|
||||
// - ∫_Γ (σ(v) n ⋅ n) (u ⋅ n) dS
|
||||
// + κ ∫_Γ h⁻¹ (λ + 2μ) (u ⋅ n) (v ⋅ n) dS,
|
||||
//
|
||||
// b(v) := - ∫_Γ σ(v) n ⋅ n g dS
|
||||
// + κ ∫_Γ h⁻¹ (λ + 2μ) (v ⋅ n) g dS,
|
||||
//
|
||||
// with κ > 0 being a penalty parameter. Here, h is a
|
||||
// characteristic element size on the boundary. The function
|
||||
// space V is a vector H1-conforming finite element space.
|
||||
//
|
||||
// This example can be viewed as an alternative to Example 28.
|
||||
// Whereas Example 28 imposes sliding boundary conditions using
|
||||
// the general-purpose constrained system solvers found in
|
||||
// mfem/linalg/constraints.hpp, this example employs Nitsche's
|
||||
// method to weakly enforce the same condition by modifying the
|
||||
// underlying variational formulation. Unlike Example 28, the
|
||||
// approach here is specialized to isotropic linear elasticity,
|
||||
// but it has the advantage of producing a well-conditioned SPD
|
||||
// stiffness matrix that can be readily preconditioned with
|
||||
// standard AMG. We recommend reviewing Example 2 before working
|
||||
// through this example.
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
// 1. Parse command-line options.
|
||||
const char *mesh_file = "../data/star.mesh";
|
||||
real_t displ_mag = 0.1;
|
||||
int order = 1;
|
||||
int ref_levels = 0;
|
||||
real_t lambda = 1.0;
|
||||
real_t mu = 1.0;
|
||||
real_t kappa = -1.0;
|
||||
bool static_cond = false;
|
||||
bool visualization = 1;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
"Mesh file to use.");
|
||||
args.AddOption(&displ_mag, "-g", "--displ",
|
||||
"Magnitude of the normal displacement.");
|
||||
args.AddOption(&order, "-o", "--order",
|
||||
"Finite element order (polynomial degree).");
|
||||
args.AddOption(&ref_levels, "-r", "--ref_levels",
|
||||
"Number of uniform mesh refinements.");
|
||||
args.AddOption(&lambda, "-l", "--lambda", "First Lamé parameter.");
|
||||
args.AddOption(&mu, "-mu", "--mu", "Second Lamé parameter.");
|
||||
args.AddOption(&kappa, "-k", "--kappa",
|
||||
"The penalty parameter, should be positive."
|
||||
" Negative values are replaced with (order+1)^2.");
|
||||
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
|
||||
"--no-static-condensation", "Enable static condensation.");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
return 1;
|
||||
}
|
||||
if (kappa < 0)
|
||||
{
|
||||
kappa = (order+1)*(order+1);
|
||||
}
|
||||
args.PrintOptions(cout);
|
||||
|
||||
// 2. Read the mesh from the given mesh file. We can handle triangular,
|
||||
// quadrilateral, tetrahedral or hexahedral elements with the same code.
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
|
||||
// 3. Select the order of the finite element discretization space. For NURBS
|
||||
// meshes, we increase the order by degree elevation.
|
||||
if (mesh->NURBSext)
|
||||
{
|
||||
mesh->DegreeElevate(order, order);
|
||||
}
|
||||
|
||||
// 4. Refine the mesh to increase the resolution. In this example we do
|
||||
// 'ref_levels' of uniform refinement.
|
||||
for (int i = 0; i < ref_levels; i++)
|
||||
{
|
||||
mesh->UniformRefinement();
|
||||
}
|
||||
|
||||
// 5. Interpolate the geometry after refinement to control geometry error.
|
||||
int curvature_order = max(order, 2);
|
||||
mesh->SetCurvature(curvature_order);
|
||||
|
||||
// 6. Define a finite element space on the mesh. Here we use vector finite
|
||||
// elements, i.e. dim copies of a scalar finite element space. The vector
|
||||
// dimension is specified by the last argument of the FiniteElementSpace
|
||||
// constructor. For NURBS meshes, we use the (degree elevated) NURBS space
|
||||
// associated with the mesh nodes.
|
||||
FiniteElementCollection *fec;
|
||||
FiniteElementSpace *fespace;
|
||||
if (mesh->NURBSext)
|
||||
{
|
||||
fec = NULL;
|
||||
fespace = mesh->GetNodes()->FESpace();
|
||||
}
|
||||
else
|
||||
{
|
||||
fec = new H1_FECollection(order, dim);
|
||||
fespace = new FiniteElementSpace(mesh, fec, dim);
|
||||
}
|
||||
cout << "Number of finite element unknowns: " << fespace->GetTrueVSize()
|
||||
<< endl << "Assembling: " << flush;
|
||||
|
||||
// 7. Mark the boundary attributes where the sliding (Nitsche) boundary
|
||||
// conditions are to be applied. These b.c. are imposed weakly, by adding
|
||||
// the appropriate boundary integrators over the marked 'ess_bdr' to the
|
||||
// bilinear and linear forms. Thus, no dofs are eliminated; there are no
|
||||
// essential boundary conditions.
|
||||
Array<int> ess_tdof_list, ess_bdr(mesh->bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
|
||||
// 8. Define the solution vector x as a finite element grid function
|
||||
// corresponding to fespace. Initialize x with initial guess of zero,
|
||||
// which satisfies the boundary conditions.
|
||||
GridFunction x(fespace);
|
||||
x = 0.0;
|
||||
|
||||
// 9. Set up the bilinear form a(.,.) on the finite element space
|
||||
// corresponding to the linear elasticity integrator with constant
|
||||
// coefficients lambda and mu.
|
||||
ConstantCoefficient lambda_c(lambda);
|
||||
ConstantCoefficient mu_c(mu);
|
||||
|
||||
BilinearForm *a = new BilinearForm(fespace);
|
||||
a->AddDomainIntegrator(new ElasticityIntegrator(lambda_c,mu_c));
|
||||
a->AddBdrFaceIntegrator(
|
||||
new SlidingElasticityIntegrator(lambda_c, mu_c, kappa),
|
||||
ess_bdr);
|
||||
|
||||
// 10. Set up the linear form b(.) corresponding to the Nitsche method
|
||||
// to impose the Dirichlet boundary conditions. Here, we set the
|
||||
// prescribed displacement on the Dirichlet boundary to be a constant
|
||||
// normal displacement of magnitude 'displ_mag'.
|
||||
ConstantCoefficient g(displ_mag);
|
||||
|
||||
LinearForm *b = new LinearForm(fespace);
|
||||
b->AddBdrFaceIntegrator(
|
||||
new SlidingElasticityLFIntegrator(
|
||||
g, lambda_c, mu_c, kappa), ess_bdr);
|
||||
b->Assemble();
|
||||
|
||||
// 11. Assemble the bilinear form and the corresponding linear system,
|
||||
// applying any necessary transformations such as: eliminating boundary
|
||||
// conditions, applying conforming constraints for non-conforming AMR,
|
||||
// static condensation, etc.
|
||||
cout << "matrix ... " << flush;
|
||||
if (static_cond) { a->EnableStaticCondensation(); }
|
||||
a->Assemble();
|
||||
|
||||
SparseMatrix A;
|
||||
Vector B, X;
|
||||
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
|
||||
cout << "done." << endl;
|
||||
|
||||
cout << "Size of linear system: " << A.Height() << endl;
|
||||
|
||||
#ifndef MFEM_USE_SUITESPARSE
|
||||
// 12. Define a simple symmetric Gauss-Seidel preconditioner and use it to
|
||||
// solve the system Ax=b with PCG.
|
||||
GSSmoother M(A);
|
||||
PCG(A, M, B, X, 1, 500, 1e-12, 0.0);
|
||||
#else
|
||||
// 12. If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
|
||||
UMFPackSolver umf_solver;
|
||||
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
umf_solver.SetOperator(A);
|
||||
umf_solver.Mult(B, X);
|
||||
#endif
|
||||
|
||||
// 13. Recover the solution as a finite element grid function.
|
||||
a->RecoverFEMSolution(X, *b, x);
|
||||
|
||||
// 14. For non-NURBS meshes, make the mesh curved based on the finite element
|
||||
// space. This means that we define the mesh elements through a fespace
|
||||
// based transformation of the reference element. This allows us to save
|
||||
// the displaced mesh as a curved mesh when using high-order finite
|
||||
// element displacement field. We assume that the initial mesh (read from
|
||||
// the file) is not higher order curved mesh compared to the chosen FE
|
||||
// space.
|
||||
if (!mesh->NURBSext)
|
||||
{
|
||||
mesh->SetNodalFESpace(fespace);
|
||||
}
|
||||
|
||||
// 15. Save the displaced mesh and the inverted solution (which gives the
|
||||
// backward displacements to the original grid). This output can be
|
||||
// viewed later using GLVis: "glvis -m displaced.mesh -g sol.gf".
|
||||
{
|
||||
GridFunction *nodes = mesh->GetNodes();
|
||||
*nodes += x;
|
||||
x *= -1;
|
||||
ofstream mesh_ofs("displaced.mesh");
|
||||
mesh_ofs.precision(8);
|
||||
mesh->Print(mesh_ofs);
|
||||
ofstream sol_ofs("sol.gf");
|
||||
sol_ofs.precision(8);
|
||||
x.Save(sol_ofs);
|
||||
}
|
||||
|
||||
// 16. Send the above data by socket to a GLVis server. Use the "n" and "b"
|
||||
// keys in GLVis to visualize the displacements.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << *mesh << x << flush;
|
||||
}
|
||||
|
||||
// 17. Free the used memory.
|
||||
delete a;
|
||||
delete b;
|
||||
if (fec)
|
||||
{
|
||||
delete fespace;
|
||||
delete fec;
|
||||
}
|
||||
delete mesh;
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -1,332 +0,0 @@
|
||||
// MFEM Example 43 - Parallel Version
|
||||
//
|
||||
// Compile with: make ex43p
|
||||
//
|
||||
// Sample runs: mpirun -np 4 ex43p -m ../data/ball-nurbs.mesh -r 2
|
||||
// mpirun -np 4 ex43p -m ../data/ref-cube.mesh -r 2
|
||||
// mpirun -np 4 ex43p -m ../data/fichera.mesh
|
||||
// mpirun -np 4 ex43p -m ../data/star.mesh
|
||||
//
|
||||
// Description: This example code solves a linear elasticity problem using
|
||||
// Nitsche's method to enforce sliding boundary conditions. In
|
||||
// particular, we consider a linear elastic body that is displaced
|
||||
// in the normal direction on the entire boundary, but is free to
|
||||
// slide in the tangential direction. This is achieved by imposing
|
||||
// homogeneous Dirichlet boundary conditions on the normal
|
||||
// component of the displacement, while applying homogeneous
|
||||
// Neumann boundary conditions on the tangential components of the
|
||||
// displacement. By enforcing a uniform, constant normal
|
||||
// displacement on the boundary, we can simulate the effect of
|
||||
// compressing or expanding the elastic body uniformly. These
|
||||
// boundary conditions are applied weakly using Nitsche's method,
|
||||
// allowing for more flexibility in handling complex geometries in
|
||||
// either 2D or 3D.
|
||||
//
|
||||
// The strong form is given by:
|
||||
//
|
||||
// −Div(σ(u)) = 0 in Ω
|
||||
// u ⋅ n = g on Γ
|
||||
// σ(u) ⊥ n on Γ
|
||||
//
|
||||
// where σ(u) = λ tr(ε(u)) I + 2μ ε(u) is the stress tensor, ε(u)
|
||||
// is the strain tensor, λ and μ are the Lamé parameters, and g is
|
||||
// the prescribed displacement on the boundary. Here, n is the
|
||||
// outward normal on the boundary Γ = ∂Ω.
|
||||
//
|
||||
// The weak form using Nitsche's method is:
|
||||
//
|
||||
// Find u ∈ V such that a(u,v) = b(v) for all v ∈ V
|
||||
//
|
||||
// where
|
||||
//
|
||||
// a(u,v) := ∫_Ω σ(u) : ε(v) dx
|
||||
// - ∫_Γ (σ(u) n ⋅ n) (v ⋅ n) dS
|
||||
// - ∫_Γ (σ(v) n ⋅ n) (u ⋅ n) dS
|
||||
// + κ ∫_Γ h⁻¹ (λ + 2μ) (u ⋅ n) (v ⋅ n) dS,
|
||||
//
|
||||
// b(v) := - ∫_Γ σ(v) n ⋅ n g dS
|
||||
// + κ ∫_Γ h⁻¹ (λ + 2μ) (v ⋅ n) g dS,
|
||||
//
|
||||
// with κ > 0 being a penalty parameter. Here, h is a
|
||||
// characteristic element size on the boundary. The function
|
||||
// space V is a vector H1-conforming finite element space.
|
||||
//
|
||||
// This example can be viewed as an alternative to Example 28.
|
||||
// Whereas Example 28 imposes sliding boundary conditions using
|
||||
// the general-purpose constrained system solvers found in
|
||||
// mfem/linalg/constraints.hpp, this example employs Nitsche's
|
||||
// method to weakly enforce the same condition by modifying the
|
||||
// underlying variational formulation. Unlike Example 28, the
|
||||
// approach here is specialized to isotropic linear elasticity,
|
||||
// but it has the advantage of producing a well-conditioned SPD
|
||||
// stiffness matrix that can be readily preconditioned with
|
||||
// standard AMG. We recommend reviewing Example 2 before working
|
||||
// through this example.
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
// 1. Initialize MPI and HYPRE.
|
||||
Mpi::Init(argc, argv);
|
||||
int num_procs = Mpi::WorldSize();
|
||||
int myid = Mpi::WorldRank();
|
||||
Hypre::Init();
|
||||
|
||||
// 2. Parse command-line options.
|
||||
const char *mesh_file = "../data/star.mesh";
|
||||
real_t displ_mag = 0.1;
|
||||
int order = 1;
|
||||
int ref_levels = 0;
|
||||
real_t lambda = 1.0;
|
||||
real_t mu = 1.0;
|
||||
real_t kappa = -1.0;
|
||||
bool static_cond = false;
|
||||
bool reorder_space = false;
|
||||
bool visualization = 1;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
"Mesh file to use.");
|
||||
args.AddOption(&displ_mag, "-g", "--displ",
|
||||
"Magnitude of the normal displacement.");
|
||||
args.AddOption(&order, "-o", "--order",
|
||||
"Finite element order (polynomial degree).");
|
||||
args.AddOption(&ref_levels, "-r", "--ref_levels",
|
||||
"Number of uniform mesh refinements.");
|
||||
args.AddOption(&lambda, "-l", "--lambda", "First Lamé parameter.");
|
||||
args.AddOption(&mu, "-mu", "--mu", "Second Lamé parameter.");
|
||||
args.AddOption(&kappa, "-k", "--kappa",
|
||||
"The penalty parameter, should be positive."
|
||||
" Negative values are replaced with (order+1)^2.");
|
||||
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
|
||||
"--no-static-condensation", "Enable static condensation.");
|
||||
args.AddOption(&reorder_space, "-nodes", "--by-nodes", "-vdim", "--by-vdim",
|
||||
"Use byNODES ordering of vector space instead of byVDIM");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
if (kappa < 0)
|
||||
{
|
||||
kappa = (order+1)*(order+1);
|
||||
}
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
// 3. Read the (serial) mesh from the given mesh file. We can handle triangular,
|
||||
// quadrilateral, tetrahedral or hexahedral elements with the same code.
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
|
||||
// 4. Select the order of the finite element discretization space. For NURBS
|
||||
// meshes, we increase the order by degree elevation.
|
||||
if (mesh->NURBSext)
|
||||
{
|
||||
mesh->DegreeElevate(order, order);
|
||||
}
|
||||
|
||||
// 5. Refine the mesh to increase the resolution. In this example we do
|
||||
// 'ref_levels' of uniform refinement.
|
||||
for (int i = 0; i < ref_levels; i++)
|
||||
{
|
||||
mesh->UniformRefinement();
|
||||
}
|
||||
|
||||
// 6. Interpolate the geometry after refinement to control geometry error.
|
||||
int curvature_order = max(order, 2);
|
||||
mesh->SetCurvature(curvature_order);
|
||||
|
||||
// 7. Define a parallel mesh by a partitioning of the serial mesh. Refine
|
||||
// this mesh further in parallel to increase the resolution. Once the
|
||||
// parallel mesh is defined, the serial mesh can be deleted.
|
||||
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
|
||||
delete mesh;
|
||||
|
||||
// 8. Define a finite element space on the mesh. Here we use vector finite
|
||||
// elements, i.e. dim copies of a scalar finite element space. The vector
|
||||
// dimension is specified by the last argument of the FiniteElementSpace
|
||||
// constructor. For NURBS meshes, we use the (degree elevated) NURBS space
|
||||
// associated with the mesh nodes.
|
||||
FiniteElementCollection *fec;
|
||||
ParFiniteElementSpace *fespace;
|
||||
const bool use_nodal_fespace = pmesh->NURBSext;
|
||||
if (use_nodal_fespace)
|
||||
{
|
||||
fec = NULL;
|
||||
fespace = (ParFiniteElementSpace *)pmesh->GetNodes()->FESpace();
|
||||
}
|
||||
else
|
||||
{
|
||||
fec = new H1_FECollection(order, dim);
|
||||
if (reorder_space)
|
||||
{
|
||||
fespace = new ParFiniteElementSpace(pmesh, fec, dim, Ordering::byNODES);
|
||||
}
|
||||
else
|
||||
{
|
||||
fespace = new ParFiniteElementSpace(pmesh, fec, dim, Ordering::byVDIM);
|
||||
}
|
||||
}
|
||||
HYPRE_BigInt size = fespace->GlobalTrueVSize();
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Number of finite element unknowns: " << size << endl
|
||||
<< "Assembling: " << flush;
|
||||
}
|
||||
|
||||
// 9. Mark the boundary attributes where the sliding (Nitsche) boundary
|
||||
// conditions are to be applied. These b.c. are imposed weakly, by adding
|
||||
// the appropriate boundary integrators over the marked 'ess_bdr' to the
|
||||
// bilinear and linear forms. Thus, no dofs are eliminated; there are no
|
||||
// essential boundary conditions.
|
||||
Array<int> ess_tdof_list, ess_bdr;
|
||||
if (pmesh->bdr_attributes.Size())
|
||||
{
|
||||
ess_bdr.SetSize(pmesh->bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
}
|
||||
|
||||
// 10. Define the solution vector x as a finite element grid function
|
||||
// corresponding to fespace. Initialize x with initial guess of zero,
|
||||
// which satisfies the boundary conditions.
|
||||
ParGridFunction x(fespace);
|
||||
x = 0.0;
|
||||
|
||||
// 11. Set up the bilinear form a(.,.) on the finite element space
|
||||
// corresponding to the linear elasticity integrator with constant
|
||||
// coefficients lambda and mu.
|
||||
ConstantCoefficient lambda_c(lambda);
|
||||
ConstantCoefficient mu_c(mu);
|
||||
|
||||
ParBilinearForm *a = new ParBilinearForm(fespace);
|
||||
a->AddDomainIntegrator(new ElasticityIntegrator(lambda_c,mu_c));
|
||||
a->AddBdrFaceIntegrator(
|
||||
new SlidingElasticityIntegrator(lambda_c, mu_c, kappa),
|
||||
ess_bdr);
|
||||
|
||||
// 12. Set up the linear form b(.) corresponding to the Nitsche method
|
||||
// to impose the Dirichlet boundary conditions. Here, we set the
|
||||
// prescribed displacement on the Dirichlet boundary to be a constant
|
||||
// normal displacement of magnitude 'displ_mag'.
|
||||
ConstantCoefficient g(displ_mag);
|
||||
|
||||
ParLinearForm *b = new ParLinearForm(fespace);
|
||||
b->AddBdrFaceIntegrator(
|
||||
new SlidingElasticityLFIntegrator(
|
||||
g, lambda_c, mu_c, kappa), ess_bdr);
|
||||
b->Assemble();
|
||||
|
||||
// 13. Assemble the parallel bilinear form and the corresponding linear
|
||||
// system, applying any necessary transformations such as: parallel
|
||||
// assembly, eliminating boundary conditions, applying conforming
|
||||
// constraints for non-conforming AMR, static condensation, etc.
|
||||
if (myid == 0) { cout << "matrix ... " << flush; }
|
||||
if (static_cond) { a->EnableStaticCondensation(); }
|
||||
a->Assemble();
|
||||
|
||||
HypreParMatrix A;
|
||||
Vector B, X;
|
||||
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "done." << endl;
|
||||
cout << "Size of linear system: " << A.GetGlobalNumRows() << endl;
|
||||
}
|
||||
|
||||
// 14. Define and apply a parallel PCG solver for A X = B with the BoomerAMG
|
||||
// preconditioner from hypre.
|
||||
HypreBoomerAMG *amg = new HypreBoomerAMG(A);
|
||||
if (!a->StaticCondensationIsEnabled())
|
||||
{
|
||||
amg->SetElasticityOptions(fespace);
|
||||
}
|
||||
else
|
||||
{
|
||||
amg->SetSystemsOptions(dim, reorder_space);
|
||||
}
|
||||
HyprePCG *pcg = new HyprePCG(A);
|
||||
pcg->SetTol(1e-8);
|
||||
pcg->SetMaxIter(500);
|
||||
pcg->SetPrintLevel(2);
|
||||
pcg->SetPreconditioner(*amg);
|
||||
pcg->Mult(B, X);
|
||||
|
||||
// 15. Recover the parallel grid function corresponding to X. This is the
|
||||
// local finite element solution on each processor.
|
||||
a->RecoverFEMSolution(X, *b, x);
|
||||
|
||||
// 16. For non-NURBS meshes, make the mesh curved based on the finite element
|
||||
// space. This means that we define the mesh elements through a fespace
|
||||
// based transformation of the reference element. This allows us to save
|
||||
// the displaced mesh as a curved mesh when using high-order finite
|
||||
// element displacement field. We assume that the initial mesh (read from
|
||||
// the file) is not higher order curved mesh compared to the chosen FE
|
||||
// space.
|
||||
if (!use_nodal_fespace)
|
||||
{
|
||||
pmesh->SetNodalFESpace(fespace);
|
||||
}
|
||||
|
||||
// 17. Save in parallel the displaced mesh and the inverted solution (which
|
||||
// gives the backward displacements to the original grid). This output
|
||||
// can be viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
|
||||
{
|
||||
GridFunction *nodes = pmesh->GetNodes();
|
||||
*nodes += x;
|
||||
x *= -1;
|
||||
|
||||
ostringstream mesh_name, sol_name;
|
||||
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
|
||||
sol_name << "sol." << setfill('0') << setw(6) << myid;
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
pmesh->Print(mesh_ofs);
|
||||
|
||||
ofstream sol_ofs(sol_name.str().c_str());
|
||||
sol_ofs.precision(8);
|
||||
x.Save(sol_ofs);
|
||||
}
|
||||
|
||||
// 18. Send the above data by socket to a GLVis server. Use the "n" and "b"
|
||||
// keys in GLVis to visualize the displacements.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << *pmesh << x << flush;
|
||||
}
|
||||
|
||||
// 19. Free the used memory.
|
||||
delete pcg;
|
||||
delete amg;
|
||||
delete a;
|
||||
delete b;
|
||||
if (fec)
|
||||
{
|
||||
delete fespace;
|
||||
delete fec;
|
||||
}
|
||||
delete pmesh;
|
||||
|
||||
return 0;
|
||||
}
|
||||
+6
-1
@@ -9,6 +9,7 @@
|
||||
// mpirun -np 4 ex4p -m ../data/beam-hex.mesh -o 2 -pa
|
||||
// mpirun -np 4 ex4p -m ../data/escher.mesh -o 2 -sc
|
||||
// mpirun -np 4 ex4p -m ../data/fichera.mesh -o 2 -hb
|
||||
// mpirun -np 4 ex4p -m ../data/fichera.mesh -o 2 -hb -ea
|
||||
// mpirun -np 4 ex4p -m ../data/fichera-q2.vtk
|
||||
// mpirun -np 4 ex4p -m ../data/fichera-q3.mesh -o 2 -sc
|
||||
// mpirun -np 4 ex4p -m ../data/square-disc-nurbs.mesh -o 3
|
||||
@@ -17,14 +18,18 @@
|
||||
// mpirun -np 4 ex4p -m ../data/periodic-cube.mesh -no-bc
|
||||
// mpirun -np 4 ex4p -m ../data/amr-quad.mesh
|
||||
// mpirun -np 3 ex4p -m ../data/amr-quad.mesh -o 2 -hb
|
||||
// mpirun -np 3 ex4p -m ../data/amr-quad.mesh -o 2 -hb -ea
|
||||
// mpirun -np 4 ex4p -m ../data/amr-hex.mesh -o 2 -sc
|
||||
// mpirun -np 4 ex4p -m ../data/amr-hex.mesh -o 2 -hb
|
||||
// mpirun -np 4 ex4p -m ../data/amr-hex.mesh -o 2 -hb -ea
|
||||
// mpirun -np 4 ex4p -m ../data/ref-prism.mesh -o 1
|
||||
// mpirun -np 4 ex4p -m ../data/octahedron.mesh -o 1
|
||||
// mpirun -np 4 ex4p -m ../data/star-surf.mesh -o 3 -hb
|
||||
//
|
||||
// Device sample runs:
|
||||
// mpirun -np 4 ex4p -m ../data/star.mesh -pa -d cuda
|
||||
// mpirun -np 4 ex4p -m ../data/star.mesh -ea -hb -d cuda
|
||||
// mpirun -np 4 ex4p -m ../data/amr-hex.mesh -ea -hb -d cuda
|
||||
// mpirun -np 4 ex4p -m ../data/star.mesh -pa -d raja-cuda
|
||||
// mpirun -np 4 ex4p -m ../data/star.mesh -pa -d raja-omp
|
||||
// mpirun -np 4 ex4p -m ../data/beam-hex.mesh -pa -d cuda
|
||||
@@ -230,7 +235,7 @@ int main(int argc, char *argv[])
|
||||
pcg->SetMaxIter(2000);
|
||||
pcg->SetPrintLevel(1);
|
||||
if (hybridization) { prec = new HypreBoomerAMG(*A.As<HypreParMatrix>()); }
|
||||
else if (pa) { prec = new OperatorJacobiSmoother(*a, ess_tdof_list); }
|
||||
else if (pa || ea) { prec = new OperatorJacobiSmoother(*a, ess_tdof_list); }
|
||||
else
|
||||
{
|
||||
ParFiniteElementSpace *prec_fespace =
|
||||
|
||||
+2
-2
@@ -22,11 +22,11 @@ MFEM_LIB_FILE = mfem_is_not_built
|
||||
|
||||
SEQ_EXAMPLES = ex0 ex1 ex2 ex3 ex4 ex5 ex6 ex7 ex8 ex9 ex10 ex14 ex15 ex16 \
|
||||
ex17 ex18 ex19 ex20 ex21 ex22 ex23 ex24 ex25 ex26 ex27 ex28 ex29 ex30 \
|
||||
ex31 ex33 ex34 ex36 ex37 ex38 ex39 ex40 ex41 ex43
|
||||
ex31 ex33 ex34 ex36 ex37 ex38 ex39 ex40 ex41
|
||||
PAR_EXAMPLES = ex0p ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex8p ex9p ex10p ex11p \
|
||||
ex12p ex13p ex14p ex15p ex16p ex17p ex18p ex19p ex20p ex21p ex22p ex24p \
|
||||
ex25p ex26p ex27p ex28p ex29p ex30p ex31p ex32p ex33p ex34p ex35p ex36p \
|
||||
ex37p ex39p ex40p ex41p ex43p
|
||||
ex37p ex39p ex40p ex41p
|
||||
SEQ_DEVICE_EXAMPLES = ex1 ex3 ex4 ex5 ex6 ex9 ex14 ex22 ex24 ex25 ex26 ex34
|
||||
PAR_DEVICE_EXAMPLES = ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex9p ex13p ex14p \
|
||||
ex22p ex24p ex25p ex26p ex34p ex35p
|
||||
|
||||
+35
-6
@@ -825,14 +825,46 @@ void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
|
||||
Vector &b, OperatorHandle &A, Vector &X,
|
||||
Vector &B, int copy_interior)
|
||||
{
|
||||
const SparseMatrix *P = fes->GetConformingProlongation();
|
||||
const SparseMatrix *R = fes->GetConformingRestriction();
|
||||
if (ext)
|
||||
{
|
||||
if (hybridization)
|
||||
{
|
||||
FormSystemMatrix(ess_tdof_list, A);
|
||||
ConstrainedOperator A_constrained(this, ess_tdof_list);
|
||||
A_constrained.EliminateRHS(x, b);
|
||||
hybridization->ReduceRHS(b, B);
|
||||
|
||||
std::unique_ptr<ConstrainedOperator> A_constrained([&]()
|
||||
{
|
||||
Operator *op;
|
||||
Operator::FormSystemOperator(ess_tdof_list, op);
|
||||
return dynamic_cast<ConstrainedOperator*>(op);
|
||||
}());
|
||||
MFEM_ASSERT(A_constrained != nullptr, "");
|
||||
|
||||
Vector conf_b, conf_x;
|
||||
if (P)
|
||||
{
|
||||
// Nonconforming
|
||||
conf_b.SetSize(P->Width());
|
||||
conf_x.SetSize(P->Width());
|
||||
P->MultTranspose(b, conf_b);
|
||||
R->Mult(x, conf_x);
|
||||
}
|
||||
else
|
||||
{
|
||||
// Conforming
|
||||
conf_b.MakeRef(b, 0, b.Size());
|
||||
conf_x.MakeRef(x, 0, x.Size());
|
||||
}
|
||||
|
||||
A_constrained->EliminateRHS(conf_x, conf_b);
|
||||
|
||||
if (P)
|
||||
{
|
||||
R->MultTranspose(conf_b, b); // store eliminated rhs in b
|
||||
}
|
||||
|
||||
hybridization->ReduceRHS(conf_b, B);
|
||||
X.SetSize(B.Size());
|
||||
X = 0.0;
|
||||
}
|
||||
@@ -842,7 +874,6 @@ void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
|
||||
}
|
||||
return;
|
||||
}
|
||||
const SparseMatrix *P = fes->GetConformingProlongation();
|
||||
FormSystemMatrix(ess_tdof_list, A);
|
||||
|
||||
// Transform the system and perform the elimination in B, based on the
|
||||
@@ -878,7 +909,6 @@ void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
|
||||
if (hybridization)
|
||||
{
|
||||
// Reduction to the Lagrange multipliers system
|
||||
const SparseMatrix *R = fes->GetConformingRestriction();
|
||||
Vector conf_b(P->Width()), conf_x(P->Width());
|
||||
P->MultTranspose(b, conf_b);
|
||||
R->Mult(x, conf_x);
|
||||
@@ -891,7 +921,6 @@ void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
|
||||
else
|
||||
{
|
||||
// Variational restriction with P
|
||||
const SparseMatrix *R = fes->GetConformingRestriction();
|
||||
B.SetSize(P->Width());
|
||||
P->MultTranspose(b, B);
|
||||
X.SetSize(R->Height());
|
||||
|
||||
@@ -4213,181 +4213,6 @@ void DGElasticityIntegrator::AssembleFaceMatrix(
|
||||
}
|
||||
}
|
||||
|
||||
void SlidingElasticityIntegrator::AssembleFaceMatrix(
|
||||
const FiniteElement &el1, const FiniteElement &el2,
|
||||
FaceElementTransformations &Trans, DenseMatrix &elmat)
|
||||
{
|
||||
MFEM_ASSERT(Trans.Elem2No < 0,
|
||||
"support for interior faces is not implemented");
|
||||
|
||||
#ifdef MFEM_THREAD_SAFE
|
||||
// For descriptions of these variables, see the class declaration.
|
||||
Vector shape1;
|
||||
DenseMatrix dshape1;
|
||||
DenseMatrix adjJ;
|
||||
DenseMatrix dshape1_ps;
|
||||
Vector nor;
|
||||
Vector nL1;
|
||||
Vector nM1;
|
||||
Vector nt1;
|
||||
Vector dshape1_dnM;
|
||||
Vector dshape1_dnt;
|
||||
DenseMatrix jmat;
|
||||
#endif
|
||||
|
||||
const int dim = el1.GetDim();
|
||||
const int ndofs1 = el1.GetDof();
|
||||
const int nvdofs = dim * ndofs1;
|
||||
|
||||
// Initially 'elmat' corresponds to the term:
|
||||
// < { sigma(u) n . ñ }, v . ñ > =
|
||||
// < { (lambda div(u) I + mu (grad(u) + grad(u)^T)) n . ñ }, v . ñ >
|
||||
// But eventually, it's going to be replaced by:
|
||||
// elmat := -elmat + alpha*elmat^T + jmat
|
||||
elmat.SetSize(nvdofs);
|
||||
elmat = 0.;
|
||||
|
||||
const bool kappa_is_nonzero = (kappa != 0.0);
|
||||
if (kappa_is_nonzero)
|
||||
{
|
||||
jmat.SetSize(nvdofs);
|
||||
jmat = 0.;
|
||||
}
|
||||
|
||||
adjJ.SetSize(dim);
|
||||
shape1.SetSize(ndofs1);
|
||||
dshape1.SetSize(ndofs1, dim);
|
||||
dshape1_ps.SetSize(ndofs1, dim);
|
||||
nor.SetSize(dim);
|
||||
nL1.SetSize(dim);
|
||||
nM1.SetSize(dim);
|
||||
nt1.SetSize(dim);
|
||||
dshape1_dnM.SetSize(ndofs1);
|
||||
dshape1_dnt.SetSize(ndofs1);
|
||||
|
||||
const IntegrationRule *ir = IntRule;
|
||||
if (ir == NULL)
|
||||
{
|
||||
// a simple choice for the integration order; is this OK?
|
||||
const int order = 2 * el1.GetOrder();
|
||||
ir = &IntRules.Get(Trans.GetGeometryType(), order);
|
||||
}
|
||||
|
||||
for (int pind = 0; pind < ir->GetNPoints(); ++pind)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(pind);
|
||||
|
||||
// Set the integration point in the face and the neighboring elements
|
||||
Trans.SetAllIntPoints(&ip);
|
||||
|
||||
// Access the neighboring element's integration point
|
||||
const IntegrationPoint &eip1 = Trans.GetElement1IntPoint();
|
||||
|
||||
el1.CalcShape(eip1, shape1);
|
||||
el1.CalcDShape(eip1, dshape1);
|
||||
|
||||
CalcAdjugate(Trans.Elem1->Jacobian(), adjJ);
|
||||
Mult(dshape1, adjJ, dshape1_ps);
|
||||
|
||||
if (dim == 1)
|
||||
{
|
||||
nor(0) = 2*eip1.x - 1.0;
|
||||
}
|
||||
else
|
||||
{
|
||||
CalcOrtho(Trans.Jacobian(), nor);
|
||||
}
|
||||
|
||||
if (!nt)
|
||||
{
|
||||
// Set ñ to the unit normal vector if not provided
|
||||
nt1 = nor;
|
||||
nt1 /= nt1.Norml2();
|
||||
}
|
||||
else
|
||||
{
|
||||
// Evaluate vector function ñ at integration point
|
||||
nt->Eval(nt1, *Trans.Elem1, eip1);
|
||||
}
|
||||
|
||||
const real_t W = ip.weight;
|
||||
const real_t W1 = W / Trans.Elem1->Weight();
|
||||
const real_t WL1 = W1 * lambda->Eval(*Trans.Elem1, eip1);
|
||||
const real_t WM1 = W1 * mu->Eval(*Trans.Elem1, eip1);
|
||||
nL1.Set(WL1, nor);
|
||||
nM1.Set(WM1, nor);
|
||||
const real_t WLM = WL1 + 2.0*WM1;
|
||||
dshape1_ps.Mult(nM1, dshape1_dnM);
|
||||
dshape1_ps.Mult(nt1, dshape1_dnt);
|
||||
|
||||
const real_t jmatcoef = kappa * (nor*nor) * WLM;
|
||||
|
||||
const real_t nL_dot_nt1 = nL1 * nt1;
|
||||
for (int jm = 0, j = 0; jm < dim; ++jm)
|
||||
{
|
||||
for (int jdof = 0; jdof < ndofs1; ++jdof, ++j)
|
||||
{
|
||||
const real_t t1 = dshape1_ps(jdof, jm) * nL_dot_nt1;
|
||||
const real_t t2 = dshape1_dnM(jdof) * nt1(jm);
|
||||
const real_t t3 = dshape1_dnt(jdof) * nM1(jm);
|
||||
const real_t tt = t1 + t2 + t3;
|
||||
for (int im = 0, i = 0; im < dim; ++im)
|
||||
{
|
||||
for (int idof = 0; idof < ndofs1; ++idof, ++i)
|
||||
{
|
||||
elmat(i, j) += tt * shape1(idof) * nt1(im);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (kappa_is_nonzero)
|
||||
{
|
||||
for (int jm = 0, j = 0; jm < dim; ++jm)
|
||||
{
|
||||
for (int jdof = 0; jdof < ndofs1; ++jdof, ++j)
|
||||
{
|
||||
const real_t sj = jmatcoef * shape1(jdof) * nt1(jm);
|
||||
for (int im = 0, i = 0; im < dim; ++im)
|
||||
{
|
||||
for (int idof = 0; idof < ndofs1; ++idof, ++i)
|
||||
{
|
||||
jmat(i, j) += shape1(idof) * sj * nt1(im);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// elmat := -elmat + alpha*elmat^t + jmat
|
||||
if (kappa_is_nonzero)
|
||||
{
|
||||
for (int i = 0; i < nvdofs; ++i)
|
||||
{
|
||||
for (int j = 0; j < i; ++j)
|
||||
{
|
||||
real_t aij = elmat(i,j), aji = elmat(j,i), mij = jmat(i,j);
|
||||
elmat(i,j) = alpha*aji - aij + mij;
|
||||
elmat(j,i) = alpha*aij - aji + mij;
|
||||
}
|
||||
elmat(i,i) = (alpha - 1.)*elmat(i,i) + jmat(i,i);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i < nvdofs; ++i)
|
||||
{
|
||||
for (int j = 0; j < i; ++j)
|
||||
{
|
||||
real_t aij = elmat(i,j), aji = elmat(j,i);
|
||||
elmat(i,j) = alpha*aji - aij;
|
||||
elmat(j,i) = alpha*aij - aji;
|
||||
}
|
||||
elmat(i,i) *= (alpha - 1.);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void TraceJumpIntegrator::AssembleFaceMatrix(
|
||||
const FiniteElement &trial_face_fe, const FiniteElement &test_fe1,
|
||||
|
||||
@@ -3738,84 +3738,6 @@ protected:
|
||||
DenseMatrix &elmat, DenseMatrix &jmat);
|
||||
};
|
||||
|
||||
/** Integrator for the Nitsche elasticity form:
|
||||
$$
|
||||
\begin{split}
|
||||
a(u,v)
|
||||
&:= -\langle \sigma(u)\, \vec{n} \cdot \tilde{n},\ v \cdot \tilde{n}
|
||||
\rangle + \alpha \langle \sigma(v)\, \vec{n} \cdot \tilde{n},\ u \cdot
|
||||
\tilde{n} \rangle + \kappa \langle h^{-1} (\lambda + 2\mu)\, u \cdot
|
||||
\tilde{n},\ v \cdot \tilde{n} \rangle \\
|
||||
&= -\int_\Gamma (\sigma(u)\, n \cdot \tilde{n})(v \cdot \tilde{n})\, dS +
|
||||
\alpha \int_\Gamma (\sigma(v)\, n \cdot \tilde{n})(u \cdot \tilde{n})\,
|
||||
dS + \kappa \int_\Gamma h^{-1} (\lambda + 2\mu)(u \cdot \tilde{n})(v
|
||||
\cdot \tilde{n})\, dS.
|
||||
\end{split}
|
||||
$$
|
||||
|
||||
For isotropic media,
|
||||
$$
|
||||
\begin{split}
|
||||
\sigma(u) &= \lambda \nabla \cdot u I + 2 \mu \varepsilon(u) \\
|
||||
&= \lambda \nabla \cdot u I + 2 \mu \frac{1}{2} (\nabla u + \nabla
|
||||
u^{\mathrm{T}}) \\
|
||||
&= \lambda \nabla \cdot u I + \mu (\nabla u + \nabla u^{\mathrm{T}})
|
||||
\end{split}
|
||||
$$
|
||||
where $I$ is the identity matrix, $\lambda$ and $\mu$ are the Lamé
|
||||
coefficients (see ElasticityIntegrator), $\tilde{n}$ is a unit vector
|
||||
field, $\alpha = \pm 1$ and $\kappa > 0$ are the Nitsche parameters, and
|
||||
$u$, $v$ are the trial and test functions, respectively.
|
||||
|
||||
This is a '%Vector' integrator, i.e. defined for FE spaces using multiple
|
||||
copies of a scalar FE space.
|
||||
*/
|
||||
class SlidingElasticityIntegrator : public BilinearFormIntegrator
|
||||
{
|
||||
public:
|
||||
SlidingElasticityIntegrator(Coefficient &lambda_, Coefficient &mu_,
|
||||
real_t kappa_)
|
||||
: nt(NULL), lambda(&lambda_), mu(&mu_), alpha(-1.0), kappa(kappa_) { }
|
||||
|
||||
SlidingElasticityIntegrator(VectorCoefficient &nt_, Coefficient &lambda_,
|
||||
Coefficient &mu_, real_t alpha_, real_t kappa_)
|
||||
: nt(&nt_), lambda(&lambda_), mu(&mu_), alpha(alpha_), kappa(kappa_) { }
|
||||
|
||||
using BilinearFormIntegrator::AssembleFaceMatrix;
|
||||
void AssembleFaceMatrix(const FiniteElement &el1,
|
||||
const FiniteElement &el2,
|
||||
FaceElementTransformations &Trans,
|
||||
DenseMatrix &elmat) override;
|
||||
|
||||
protected:
|
||||
VectorCoefficient *nt;
|
||||
Coefficient *lambda, *mu;
|
||||
real_t alpha, kappa;
|
||||
|
||||
#ifndef MFEM_THREAD_SAFE
|
||||
// values of all scalar basis functions for one component of u (which is a
|
||||
// vector) at the integration point in the reference space
|
||||
Vector shape1;
|
||||
// values of derivatives of all scalar basis functions for one component
|
||||
// of u (which is a vector) at the integration point in the reference space
|
||||
DenseMatrix dshape1;
|
||||
// Adjugate of the Jacobian of the transformation: adjJ = det(J) J^{-1}
|
||||
DenseMatrix adjJ;
|
||||
// gradient of shape functions in the real (physical, not reference)
|
||||
// coordinates, scaled by det(J):
|
||||
// dshape_ps(jdof,jm) = sum_{t} adjJ(t,jm)*dshape(jdof,t)
|
||||
DenseMatrix dshape1_ps;
|
||||
Vector nor; // nor = |weight(J_face)| n
|
||||
Vector nL1; // nL1 = (lambda1 * ip.weight / detJ1) nor
|
||||
Vector nM1; // nM1 = (mu1 * ip.weight / detJ1) nor
|
||||
Vector nt1; // nt1 = vector function ñ evaluated at ip1
|
||||
Vector dshape1_dnM; // dshape1_dnM = dshape1_ps . nM1
|
||||
Vector dshape1_dnt; // dshape1_dnt = dshape1_ps . nt1
|
||||
// 'jmat' corresponds to the term: kappa <h⁻¹ u ⋅ ñ, v ⋅ ñ>
|
||||
DenseMatrix jmat;
|
||||
#endif
|
||||
};
|
||||
|
||||
/** Integrator for the DPG form:$ \langle v, [w] \rangle $ over all faces (the interface) where
|
||||
the trial variable $v$ is defined on the interface and the test variable $w$ is
|
||||
defined inside the elements, generally in a DG space. */
|
||||
|
||||
@@ -387,7 +387,7 @@ void DGMassInverse::DGMassCGIteration(const Vector &b_, Vector &u_) const
|
||||
|
||||
static constexpr int NB = Q1D ? Q1D : 1; // block size
|
||||
|
||||
mfem::forall_2D(NE, NB, NB, [=] MFEM_HOST_DEVICE (int e)
|
||||
mfem::forall_2D<NB*NB>(NE, NB, NB, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
// Perform change of basis if needed
|
||||
if (CHANGE_BASIS)
|
||||
|
||||
@@ -69,9 +69,9 @@ inline int ToLexOrdering2D(const int face_id, const int size1d, const int i)
|
||||
}
|
||||
|
||||
/// @brief Given a face DOF index on a shared face, ordered lexicographically
|
||||
/// relative to element the element (where the local face is face_id), and
|
||||
/// return the corresponding face DOF index ordered lexicographically relative
|
||||
/// to the face itself.
|
||||
/// relative to the element (where the local face is face_id), return the
|
||||
/// corresponding face DOF index ordered lexicographically relative to the face
|
||||
/// itself.
|
||||
MFEM_HOST_DEVICE
|
||||
inline int PermuteFace2D(const int face_id, const int orientation,
|
||||
const int size1d, const int index)
|
||||
|
||||
+22
-34
@@ -231,7 +231,7 @@ void FiniteElement::CalcPhysLaplacian(ElementTransformation &Trans,
|
||||
{
|
||||
for (int nd = 0; nd < dof; nd++)
|
||||
{
|
||||
Laplacian[nd] = hess(nd,0) + hess(nd,4) + hess(nd,5);
|
||||
Laplacian[nd] = hess(nd,0) + hess(nd,3) + hess(nd,5);
|
||||
}
|
||||
}
|
||||
else if (dim == 2)
|
||||
@@ -268,11 +268,9 @@ void FiniteElement::CalcPhysLinLaplacian(ElementTransformation &Trans,
|
||||
scale[0] = Gij(0,0);
|
||||
scale[1] = 2*Gij(0,1);
|
||||
scale[2] = 2*Gij(0,2);
|
||||
|
||||
scale[3] = 2*Gij(1,2);
|
||||
scale[4] = Gij(2,2);
|
||||
|
||||
scale[5] = Gij(1,1);
|
||||
scale[3] = Gij(1,1);
|
||||
scale[4] = 2*Gij(1,2);
|
||||
scale[5] = Gij(2,2);
|
||||
}
|
||||
else if (dim == 2)
|
||||
{
|
||||
@@ -309,12 +307,12 @@ void FiniteElement::CalcPhysHessian(ElementTransformation &Trans,
|
||||
map[2] = 2;
|
||||
|
||||
map[3] = 1;
|
||||
map[4] = 5;
|
||||
map[5] = 3;
|
||||
map[4] = 3;
|
||||
map[5] = 4;
|
||||
|
||||
map[6] = 2;
|
||||
map[7] = 3;
|
||||
map[8] = 4;
|
||||
map[7] = 4;
|
||||
map[8] = 5;
|
||||
}
|
||||
else if (dim == 2)
|
||||
{
|
||||
@@ -382,11 +380,7 @@ const DofToQuad &FiniteElement::GetDofToQuad(const IntegrationRule &ir,
|
||||
#pragma omp critical (DofToQuad)
|
||||
#endif
|
||||
{
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
{
|
||||
d2q = dof2quad_array[i];
|
||||
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
|
||||
}
|
||||
d2q = DofToQuad::SearchArray(dof2quad_array, ir, mode);
|
||||
if (!d2q)
|
||||
{
|
||||
#ifdef MFEM_THREAD_SAFE
|
||||
@@ -661,14 +655,22 @@ void ScalarFiniteElement::ScalarLocalL2Restriction(
|
||||
void NodalFiniteElement::CreateLexicographicFullMap(const IntegrationRule &ir)
|
||||
const
|
||||
{
|
||||
// Get the FULL version of the map. This call contains omp critical region,
|
||||
// so it is done before the critical region below.
|
||||
auto &d2q = GetDofToQuad(ir, DofToQuad::FULL);
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
#pragma omp critical (DofToQuad)
|
||||
#endif
|
||||
{
|
||||
// Get the FULL version of the map.
|
||||
auto &d2q = GetDofToQuad(ir, DofToQuad::FULL);
|
||||
//Undo the native ordering which is what FiniteElement::GetDofToQuad returns.
|
||||
// If the new Dof2Quad is already present, e.g. added in a previous call
|
||||
// or added by another omp thread, return.
|
||||
if (DofToQuad::SearchArray(dof2quad_array, ir,
|
||||
DofToQuad::LEXICOGRAPHIC_FULL))
|
||||
{ return; }
|
||||
|
||||
// Undo the native ordering which is what FiniteElement::GetDofToQuad
|
||||
// returns.
|
||||
auto *d2q_new = new DofToQuad(d2q);
|
||||
d2q_new->mode = DofToQuad::LEXICOGRAPHIC_FULL;
|
||||
const int nqpt = ir.GetNPoints();
|
||||
@@ -724,13 +726,7 @@ const DofToQuad &NodalFiniteElement::GetDofToQuad(const IntegrationRule &ir,
|
||||
#pragma omp critical (DofToQuad)
|
||||
#endif
|
||||
{
|
||||
//Should make this loop a function of FiniteElement
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
{
|
||||
d2q = dof2quad_array[i];
|
||||
if (d2q->IntRule == &ir && d2q->mode == mode) { break; }
|
||||
d2q = nullptr;
|
||||
}
|
||||
d2q = DofToQuad::SearchArray(dof2quad_array, ir, mode);
|
||||
}
|
||||
if (d2q) { return *d2q; }
|
||||
if (mode != DofToQuad::LEXICOGRAPHIC_FULL)
|
||||
@@ -2631,15 +2627,7 @@ const DofToQuad &TensorBasisElement::GetTensorDofToQuad(
|
||||
#pragma omp critical (DofToQuad)
|
||||
#endif
|
||||
{
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
{
|
||||
auto* d2q_ = dof2quad_array[i];
|
||||
if (d2q_->IntRule == &ir && d2q_->mode == mode)
|
||||
{
|
||||
d2q = d2q_;
|
||||
break;
|
||||
}
|
||||
}
|
||||
d2q = DofToQuad::SearchArray(dof2quad_array, ir, mode);
|
||||
if (!d2q)
|
||||
{
|
||||
d2q = new DofToQuad;
|
||||
|
||||
@@ -222,6 +222,12 @@ public:
|
||||
|
||||
/// Returns absolute value of the maps
|
||||
DofToQuad Abs() const;
|
||||
|
||||
/// Auxiliary function for searching DofToQuad arrays.
|
||||
static inline DofToQuad *SearchArray(
|
||||
const Array<DofToQuad*> &dof2quad_array,
|
||||
const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode);
|
||||
};
|
||||
|
||||
/// Describes the function space on each element
|
||||
@@ -407,6 +413,7 @@ public:
|
||||
/** Each row of the result DenseMatrix @a Hessian contains upper triangular
|
||||
part of the Hessian of one shape function.
|
||||
The order in 2D is {u_xx, u_xy, u_yy}.
|
||||
The order in 3D is {u_xx, u_xy, u_xz, u_yy, u_yz, u_zz}.
|
||||
The size (#dof x (#dim (#dim+1)/2) of @a Hessian must be set in advance.*/
|
||||
virtual void CalcHessian(const IntegrationPoint &ip,
|
||||
DenseMatrix &Hessian) const;
|
||||
@@ -1376,6 +1383,21 @@ public:
|
||||
void InvertLinearTrans(ElementTransformation &trans,
|
||||
const IntegrationPoint &pt, Vector &x);
|
||||
|
||||
|
||||
// static inline method
|
||||
inline DofToQuad *DofToQuad::SearchArray(
|
||||
const Array<DofToQuad*> &dof2quad_array,
|
||||
const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode)
|
||||
{
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
{
|
||||
DofToQuad *d2q = dof2quad_array[i];
|
||||
if (d2q->IntRule == &ir && d2q->mode == mode) { return d2q; }
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
|
||||
@@ -60,6 +60,12 @@ void Linear1DFiniteElement::CalcDShape(const IntegrationPoint &ip,
|
||||
dshape(1,0) = 1.;
|
||||
}
|
||||
|
||||
void Linear1DFiniteElement::CalcHessian(const IntegrationPoint &ip,
|
||||
DenseMatrix &h) const
|
||||
{
|
||||
h = 0.0;
|
||||
}
|
||||
|
||||
Linear2DFiniteElement::Linear2DFiniteElement()
|
||||
: NodalFiniteElement(2, Geometry::TRIANGLE, 3, 1)
|
||||
{
|
||||
@@ -87,6 +93,11 @@ void Linear2DFiniteElement::CalcDShape(const IntegrationPoint &ip,
|
||||
dshape(2,0) = 0.; dshape(2,1) = 1.;
|
||||
}
|
||||
|
||||
void Linear2DFiniteElement::CalcHessian(const IntegrationPoint &ip,
|
||||
DenseMatrix &h) const
|
||||
{
|
||||
h = 0.0;
|
||||
}
|
||||
|
||||
BiLinear2DFiniteElement::BiLinear2DFiniteElement()
|
||||
: NodalFiniteElement(2, Geometry::SQUARE, 4, 1, FunctionSpace::Qk)
|
||||
@@ -1256,6 +1267,12 @@ void Linear3DFiniteElement::CalcDShape(const IntegrationPoint &ip,
|
||||
}
|
||||
}
|
||||
|
||||
void Linear3DFiniteElement::CalcHessian(const IntegrationPoint &ip,
|
||||
DenseMatrix &h) const
|
||||
{
|
||||
h = 0.0;
|
||||
}
|
||||
|
||||
void Linear3DFiniteElement::GetFaceDofs (int face, int **dofs, int *ndofs)
|
||||
const
|
||||
{
|
||||
@@ -1632,6 +1649,37 @@ void TriLinear3DFiniteElement::CalcDShape(const IntegrationPoint &ip,
|
||||
dshape(7,2) = ox * y;
|
||||
}
|
||||
|
||||
void TriLinear3DFiniteElement::CalcHessian(const IntegrationPoint &ip,
|
||||
DenseMatrix &h) const
|
||||
{
|
||||
real_t x = ip.x, y = ip.y, z = ip.z;
|
||||
real_t ox = 1.-x, oy = 1.-y, oz = 1.-z;
|
||||
|
||||
h(0,0) = 0.; h(0,1) = oz; h(0,2) = oy;
|
||||
h(0,3) = 0.; h(0,4) = ox; h(0,5) = 0.;
|
||||
|
||||
h(1,0) = 0.; h(1,1) = -oz; h(1,2) = -oy;
|
||||
h(1,3) = 0.; h(1,4) = x; h(1,5) = 0.;
|
||||
|
||||
h(2,0) = 0.; h(2,1) = oz; h(2,2) = -y;
|
||||
h(2,3) = 0.; h(2,4) = -x; h(2,5) = 0.;
|
||||
|
||||
h(3,0) = 0.; h(3,1) = -oz; h(3,2) = y;
|
||||
h(3,3) = 0.; h(3,4) = -ox; h(3,5) = 0.;
|
||||
|
||||
h(4,0) = 0.; h(4,1) = z; h(4,2) = -oy;
|
||||
h(4,3) = 0.; h(4,4) = -ox; h(4,5) = 0.;
|
||||
|
||||
h(5,0) = 0.; h(5,1) = -z; h(5,2) = oy;
|
||||
h(5,3) = 0.; h(5,4) = -x; h(5,5) = 0.;
|
||||
|
||||
h(6,0) = 0.; h(6,1) = z; h(6,2) = y;
|
||||
h(6,3) = 0.; h(6,4) = x; h(6,5) = 0.;
|
||||
|
||||
h(7,0) = 0.; h(7,1) = -z; h(7,2) = -y;
|
||||
h(7,3) = 0.; h(7,4) = ox; h(7,5) = 0.;
|
||||
}
|
||||
|
||||
|
||||
P0SegmentFiniteElement::P0SegmentFiniteElement(int Ord)
|
||||
: NodalFiniteElement(1, Geometry::SEGMENT, 1, Ord) // default Ord = 0
|
||||
|
||||
@@ -50,6 +50,8 @@ public:
|
||||
contains the derivative of one shape function */
|
||||
void CalcDShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &dshape) const override;
|
||||
void CalcHessian(const IntegrationPoint &ip,
|
||||
DenseMatrix &h) const override;
|
||||
};
|
||||
|
||||
/// A 2D linear element on triangle with nodes at the vertices of the triangle
|
||||
@@ -70,6 +72,8 @@ public:
|
||||
so that each row contains the derivatives of one shape function */
|
||||
void CalcDShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &dshape) const override;
|
||||
void CalcHessian(const IntegrationPoint &ip,
|
||||
DenseMatrix &h) const override;
|
||||
void ProjectDelta(int vertex, Vector &dofs) const override
|
||||
{ dofs = 0.0; dofs(vertex) = 1.0; }
|
||||
};
|
||||
@@ -404,6 +408,9 @@ public:
|
||||
void CalcDShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &dshape) const override;
|
||||
|
||||
void CalcHessian(const IntegrationPoint &ip,
|
||||
DenseMatrix &h) const override;
|
||||
|
||||
void ProjectDelta(int vertex, Vector &dofs) const override
|
||||
{ dofs = 0.0; dofs(vertex) = 1.0; }
|
||||
|
||||
@@ -445,7 +452,8 @@ public:
|
||||
so that each row contains the derivatives of one shape function */
|
||||
void CalcDShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &dshape) const override;
|
||||
|
||||
void CalcHessian(const IntegrationPoint &ip,
|
||||
DenseMatrix &h) const override;
|
||||
void ProjectDelta(int vertex, Vector &dofs) const override
|
||||
{ dofs = 0.0; dofs(vertex) = 1.0; }
|
||||
};
|
||||
|
||||
+3
-4
@@ -445,11 +445,10 @@ void NURBS3DFiniteElement::CalcHessian (const IntegrationPoint &ip,
|
||||
d2sum[0] += ( hessian(o,0) = d2sx*sy*sz*weights(o) );
|
||||
d2sum[1] += ( hessian(o,1) = dsx*dsy*sz*weights(o) );
|
||||
d2sum[2] += ( hessian(o,2) = dsx*sy*dsz*weights(o) );
|
||||
d2sum[3] += ( hessian(o,3) = sx*d2sy*sz*weights(o) );
|
||||
d2sum[4] += ( hessian(o,4) = sx*dsy*dsz*weights(o) );
|
||||
d2sum[5] += ( hessian(o,5) = sx*sy*d2sz*weights(o) );
|
||||
|
||||
d2sum[3] += ( hessian(o,3) = sx*dsy*dsz*weights(o) );
|
||||
|
||||
d2sum[4] += ( hessian(o,4) = sx*sy*d2sz*weights(o) );
|
||||
d2sum[5] += ( hessian(o,5) = sx*d2sy*sz*weights(o) );
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+50
-13
@@ -1516,36 +1516,76 @@ const FaceRestriction *FiniteElementSpace::GetFaceRestriction(
|
||||
const bool is_dg_space = IsDGSpace();
|
||||
const L2FaceValues m = (is_dg_space && mul==L2FaceValues::DoubleValued) ?
|
||||
L2FaceValues::DoubleValued : L2FaceValues::SingleValued;
|
||||
key_face key = std::make_tuple(is_dg_space, f_ordering, type, m);
|
||||
auto key = std::make_tuple(is_dg_space, f_ordering, type, m);
|
||||
auto itr = L2F.find(key);
|
||||
if (itr != L2F.end())
|
||||
{
|
||||
return itr->second;
|
||||
return itr->second.get();
|
||||
}
|
||||
else
|
||||
{
|
||||
FaceRestriction *res;
|
||||
std::unique_ptr<FaceRestriction> res;
|
||||
if (is_dg_space)
|
||||
{
|
||||
if (Conforming())
|
||||
{
|
||||
res = new L2FaceRestriction(*this, f_ordering, type, m);
|
||||
res.reset(new L2FaceRestriction(*this, f_ordering, type, m));
|
||||
}
|
||||
else
|
||||
{
|
||||
res = new NCL2FaceRestriction(*this, f_ordering, type, m);
|
||||
res.reset(new NCL2FaceRestriction(*this, f_ordering, type, m));
|
||||
}
|
||||
}
|
||||
else if (dynamic_cast<const DG_Interface_FECollection*>(fec))
|
||||
{
|
||||
res = new L2InterfaceFaceRestriction(*this, f_ordering, type);
|
||||
res.reset(new L2InterfaceFaceRestriction(*this, f_ordering, type));
|
||||
}
|
||||
else
|
||||
{
|
||||
res = new ConformingFaceRestriction(*this, f_ordering, type);
|
||||
res.reset(new ConformingFaceRestriction(*this, f_ordering, type));
|
||||
}
|
||||
L2F[key] = res;
|
||||
return res;
|
||||
return L2F.emplace(key, std::move(res)).first->second.get();
|
||||
}
|
||||
}
|
||||
|
||||
const InterpolationManager &FiniteElementSpace::GetInterpolationManager(
|
||||
ElementDofOrdering f_ordering, FaceType type) const
|
||||
{
|
||||
const auto key = make_tuple(f_ordering, type);
|
||||
|
||||
auto it = interpolations.find(key);
|
||||
if (it != interpolations.end())
|
||||
{
|
||||
return *it->second;
|
||||
}
|
||||
else
|
||||
{
|
||||
auto interp = make_unique<InterpolationManager>(*this, f_ordering, type);
|
||||
|
||||
int face_idx = 0;
|
||||
for (int f = 0; f < mesh->GetNumFacesWithGhost(); ++f)
|
||||
{
|
||||
Mesh::FaceInformation face = mesh->GetFaceInformation(f);
|
||||
if (!face.IsOfFaceType(type) || face.IsNonconformingCoarse())
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if (face.IsConforming() || face.IsBoundary())
|
||||
{
|
||||
interp->RegisterFaceConformingInterpolation(face, face_idx);
|
||||
}
|
||||
else
|
||||
{
|
||||
interp->RegisterFaceCoarseToFineInterpolation(face, face_idx);
|
||||
}
|
||||
++face_idx;
|
||||
}
|
||||
|
||||
// Transform the interpolation matrix map into contiguous memory.
|
||||
interp->LinearizeInterpolatorMapIntoVector();
|
||||
interp->InitializeNCInterpConfig();
|
||||
|
||||
return *interpolations.emplace(key, std::move(interp)).first->second;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3969,11 +4009,8 @@ void FiniteElementSpace::Destroy()
|
||||
delete E2Q_array[i];
|
||||
}
|
||||
E2Q_array.SetSize(0);
|
||||
for (auto &x : L2F)
|
||||
{
|
||||
delete x.second;
|
||||
}
|
||||
L2F.clear();
|
||||
interpolations.clear();
|
||||
for (int i = 0; i < E2IFQ_array.Size(); i++)
|
||||
{
|
||||
delete E2IFQ_array[i];
|
||||
|
||||
+9
-12
@@ -13,6 +13,7 @@
|
||||
#define MFEM_FESPACE
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#include "../general/hash_util.hpp"
|
||||
#include "../linalg/ordering.hpp"
|
||||
#include "../linalg/sparsemat.hpp"
|
||||
#include "../mesh/mesh.hpp"
|
||||
@@ -320,18 +321,11 @@ protected:
|
||||
mutable OperatorHandle L2E_nat, L2E_lex;
|
||||
/// The face restriction operators, see GetFaceRestriction().
|
||||
using key_face = std::tuple<bool, ElementDofOrdering, FaceType, L2FaceValues>;
|
||||
struct key_hash
|
||||
{
|
||||
std::size_t operator()(const key_face& k) const
|
||||
{
|
||||
return std::get<0>(k)
|
||||
+ 2 * (int)std::get<1>(k)
|
||||
+ 4 * (int)std::get<2>(k)
|
||||
+ 8 * (int)std::get<3>(k);
|
||||
}
|
||||
};
|
||||
using map_L2F = std::unordered_map<const key_face,FaceRestriction*,key_hash>;
|
||||
mutable map_L2F L2F;
|
||||
mutable std::unordered_map<key_face,std::unique_ptr<FaceRestriction>,
|
||||
TupleHasher> L2F;
|
||||
|
||||
mutable std::unordered_map<std::tuple<ElementDofOrdering,FaceType>,
|
||||
std::unique_ptr<InterpolationManager>, TupleHasher> interpolations;
|
||||
|
||||
mutable Array<QuadratureInterpolator*> E2Q_array;
|
||||
mutable Array<FaceQuadratureInterpolator*> E2IFQ_array;
|
||||
@@ -751,6 +745,9 @@ public:
|
||||
ElementDofOrdering f_ordering, FaceType,
|
||||
L2FaceValues mul = L2FaceValues::DoubleValued) const;
|
||||
|
||||
const InterpolationManager &GetInterpolationManager(
|
||||
ElementDofOrdering f_ordering, FaceType type) const;
|
||||
|
||||
/** @brief Return a QuadratureInterpolator that interpolates E-vectors to
|
||||
quadrature point values and/or derivatives (Q-vectors). */
|
||||
/** An E-vector represents the element-wise discontinuous version of the FE
|
||||
|
||||
+59
-32
@@ -234,7 +234,7 @@ void FindPointsGSLIB::Setup(Mesh &m, const double bb_t, const double newt_tol,
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::FindPoints(const Vector &point_pos,
|
||||
int point_pos_ordering)
|
||||
const int point_pos_ordering)
|
||||
{
|
||||
MFEM_VERIFY(setupflag, "Use FindPointsGSLIB::Setup before finding points.");
|
||||
bool dev_mode = (point_pos.UseDevice() && Device::IsEnabled());
|
||||
@@ -482,7 +482,7 @@ void FindPointsGSLIB::SetupDevice()
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::FindPointsOnDevice(const Vector &point_pos,
|
||||
int point_pos_ordering)
|
||||
const int point_pos_ordering)
|
||||
{
|
||||
if (!DEV.setup_device)
|
||||
{
|
||||
@@ -505,13 +505,13 @@ void FindPointsGSLIB::FindPointsOnDevice(const Vector &point_pos,
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
FindPointsLocal2(point_pos, point_pos_ordering, gsl_code, gsl_elem, gsl_ref,
|
||||
gsl_dist, points_cnt);
|
||||
FindPointsLocal2(point_pos, point_pos_ordering, gsl_code, gsl_elem,
|
||||
gsl_ref, gsl_dist, points_cnt);
|
||||
}
|
||||
else
|
||||
{
|
||||
FindPointsLocal3(point_pos, point_pos_ordering, gsl_code, gsl_elem, gsl_ref,
|
||||
gsl_dist, points_cnt);
|
||||
FindPointsLocal3(point_pos, point_pos_ordering, gsl_code, gsl_elem,
|
||||
gsl_ref, gsl_dist, points_cnt);
|
||||
}
|
||||
|
||||
// Sync from device to host
|
||||
@@ -1085,7 +1085,7 @@ void FindPointsGSLIB::InterpolateOnDevice(const Vector &field_in_evec,
|
||||
#else
|
||||
void FindPointsGSLIB::SetupDevice() {};
|
||||
void FindPointsGSLIB::FindPointsOnDevice(const Vector &point_pos,
|
||||
int point_pos_ordering) {};
|
||||
const int point_pos_ordering) {};
|
||||
void FindPointsGSLIB::InterpolateOnDevice(const Vector &field_in_evec,
|
||||
Vector &field_out,
|
||||
const int nel, const int ncomp,
|
||||
@@ -1094,7 +1094,8 @@ void FindPointsGSLIB::InterpolateOnDevice(const Vector &field_in_evec,
|
||||
#endif
|
||||
|
||||
void FindPointsGSLIB::FindPoints(Mesh &m, const Vector &point_pos,
|
||||
int point_pos_ordering, const double bb_t,
|
||||
const int point_pos_ordering,
|
||||
const double bb_t,
|
||||
const double newt_tol, const int npt_max)
|
||||
{
|
||||
if (!setupflag || (mesh != &m) )
|
||||
@@ -1105,16 +1106,28 @@ void FindPointsGSLIB::FindPoints(Mesh &m, const Vector &point_pos,
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::Interpolate(const Vector &point_pos,
|
||||
const GridFunction &field_in, Vector &field_out,
|
||||
int point_pos_ordering)
|
||||
const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
const int point_pos_ordering)
|
||||
{
|
||||
FindPoints(point_pos, point_pos_ordering);
|
||||
Interpolate(field_in, field_out);
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::Interpolate(const Vector &point_pos,
|
||||
const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
const int point_pos_ordering,
|
||||
const int field_out_ordering)
|
||||
{
|
||||
FindPoints(point_pos, point_pos_ordering);
|
||||
Interpolate(field_in, field_out, field_out_ordering);
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::Interpolate(Mesh &m, const Vector &point_pos,
|
||||
const GridFunction &field_in, Vector &field_out,
|
||||
int point_pos_ordering)
|
||||
const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
const int point_pos_ordering)
|
||||
{
|
||||
FindPoints(m, point_pos, point_pos_ordering);
|
||||
Interpolate(field_in, field_out);
|
||||
@@ -1470,7 +1483,7 @@ void FindPointsGSLIB::SetupSplitMeshesAndIntegrationRules(const int order)
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::GetNodalValues(const GridFunction *gf_in,
|
||||
Vector &node_vals)
|
||||
Vector &node_vals) const
|
||||
{
|
||||
const GridFunction *nodes = gf_in;
|
||||
const FiniteElementSpace *fes = nodes->FESpace();
|
||||
@@ -1758,6 +1771,13 @@ void FindPointsGSLIB::MapRefPosAndElemIndices()
|
||||
|
||||
void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
|
||||
Vector &field_out)
|
||||
{
|
||||
Interpolate(field_in, field_out, field_in.FESpace()->GetOrdering());
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
const int field_out_ordering)
|
||||
{
|
||||
const int gf_order = field_in.FESpace()->GetMaxElementOrder(),
|
||||
mesh_order = mesh->GetNodalFESpace()->GetMaxElementOrder();
|
||||
@@ -1800,7 +1820,7 @@ void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
|
||||
const int maxOrder = field_in.FESpace()->GetMaxElementOrder();
|
||||
|
||||
InterpolateOnDevice(node_vals, field_out, NE_split_total, ncomp,
|
||||
maxOrder+1, field_in.FESpace()->GetOrdering());
|
||||
maxOrder+1, field_out_ordering);
|
||||
return;
|
||||
#endif
|
||||
}
|
||||
@@ -1812,12 +1832,13 @@ void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
|
||||
field_in.FESpace()->IsVariableOrder() ==
|
||||
mesh->GetNodalFESpace()->IsVariableOrder())
|
||||
{
|
||||
InterpolateH1(field_in, field_out);
|
||||
InterpolateH1(field_in, field_out, field_out_ordering);
|
||||
return;
|
||||
}
|
||||
else
|
||||
{
|
||||
InterpolateGeneral(field_in, field_out);
|
||||
InterpolateGeneral(field_in, field_out,
|
||||
field_out_ordering);
|
||||
if (!fec_l2 || avgtype == AvgType::NONE) { return; }
|
||||
}
|
||||
|
||||
@@ -1861,11 +1882,11 @@ void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
|
||||
|
||||
if (gf_order_h1 == mesh_order) // basis is GaussLobatto by default
|
||||
{
|
||||
InterpolateH1(field_in_h1, field_out_l2);
|
||||
InterpolateH1(field_in_h1, field_out_l2, field_out_ordering);
|
||||
}
|
||||
else
|
||||
{
|
||||
InterpolateGeneral(field_in_h1, field_out_l2);
|
||||
InterpolateGeneral(field_in_h1, field_out_l2, field_out_ordering);
|
||||
}
|
||||
|
||||
// Copy interpolated values for the points on element border
|
||||
@@ -1873,7 +1894,7 @@ void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
|
||||
{
|
||||
for (int i = 0; i < indl2.Size(); i++)
|
||||
{
|
||||
int idx = field_in_h1.FESpace()->GetOrdering() == Ordering::byNODES?
|
||||
int idx = field_out_ordering == Ordering::byNODES?
|
||||
indl2[i] + j*points_cnt:
|
||||
indl2[i]*ncomp + j;
|
||||
field_out(idx) = field_out_l2(idx);
|
||||
@@ -1883,7 +1904,8 @@ void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::InterpolateH1(const GridFunction &field_in,
|
||||
Vector &field_out)
|
||||
Vector &field_out,
|
||||
const int field_out_ordering)
|
||||
{
|
||||
FiniteElementSpace ind_fes(mesh, field_in.FESpace()->FEColl());
|
||||
if (field_in.FESpace()->IsVariableOrder())
|
||||
@@ -1913,7 +1935,8 @@ void FindPointsGSLIB::InterpolateH1(const GridFunction &field_in,
|
||||
dataptrout = i*points_cnt;
|
||||
if (field_in.FESpace()->GetOrdering() == Ordering::byNODES)
|
||||
{
|
||||
field_in_scalar.NewDataAndSize(field_in.GetData()+dataptrin, points_fld);
|
||||
field_in_scalar.NewDataAndSize(field_in.GetData()+dataptrin,
|
||||
points_fld);
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -1945,7 +1968,7 @@ void FindPointsGSLIB::InterpolateH1(const GridFunction &field_in,
|
||||
(gslib::findpts_data_3 *)this->fdataD);
|
||||
}
|
||||
}
|
||||
if (field_in.FESpace()->GetOrdering() == Ordering::byVDIM)
|
||||
if (field_out_ordering == Ordering::byVDIM)
|
||||
{
|
||||
Vector field_out_temp = field_out;
|
||||
for (int i = 0; i < ncomp; i++)
|
||||
@@ -1959,7 +1982,8 @@ void FindPointsGSLIB::InterpolateH1(const GridFunction &field_in,
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::InterpolateGeneral(const GridFunction &field_in,
|
||||
Vector &field_out)
|
||||
Vector &field_out,
|
||||
const int field_out_ordering)
|
||||
{
|
||||
int ncomp = field_in.VectorDim(),
|
||||
nptorig = points_cnt,
|
||||
@@ -1979,7 +2003,7 @@ void FindPointsGSLIB::InterpolateGeneral(const GridFunction &field_in,
|
||||
if (dim == 3) { ip.z = gsl_mfem_ref(index*dim + 2); }
|
||||
Vector localval(ncomp);
|
||||
field_in.GetVectorValue(gsl_mfem_elem[index], ip, localval);
|
||||
if (field_in.FESpace()->GetOrdering() == Ordering::byNODES)
|
||||
if (field_out_ordering == Ordering::byNODES)
|
||||
{
|
||||
for (int i = 0; i < ncomp; i++)
|
||||
{
|
||||
@@ -2014,7 +2038,10 @@ void FindPointsGSLIB::InterpolateGeneral(const GridFunction &field_in,
|
||||
for (int index = 0; index < npt; index++)
|
||||
{
|
||||
if (gsl_code[index] == 2) { continue; }
|
||||
for (int d = 0; d < dim; ++d) { pt->r[d]= gsl_mfem_ref(index*dim + d); }
|
||||
for (int d = 0; d < dim; ++d)
|
||||
{
|
||||
pt->r[d]= gsl_mfem_ref(index*dim + d);
|
||||
}
|
||||
pt->index = index;
|
||||
pt->proc = gsl_proc[index];
|
||||
pt->el = gsl_mfem_elem[index];
|
||||
@@ -2104,7 +2131,7 @@ void FindPointsGSLIB::InterpolateGeneral(const GridFunction &field_in,
|
||||
sdpt = (struct send_pt *)sendpt->ptr;
|
||||
for (int index = 0; index < static_cast<int>(sendpt->n); index++)
|
||||
{
|
||||
int idx = field_in.FESpace()->GetOrdering() == Ordering::byNODES ?
|
||||
int idx = field_out_ordering == Ordering::byNODES ?
|
||||
sdpt->index + j*nptorig :
|
||||
sdpt->index*ncomp + j;
|
||||
field_out(idx) = sdpt->ival;
|
||||
@@ -2246,7 +2273,7 @@ void FindPointsGSLIB::DistributeInterpolatedValues(const Vector &int_vals,
|
||||
}
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::GetAxisAlignedBoundingBoxes(Vector &aabb)
|
||||
void FindPointsGSLIB::GetAxisAlignedBoundingBoxes(Vector &aabb) const
|
||||
{
|
||||
MFEM_VERIFY(setupflag, "Call FindPointsGSLIB::Setup method first");
|
||||
auto *findptsData3 = (gslib::findpts_data_3 *)this->fdataD;
|
||||
@@ -2317,7 +2344,7 @@ void FindPointsGSLIB::GetAxisAlignedBoundingBoxes(Vector &aabb)
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::GetOrientedBoundingBoxes(DenseTensor &obbA, Vector &obbC,
|
||||
Vector &obbV)
|
||||
Vector &obbV) const
|
||||
{
|
||||
MFEM_VERIFY(setupflag, "Call FindPointsGSLIB::Setup method first");
|
||||
auto *findptsData3 = (gslib::findpts_data_3 *)this->fdataD;
|
||||
@@ -2502,8 +2529,8 @@ void OversetFindPointsGSLIB::Setup(Mesh &m, const int meshid,
|
||||
}
|
||||
|
||||
void OversetFindPointsGSLIB::FindPoints(const Vector &point_pos,
|
||||
Array<unsigned int> &point_id,
|
||||
int point_pos_ordering)
|
||||
const Array<unsigned int> &point_id,
|
||||
const int point_pos_ordering)
|
||||
{
|
||||
MFEM_VERIFY(setupflag, "Use OversetFindPointsGSLIB::Setup before "
|
||||
"finding points.");
|
||||
@@ -2582,10 +2609,10 @@ void OversetFindPointsGSLIB::FindPoints(const Vector &point_pos,
|
||||
}
|
||||
|
||||
void OversetFindPointsGSLIB::Interpolate(const Vector &point_pos,
|
||||
Array<unsigned int> &point_id,
|
||||
const Array<unsigned int> &point_id,
|
||||
const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
int point_pos_ordering)
|
||||
const int point_pos_ordering)
|
||||
{
|
||||
FindPoints(point_pos, point_id, point_pos_ordering);
|
||||
Interpolate(field_in, field_out);
|
||||
|
||||
+32
-15
@@ -119,11 +119,13 @@ protected:
|
||||
} DEV;
|
||||
|
||||
/// Use GSLIB for communication and interpolation
|
||||
virtual void InterpolateH1(const GridFunction &field_in, Vector &field_out);
|
||||
virtual void InterpolateH1(const GridFunction &field_in, Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
/// Uses GSLIB Crystal Router for communication followed by MFEM's
|
||||
/// interpolation functions
|
||||
virtual void InterpolateGeneral(const GridFunction &field_in,
|
||||
Vector &field_out);
|
||||
Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
|
||||
/// Since GSLIB is designed to work with quads/hexes, we split every
|
||||
/// triangle/tet/prism/pyramid element into quads/hexes.
|
||||
@@ -140,7 +142,7 @@ protected:
|
||||
virtual void SetupSplitMeshesAndIntegrationRules(const int order);
|
||||
|
||||
/// Get GridFunction value at the points expected by GSLIB.
|
||||
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals);
|
||||
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals) const;
|
||||
|
||||
/// Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For simplices,
|
||||
/// find the original element number (that was split into micro quads/hexes)
|
||||
@@ -182,7 +184,7 @@ protected:
|
||||
These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
|
||||
byVDim: (XYZ,XYZ,....XYZ) specified by @a point_pos_ordering. */
|
||||
void FindPointsOnDevice(const Vector &point_pos,
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/** Interpolation of field values at prescribed reference space positions.
|
||||
@param[in] field_in_evec E-vector of grid function to be interpolated.
|
||||
@@ -253,10 +255,15 @@ public:
|
||||
#gsl_dist Distance between the sought and the found point
|
||||
in physical space. */
|
||||
void FindPoints(const Vector &point_pos,
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
/// Convenience function when point positions are in a ParticleVector
|
||||
void FindPoints(const ParticleVector &point_pos)
|
||||
{
|
||||
FindPoints(point_pos, point_pos.GetOrdering());
|
||||
}
|
||||
/// Setup FindPoints and search positions
|
||||
void FindPoints(Mesh &m, const Vector &point_pos,
|
||||
int point_pos_ordering = Ordering::byNODES,
|
||||
const int point_pos_ordering = Ordering::byNODES,
|
||||
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
@@ -266,20 +273,28 @@ public:
|
||||
\p field_in is in H1 and in the same space as the
|
||||
mesh that was given to Setup().
|
||||
@param[out] field_out Interpolated values. For points that are not found
|
||||
the value is set to #default_interp_value. */
|
||||
the value is set to #default_interp_value.
|
||||
The output ordering is determined from field_in.*/
|
||||
virtual void Interpolate(const GridFunction &field_in, Vector &field_out);
|
||||
/// Interpolation of field values, with output ordering specification.
|
||||
virtual void Interpolate(const GridFunction &field_in, Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
/** Search positions and interpolate. The ordering (byNODES or byVDIM) of
|
||||
the output values in \p field_out corresponds to the ordering used
|
||||
in the input GridFunction \p field_in. */
|
||||
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
/// Search positions and interpolate with given point and output ordering.
|
||||
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
|
||||
Vector &field_out, const int point_pos_ordering,
|
||||
const int field_out_ordering);
|
||||
/** Setup FindPoints, search positions and interpolate. The ordering (byNODES
|
||||
or byVDIM) of the output values in \p field_out corresponds to the
|
||||
ordering used in the input GridFunction \p field_in. */
|
||||
void Interpolate(Mesh &m, const Vector &point_pos,
|
||||
const GridFunction &field_in, Vector &field_out,
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/// Average type to be used for L2 functions in-case a point is located at
|
||||
/// an element boundary where the function might be multi-valued.
|
||||
@@ -376,7 +391,7 @@ public:
|
||||
/// The size of the returned vector is (nel x nverts x dim), where nel is the
|
||||
/// number of elements (after splitting for simplcies), nverts is number of
|
||||
/// vertices (4 in 2D, 8 in 3D), and dim is the spatial dimension.
|
||||
void GetAxisAlignedBoundingBoxes(Vector &aabb);
|
||||
void GetAxisAlignedBoundingBoxes(Vector &aabb) const;
|
||||
|
||||
/// Return the oriented bounding boxes (OBB) computed during \ref Setup.
|
||||
/// Each OBB is represented using the inverse transformation (A^{-1}) and
|
||||
@@ -386,7 +401,8 @@ public:
|
||||
/// size (dim x dim x nel), and the OBB centers are returned in \p obbC,
|
||||
/// a vector of size (nel x dim). The vertices of the OBBs are returned in
|
||||
/// \p obbV, a vector of size (nel x nverts x dim) .
|
||||
void GetOrientedBoundingBoxes(DenseTensor &obbA, Vector &obbC, Vector &obbV);
|
||||
void GetOrientedBoundingBoxes(DenseTensor &obbA, Vector &obbC,
|
||||
Vector &obbV) const;
|
||||
};
|
||||
|
||||
/** \brief OversetFindPointsGSLIB enables use of findpts for arbitrary number of
|
||||
@@ -446,13 +462,14 @@ public:
|
||||
byNodes: (XXX...,YYY...,ZZZ) or
|
||||
byVDim: (XYZ,XYZ,....XYZ) */
|
||||
void FindPoints(const Vector &point_pos,
|
||||
Array<unsigned int> &point_id,
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
const Array<unsigned int> &point_id,
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/** Search positions and interpolate */
|
||||
void Interpolate(const Vector &point_pos, Array<unsigned int> &point_id,
|
||||
void Interpolate(const Vector &point_pos,
|
||||
const Array<unsigned int> &point_id,
|
||||
const GridFunction &field_in, Vector &field_out,
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
using FindPointsGSLIB::Interpolate;
|
||||
};
|
||||
|
||||
|
||||
@@ -789,7 +789,6 @@ void Hybridization::ComputeH()
|
||||
}
|
||||
else
|
||||
{
|
||||
// TODO: add ones on the diagonal of zero rows
|
||||
V->Finalize();
|
||||
Array<HYPRE_BigInt> V_J(V->NumNonZeroElems());
|
||||
MFEM_ASSERT(c_pfes, "");
|
||||
@@ -823,6 +822,13 @@ void Hybridization::ComputeH()
|
||||
MFEM_VERIFY(pH.Type() != Operator::PETSC_MATIS, "To be implemented");
|
||||
pH.MakePtAP(plpH, pP);
|
||||
delete lpH;
|
||||
|
||||
HypreParMatrix *hH = pH.As<HypreParMatrix>();
|
||||
MFEM_ASSERT(hH, "");
|
||||
|
||||
SparseMatrix H_diag;
|
||||
hH->GetDiag(H_diag);
|
||||
H_diag.SetDiagIdentity();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
+455
-275
File diff suppressed because it is too large
Load Diff
@@ -14,8 +14,11 @@
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#include "../general/array.hpp"
|
||||
#include "../linalg/operator.hpp"
|
||||
#include "../linalg/vector.hpp"
|
||||
|
||||
#include <memory>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
@@ -45,15 +48,30 @@ protected:
|
||||
Array<int> hat_dof_gather_map;
|
||||
Array<DofType> hat_dof_marker;
|
||||
|
||||
Array<int> el_to_face;
|
||||
Array<int> face_to_el;
|
||||
Array<int> el_to_face; ///< Element to face connectivity.
|
||||
Array<int> el_face_offsets; ///< Per-element offsets into @a el_to_face.
|
||||
Array<int> face_to_el; ///< Face-to-element connectivity.
|
||||
Array<int> face_face_offsets; ///< Face-to-face offsets.
|
||||
|
||||
int n_el_face; ///< Total number of element-to-face connections.
|
||||
int n_face_face; ///< Total number of face-to-face connections.
|
||||
|
||||
Vector Ct_mat; ///< Constraint matrix (transposed) stored element-wise.
|
||||
|
||||
/// @name For parallel non-conforming meshes
|
||||
///@{
|
||||
std::unique_ptr<Operator> P_pc; ///< Partially conforming prolongation.
|
||||
std::unique_ptr<Operator> P_nbr; ///< Face-neighbor prolongation.
|
||||
///@}
|
||||
|
||||
Array<int> idofs, bdofs;
|
||||
|
||||
Vector Ahat, Ahat_ii, Ahat_ib, Ahat_bi, Ahat_bb;
|
||||
Array<int> Ahat_ii_piv, Ahat_bb_piv;
|
||||
|
||||
/// Return the (partially) conforming prolongation on the constraint space.
|
||||
const Operator &GetProlongation() const;
|
||||
|
||||
public:
|
||||
/// Construct the constraint matrix.
|
||||
void ConstructC();
|
||||
|
||||
@@ -1004,13 +1004,16 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
const int max_d1d = T_D1D ? T_D1D : DeviceDofQuadLimits::Get().MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= max_d1d, "");
|
||||
MFEM_VERIFY(Q1D <= max_q1d, "");
|
||||
auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
auto g = Reshape(g_.Read(), Q1D, D1D);
|
||||
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, symmetric ? 6 : 9, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
|
||||
const auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
const auto g = Reshape(g_.Read(), Q1D, D1D);
|
||||
const auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, symmetric ? 6 : 9, NE);
|
||||
const auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
MFEM_VERIFY(D1D <= Q1D, "THREAD_DIRECT requires D1D <= Q1D");
|
||||
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
|
||||
mfem::forall_3D<T_Q1D*T_Q1D*T_Q1D>(NE,
|
||||
Q1D, Q1D, Q1D,
|
||||
[=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
@@ -1133,11 +1133,11 @@ inline void SmemPAMassApply3D(const int NE,
|
||||
const int max_d1d = T_D1D ? T_D1D : DeviceDofQuadLimits::Get().MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= max_d1d, "");
|
||||
MFEM_VERIFY(Q1D <= max_q1d, "");
|
||||
auto b = b_.Read();
|
||||
auto d = d_.Read();
|
||||
auto x = x_.Read();
|
||||
const auto b = b_.Read();
|
||||
const auto d = d_.Read();
|
||||
const auto x = x_.Read();
|
||||
auto y = y_.ReadWrite();
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
internal::SmemPAMassApply3D_Element<T_D1D,T_Q1D>(e, NE, b, d, x, y, d1d, q1d);
|
||||
});
|
||||
@@ -1156,8 +1156,8 @@ inline void EAMassAssemble1D(const int NE,
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
const auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
const auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
auto M = Reshape(add ? eadata.ReadWrite() : eadata.Write(), D1D, D1D, NE);
|
||||
mfem::forall_2D(NE, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
|
||||
@@ -28,7 +28,7 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
|
||||
const FaceType ftype = FaceType::Interior;
|
||||
const int nf = mesh.GetNFbyType(ftype);
|
||||
|
||||
const Geometry::Type geom = mesh.GetFaceGeometry(0);
|
||||
const Geometry::Type geom = mesh.GetTypicalFaceGeometry();
|
||||
const int trial_order = trial_fes.GetMaxElementOrder();
|
||||
const int test_order = test_fes.GetMaxElementOrder();
|
||||
const int qorder = test_order + trial_order - 1;
|
||||
@@ -47,7 +47,7 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
|
||||
});
|
||||
}
|
||||
|
||||
const FiniteElement &trial_face_el = *trial_fes.GetFaceElement(0);
|
||||
const FiniteElement &trial_face_el = *trial_fes.GetTypicalTraceElement();
|
||||
const auto maps = &trial_face_el.GetDofToQuad(ir, DofToQuad::TENSOR);
|
||||
const int ndof_face = trial_face_el.GetDof();
|
||||
|
||||
@@ -72,7 +72,7 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
|
||||
const FiniteElement &test_el = *test_fes.GetFE(0);
|
||||
const FiniteElement &test_el = *test_fes.GetTypicalFE();
|
||||
const int n_faces_per_el = 2*dim; // assuming tensor product
|
||||
// Get all the local face maps (mapping from lexicographic face index to
|
||||
// lexicographic volume index, depending on the local face index).
|
||||
@@ -90,10 +90,10 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
|
||||
Array<int> face_info(nf * 4);
|
||||
{
|
||||
int fidx = 0;
|
||||
for (int f = 0; f < mesh.GetNumFaces(); ++f)
|
||||
for (int f = 0; f < mesh.GetNumFacesWithGhost(); ++f)
|
||||
{
|
||||
Mesh::FaceInformation finfo = mesh.GetFaceInformation(f);
|
||||
if (!finfo.IsInterior()) { continue; }
|
||||
if (!finfo.IsInterior() || finfo.IsNonconformingCoarse()) { continue; }
|
||||
face_info[0 + fidx*4] = finfo.element[0].local_face_id;
|
||||
face_info[1 + fidx*4] = finfo.element[0].orientation;
|
||||
face_info[2 + fidx*4] = finfo.element[1].local_face_id;
|
||||
@@ -114,7 +114,7 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
|
||||
else
|
||||
{
|
||||
d_emat = emat.Write();
|
||||
mfem::forall(emat.Size(), [=] MFEM_HOST_DEVICE (int i) { d_emat[i] = 0.0; });
|
||||
emat = 0.0; // Will execute on device, since Write() sets the device flag
|
||||
}
|
||||
|
||||
const auto face_mats = Reshape(mass_emat.Read(), ndof_face, ndof_face, nf);
|
||||
@@ -133,26 +133,104 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
|
||||
}
|
||||
};
|
||||
|
||||
mfem::forall_3D(nf, ndof_face, ndof_face, 2, [=] MFEM_HOST_DEVICE (int f)
|
||||
auto permute_face_2 = [=] MFEM_HOST_DEVICE(int local_face_1, int local_face_2,
|
||||
int orient, int size1d, int index)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(el_i, z, 2)
|
||||
if (dim == 2)
|
||||
{
|
||||
const int lf_i = d_face_info(0, el_i, f);
|
||||
const int orient = d_face_info(1, el_i, f);
|
||||
// Loop over face indices in "native ordering"
|
||||
MFEM_FOREACH_THREAD(i_lex, x, ndof_face)
|
||||
return internal::PermuteFace2D(local_face_1, local_face_2, orient,
|
||||
size1d, index);
|
||||
}
|
||||
else // dim == 3
|
||||
{
|
||||
return internal::PermuteFace3D(local_face_1, local_face_2, orient,
|
||||
size1d, index);
|
||||
}
|
||||
};
|
||||
|
||||
if (mesh.Conforming())
|
||||
{
|
||||
mfem::forall_3D(nf, ndof_face, ndof_face, 2, [=] MFEM_HOST_DEVICE (int f)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(el_i, z, 2)
|
||||
{
|
||||
// Convert to lexicographic relative to the face itself
|
||||
const int i_face = permute_face(lf_i, orient, d1d, i_lex);
|
||||
// Convert from lexicographic face DOF to volume DOF
|
||||
const int i = d_face_maps(i_lex, lf_i);
|
||||
MFEM_FOREACH_THREAD(j, y, ndof_face)
|
||||
const int lf_i = d_face_info(0, el_i, f);
|
||||
const int orient = d_face_info(1, el_i, f);
|
||||
// Loop over face indices in "native ordering"
|
||||
MFEM_FOREACH_THREAD(i_lex, x, ndof_face)
|
||||
{
|
||||
el_mats(i, j, el_i, f) += face_mats(i_face, j, f);
|
||||
// Convert to lexicographic relative to the face itself
|
||||
const int i_face = permute_face(lf_i, orient, d1d, i_lex);
|
||||
// Convert from lexicographic face DOF to volume DOF
|
||||
const int i = d_face_maps(i_lex, lf_i);
|
||||
MFEM_FOREACH_THREAD(j, y, ndof_face)
|
||||
{
|
||||
el_mats(i, j, el_i, f) += face_mats(i_face, j, f);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
const InterpolationManager &interp =
|
||||
test_fes.GetInterpolationManager(ElementDofOrdering::LEXICOGRAPHIC, ftype);
|
||||
|
||||
auto interp_configs = interp.GetFaceInterpConfig().Read();
|
||||
const int nc_size = interp.GetNumInterpolators();
|
||||
auto d_interp = Reshape(interp.GetInterpolators().Read(),
|
||||
ndof_face, ndof_face, nc_size);
|
||||
|
||||
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int f)
|
||||
{
|
||||
const InterpConfig conf = interp_configs[f];
|
||||
const int master_side = conf.master_side;
|
||||
const int interp_index = conf.index;
|
||||
|
||||
const int lf_0 = d_face_info(0, 0, f);
|
||||
|
||||
for (int el_i = 0; el_i < 2; ++el_i)
|
||||
{
|
||||
const int lf_i = d_face_info(0, el_i, f);
|
||||
const int orient = d_face_info(1, el_i, f);
|
||||
|
||||
for (int j = 0; j < ndof_face; j++)
|
||||
{
|
||||
for (int i_lex = 0; i_lex < ndof_face; i_lex++)
|
||||
{
|
||||
real_t val = 0.0;
|
||||
if (conf.is_non_conforming && el_i == master_side)
|
||||
{
|
||||
// Interpolate from el_i (coarse element) to the fine face.
|
||||
// The mapping is given by d_interp, which uses indices
|
||||
// relative to element 0.
|
||||
|
||||
// i0 is lexicographic relative to element 0
|
||||
const int i0 = permute_face_2(lf_i, lf_0, orient, d1d, i_lex);
|
||||
|
||||
// k0 is lexicographic relative to element 0
|
||||
for (int k0 = 0; k0 < ndof_face; k0++)
|
||||
{
|
||||
// k is relative to the face itself
|
||||
const int k = permute_face(lf_0, orient, d1d, k0);
|
||||
val += d_interp(k0, i0, interp_index)
|
||||
* face_mats(k, j, f);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// Convert to lexicographic relative to the face itself
|
||||
const int i_face = permute_face(lf_i, orient, d1d, i_lex);
|
||||
val = face_mats(i_face, j, f);
|
||||
}
|
||||
// Convert from lexicographic face DOF to volume DOF
|
||||
const int i = d_face_maps(i_lex, lf_i);
|
||||
el_mats(i, j, el_i, f) += val;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -54,7 +54,7 @@ void SmemPAVectorDiffusionApply2D(const int NE,
|
||||
const auto XE = Reshape(x.Read(), D1D, D1D, SDIM, NE);
|
||||
auto YE = Reshape(y.ReadWrite(), D1D, D1D, SDIM, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
|
||||
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
|
||||
@@ -120,7 +120,7 @@ void SmemPAVectorDiffusionApply3D(const int NE,
|
||||
const auto XE = Reshape(x.Read(), D1D, D1D, D1D, SDIM, NE);
|
||||
auto YE = Reshape(y.ReadWrite(), D1D, D1D, D1D, SDIM, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
|
||||
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
|
||||
|
||||
@@ -51,7 +51,7 @@ void SmemPAVectorMassApply2D(const int NE,
|
||||
const auto X = Reshape(x.Read(), D1D, D1D, VDIM, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
|
||||
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
|
||||
@@ -119,7 +119,7 @@ void SmemPAVectorMassApply3D(const int NE,
|
||||
const auto X = Reshape(x.Read(), D1D, D1D, D1D, VDIM, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
|
||||
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
|
||||
|
||||
+3
-31
@@ -14,6 +14,7 @@
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#include "kernel_reporter.hpp"
|
||||
#include "../general/hash_util.hpp"
|
||||
#include <unordered_map>
|
||||
#include <tuple>
|
||||
#include <type_traits>
|
||||
@@ -86,35 +87,6 @@ namespace mfem
|
||||
} \
|
||||
}
|
||||
|
||||
/// @brief Hashes variadic packs for which each type contained in the variadic
|
||||
/// pack has a specialization of `std::hash` available.
|
||||
///
|
||||
/// For example, packs containing int, bool, enum values, etc.
|
||||
template<typename ...KernelParameters>
|
||||
struct KernelDispatchKeyHash
|
||||
{
|
||||
private:
|
||||
template<int N>
|
||||
size_t operator()(std::tuple<KernelParameters...> value) const { return 0; }
|
||||
|
||||
// The hashing formula here is taken directly from the Boost library, with
|
||||
// the magic number 0x9e3779b9 chosen to minimize hashing collisions.
|
||||
template<std::size_t N, typename THead, typename... TTail>
|
||||
size_t operator()(std::tuple<KernelParameters...> value) const
|
||||
{
|
||||
constexpr int Index = N - sizeof...(TTail) - 1;
|
||||
auto lhs_hash = std::hash<THead>()(std::get<Index>(value));
|
||||
auto rhs_hash = operator()<N, TTail...>(value);
|
||||
return lhs_hash^(rhs_hash + 0x9e3779b9 + (lhs_hash<<6) + (lhs_hash>>2));
|
||||
}
|
||||
public:
|
||||
/// Returns the hash of the given @a value.
|
||||
size_t operator()(std::tuple<KernelParameters...> value) const
|
||||
{
|
||||
return operator()<sizeof...(KernelParameters),KernelParameters...>(value);
|
||||
}
|
||||
};
|
||||
|
||||
namespace internal { template<typename... Types> struct KernelTypeList { }; }
|
||||
|
||||
template<typename... T> class KernelDispatchTable { };
|
||||
@@ -128,8 +100,8 @@ class KernelDispatchTable<Kernels,
|
||||
internal::KernelTypeList<Params...>,
|
||||
internal::KernelTypeList<OptParams...>>
|
||||
{
|
||||
using TableType = std::unordered_map<std::tuple<Params...>,
|
||||
Signature, KernelDispatchKeyHash<Params...>>;
|
||||
using TableType =
|
||||
std::unordered_map<std::tuple<Params...>, Signature, TupleHasher>;
|
||||
TableType table;
|
||||
|
||||
/// @brief Call function @a f with arguments @a args (perfect forwaring).
|
||||
|
||||
@@ -1054,154 +1054,7 @@ void DGElasticityDirichletLFIntegrator::AssembleRHSElementVect(
|
||||
}
|
||||
}
|
||||
|
||||
void SlidingElasticityLFIntegrator::AssembleRHSElementVect(
|
||||
const FiniteElement &el, ElementTransformation &Tr, Vector &elvect)
|
||||
{
|
||||
mfem_error("SlidingElasticityLFIntegrator::AssembleRHSElementVect");
|
||||
}
|
||||
|
||||
void SlidingElasticityLFIntegrator::AssembleRHSElementVect(
|
||||
const FiniteElement &el, FaceElementTransformations &Tr, Vector &elvect)
|
||||
{
|
||||
MFEM_ASSERT(Tr.Elem2No < 0, "interior boundary is not supported");
|
||||
|
||||
#ifdef MFEM_THREAD_SAFE
|
||||
Vector shape;
|
||||
DenseMatrix dshape;
|
||||
DenseMatrix adjJ;
|
||||
DenseMatrix dshape_ps;
|
||||
Vector nor;
|
||||
Vector dshape_dn;
|
||||
Vector dshape_du;
|
||||
real_t g_val;
|
||||
Vector nt_val;
|
||||
#endif
|
||||
|
||||
const int dim = el.GetDim();
|
||||
const int ndofs = el.GetDof();
|
||||
const int nvdofs = dim*ndofs;
|
||||
|
||||
elvect.SetSize(nvdofs);
|
||||
elvect = 0.0;
|
||||
|
||||
adjJ.SetSize(dim);
|
||||
shape.SetSize(ndofs);
|
||||
dshape.SetSize(ndofs, dim);
|
||||
dshape_ps.SetSize(ndofs, dim);
|
||||
nor.SetSize(dim);
|
||||
dshape_dn.SetSize(ndofs);
|
||||
dshape_du.SetSize(ndofs);
|
||||
nt_val.SetSize(dim);
|
||||
|
||||
const IntegrationRule *ir = IntRule;
|
||||
if (ir == NULL)
|
||||
{
|
||||
const int order = 2*el.GetOrder(); // <-----
|
||||
ir = &IntRules.Get(Tr.GetGeometryType(), order);
|
||||
}
|
||||
|
||||
for (int pi = 0; pi < ir->GetNPoints(); ++pi)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(pi);
|
||||
|
||||
// Set the integration point in the face and the neighboring element
|
||||
Tr.SetAllIntPoints(&ip);
|
||||
|
||||
// Access the neighboring element's integration point
|
||||
const IntegrationPoint &eip = Tr.GetElement1IntPoint();
|
||||
|
||||
el.CalcShape(eip, shape);
|
||||
el.CalcDShape(eip, dshape);
|
||||
|
||||
CalcAdjugate(Tr.Elem1->Jacobian(), adjJ);
|
||||
Mult(dshape, adjJ, dshape_ps);
|
||||
|
||||
if (dim == 1)
|
||||
{
|
||||
nor(0) = 2*eip.x - 1.0;
|
||||
}
|
||||
else
|
||||
{
|
||||
CalcOrtho(Tr.Jacobian(), nor);
|
||||
}
|
||||
|
||||
if (!nt)
|
||||
{
|
||||
// Set nt to the unit normal vector if not provided
|
||||
nt_val = nor;
|
||||
nt_val /= nt_val.Norml2();
|
||||
}
|
||||
else
|
||||
{
|
||||
// Evaluate the vector field using the face transformation.
|
||||
nt->Eval(nt_val, Tr, ip);
|
||||
}
|
||||
|
||||
// Evaluate the Dirichlet b.c. using the face transformation.
|
||||
g_val = g->Eval(Tr, ip);
|
||||
|
||||
real_t WL, WM, jcoef;
|
||||
{
|
||||
const real_t W = ip.weight / Tr.Elem1->Weight();
|
||||
WL = W * lambda->Eval(*Tr.Elem1, eip);
|
||||
WM = W * mu->Eval(*Tr.Elem1, eip);
|
||||
jcoef = kappa * (WL + 2.0*WM) * (nor*nor);
|
||||
dshape_ps.Mult(nor, dshape_dn);
|
||||
dshape_ps.Mult(nt_val, dshape_du);
|
||||
}
|
||||
|
||||
// alpha < g, (lambda div(v) I + mu (grad(v) + grad(v)^T)) n . ñ > +
|
||||
// + kappa < h^{-1} (lambda + 2 mu) g, v . ñ >
|
||||
|
||||
// i = idof + ndofs * im
|
||||
// v_phi(i,d) = delta(im,d) phi(idof)
|
||||
// div(v_phi(i)) = dphi(idof,im)
|
||||
// (grad(v_phi(i)))(k,l) = delta(im,k) dphi(idof,l)
|
||||
//
|
||||
// term 1:
|
||||
// alpha < g, lambda div(v_phi(i)) n . ñ > =
|
||||
// alpha lambda g div(v_phi(i)) (n.ñ) =
|
||||
// alpha lambda g dphi(idof,im) (n.ñ) --> quadrature -->
|
||||
// ip.weight/det(J1) alpha lambda g (nor.ñ) dshape_ps(idof,im) =
|
||||
// alpha * WL * g_val * (nor*nt_val) * dshape_ps(idof,im)
|
||||
// term 2:
|
||||
// alpha < g, mu grad(v_phi(i)) n . ñ > =
|
||||
// alpha mu g ñ^T grad(v_phi(i)) n =
|
||||
// alpha mu g ñ(k) delta(im,k) dphi(idof,l) n(l) =
|
||||
// alpha mu g ñ(im) dphi(idof,l) n(l) --> quadrature -->
|
||||
// ip.weight/det(J1) alpha mu ñ(im) g dshape_ps(idof,l) nor(l) =
|
||||
// alpha * WM * g_val * nt_val(im) * dshape_dn(idof)
|
||||
// term 3:
|
||||
// alpha < g, mu (grad(v_phi(i)))^T n . ñ > =
|
||||
// alpha mu g n^T grad(v_phi(i)) ñ =
|
||||
// alpha mu g n(k) delta(im,k) dphi(idof,l) ñ(l) =
|
||||
// alpha mu g n(im) dphi(idof,l) ñ(l) --> quadrature -->
|
||||
// ip.weight/det(J1) alpha mu g nor(im) dshape_ps(idof,l) ñ(l) =
|
||||
// alpha * WM * g_val * nor(im) * dshape_du(idof)
|
||||
// term j:
|
||||
// < kappa h^{-1} (lambda + 2 mu) g, ñ . v_phi(i) > =
|
||||
// kappa/h (lambda + 2 mu) g ñ(k) v_phi(i,k) =
|
||||
// kappa/h (lambda + 2 mu) g ñ(k) delta(im,k) phi(idof) =
|
||||
// kappa/h (lambda + 2 mu) g ñ(im) phi(idof) --> quadrature -->
|
||||
// [ 1/h = |nor|/det(J1) ]
|
||||
// ip.weight/det(J1) |nor|^2 (lambda + 2 mu) kappa g ñ(im) phi(idof) =
|
||||
// jcoef * g_val * nt_val(im) * shape(idof)
|
||||
|
||||
WM *= alpha;
|
||||
const real_t t1 = alpha * WL * g_val * (nor*nt_val);
|
||||
for (int im = 0, i = 0; im < dim; ++im)
|
||||
{
|
||||
const real_t t2 = WM * g_val * nt_val(im);
|
||||
const real_t t3 = WM * g_val * nor(im);
|
||||
const real_t tj = jcoef * g_val * nt_val(im);
|
||||
for (int idof = 0; idof < ndofs; ++idof, ++i)
|
||||
{
|
||||
elvect(i) += (t1*dshape_ps(idof,im) + t2*dshape_dn(idof) +
|
||||
t3*dshape_du(idof) + tj*shape(idof));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void WhiteGaussianNoiseDomainLFIntegrator::AssembleRHSElementVect
|
||||
(const FiniteElement &el,
|
||||
|
||||
@@ -646,62 +646,6 @@ public:
|
||||
using LinearFormIntegrator::AssembleRHSElementVect;
|
||||
};
|
||||
|
||||
/** Boundary linear form integrator for imposing non-zero Dirichlet boundary
|
||||
conditions, in a Nitsche elasticity formulation. Specifically, the linear
|
||||
form is given by
|
||||
$$
|
||||
\begin{split}
|
||||
b(v) &:= \alpha \int_\Gamma (\lambda\, \mathrm{div}(v)\, I + \mu (\nabla v
|
||||
+ \nabla v^{\mathrm{T}}))\, n \cdot \tilde{n}\, g\, dS + \kappa \int_\Gamma
|
||||
h^{-1} (\lambda + 2\mu) (v \cdot \tilde{n})\, g\, dS
|
||||
\end{split}
|
||||
$$
|
||||
where $g$ is the given Dirichlet data, $n$ is the unit normal, $\tilde{n}$ is
|
||||
a unit vector field, and $\alpha = \pm 1$, $\kappa > 0$ are the Nitsche
|
||||
parameters. The parameters $\lambda$ and $\mu$ should match the parameters
|
||||
with the same names used in the bilinear form integrator,
|
||||
SlidingElasticityIntegrator.
|
||||
*/
|
||||
class SlidingElasticityLFIntegrator : public LinearFormIntegrator
|
||||
{
|
||||
protected:
|
||||
Coefficient *g;
|
||||
VectorCoefficient *nt;
|
||||
Coefficient *lambda, *mu;
|
||||
real_t alpha, kappa;
|
||||
|
||||
#ifndef MFEM_THREAD_SAFE
|
||||
Vector shape;
|
||||
DenseMatrix dshape;
|
||||
DenseMatrix adjJ;
|
||||
DenseMatrix dshape_ps;
|
||||
Vector nor;
|
||||
Vector dshape_dn;
|
||||
Vector dshape_du;
|
||||
real_t g_val;
|
||||
Vector nt_val;
|
||||
#endif
|
||||
|
||||
public:
|
||||
SlidingElasticityLFIntegrator(Coefficient &g_,
|
||||
Coefficient &lambda_, Coefficient &mu_,
|
||||
real_t kappa_)
|
||||
: g(&g_), nt(NULL), lambda(&lambda_), mu(&mu_), alpha(-1.0), kappa(kappa_) {}
|
||||
|
||||
SlidingElasticityLFIntegrator(Coefficient &g_, VectorCoefficient &nt_,
|
||||
Coefficient &lambda_, Coefficient &mu_,
|
||||
real_t alpha_, real_t kappa_)
|
||||
: g(&g_), nt(&nt_), lambda(&lambda_), mu(&mu_), alpha(alpha_), kappa(kappa_) {}
|
||||
|
||||
void AssembleRHSElementVect(const FiniteElement &el,
|
||||
ElementTransformation &Tr,
|
||||
Vector &elvect) override;
|
||||
void AssembleRHSElementVect(const FiniteElement &el,
|
||||
FaceElementTransformations &Tr,
|
||||
Vector &elvect) override;
|
||||
|
||||
using LinearFormIntegrator::AssembleRHSElementVect;
|
||||
};
|
||||
|
||||
/** Class for spatial white Gaussian noise integration.
|
||||
|
||||
|
||||
@@ -488,10 +488,16 @@ void ParBilinearForm::FormLinearSystem(
|
||||
R.Mult(x, true_X);
|
||||
|
||||
FormSystemMatrix(ess_tdof_list, A);
|
||||
ConstrainedOperator *A_constrained;
|
||||
Operator::FormConstrainedSystemOperator(ess_tdof_list, A_constrained);
|
||||
|
||||
std::unique_ptr<ConstrainedOperator> A_constrained([&]()
|
||||
{
|
||||
Operator *op;
|
||||
Operator::FormSystemOperator(ess_tdof_list, op);
|
||||
return dynamic_cast<ConstrainedOperator*>(op);
|
||||
}());
|
||||
MFEM_ASSERT(A_constrained != nullptr, "");
|
||||
|
||||
A_constrained->EliminateRHS(true_X, true_B);
|
||||
delete A_constrained;
|
||||
R.MultTranspose(true_B, b);
|
||||
hybridization->ReduceRHS(true_B, B);
|
||||
X.SetSize(B.Size());
|
||||
|
||||
+8
-9
@@ -646,39 +646,38 @@ const FaceRestriction *ParFiniteElementSpace::GetFaceRestriction(
|
||||
auto itr = L2F.find(key);
|
||||
if (itr != L2F.end())
|
||||
{
|
||||
return itr->second;
|
||||
return itr->second.get();
|
||||
}
|
||||
else
|
||||
{
|
||||
FaceRestriction *res;
|
||||
std::unique_ptr<FaceRestriction> res;
|
||||
if (is_dg_space)
|
||||
{
|
||||
if (Conforming())
|
||||
{
|
||||
res = new ParL2FaceRestriction(*this, f_ordering, type, m);
|
||||
res.reset(new ParL2FaceRestriction(*this, f_ordering, type, m));
|
||||
}
|
||||
else
|
||||
{
|
||||
res = new ParNCL2FaceRestriction(*this, f_ordering, type, m);
|
||||
res.reset(new ParNCL2FaceRestriction(*this, f_ordering, type, m));
|
||||
}
|
||||
}
|
||||
else if (dynamic_cast<const DG_Interface_FECollection*>(fec))
|
||||
{
|
||||
res = new L2InterfaceFaceRestriction(*this, f_ordering, type);
|
||||
res.reset(new L2InterfaceFaceRestriction(*this, f_ordering, type));
|
||||
}
|
||||
else
|
||||
{
|
||||
if (Conforming())
|
||||
{
|
||||
res = new ConformingFaceRestriction(*this, f_ordering, type);
|
||||
res.reset(new ConformingFaceRestriction(*this, f_ordering, type));
|
||||
}
|
||||
else
|
||||
{
|
||||
res = new ParNCH1FaceRestriction(*this, f_ordering, type);
|
||||
res.reset(new ParNCH1FaceRestriction(*this, f_ordering, type));
|
||||
}
|
||||
}
|
||||
L2F[key] = res;
|
||||
return res;
|
||||
return L2F.emplace(key, std::move(res)).first->second.get();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -483,6 +483,8 @@ public:
|
||||
const FiniteElement *GetFaceNbrFaceFE(int i) const;
|
||||
const Array<HYPRE_BigInt> &GetFaceNbrGlobalDofMapArray() { return face_nbr_glob_dof_map; }
|
||||
const HYPRE_BigInt *GetFaceNbrGlobalDofMap() { return face_nbr_glob_dof_map; }
|
||||
const Array<HYPRE_BigInt> &GetFaceNbrGlobalDofMapArray() const
|
||||
{ return face_nbr_glob_dof_map; }
|
||||
ElementTransformation *GetFaceNbrElementTransformation(int i) const
|
||||
{ return pmesh->GetFaceNbrElementTransformation(i); }
|
||||
|
||||
|
||||
@@ -994,7 +994,6 @@ void ParNCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
|
||||
{
|
||||
if ( face.IsConforming() )
|
||||
{
|
||||
interpolations.RegisterFaceConformingInterpolation(face,f_ind);
|
||||
SetFaceDofsScatterIndices1(face,f_ind);
|
||||
if ( m==L2FaceValues::DoubleValued )
|
||||
{
|
||||
@@ -1010,7 +1009,6 @@ void ParNCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
|
||||
}
|
||||
else // Non-conforming face
|
||||
{
|
||||
interpolations.RegisterFaceCoarseToFineInterpolation(face,f_ind);
|
||||
SetFaceDofsScatterIndices1(face,f_ind);
|
||||
if ( m==L2FaceValues::DoubleValued )
|
||||
{
|
||||
@@ -1028,7 +1026,6 @@ void ParNCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
|
||||
}
|
||||
else if (type==FaceType::Boundary && face.IsBoundary())
|
||||
{
|
||||
interpolations.RegisterFaceConformingInterpolation(face,f_ind);
|
||||
SetFaceDofsScatterIndices1(face,f_ind);
|
||||
if ( m==L2FaceValues::DoubleValued )
|
||||
{
|
||||
@@ -1046,10 +1043,6 @@ void ParNCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
|
||||
{
|
||||
gather_offsets[i] += gather_offsets[i - 1];
|
||||
}
|
||||
|
||||
// Transform the interpolation matrix map into a contiguous memory structure.
|
||||
interpolations.LinearizeInterpolatorMapIntoVector();
|
||||
interpolations.InitializeNCInterpConfig();
|
||||
}
|
||||
|
||||
void ParNCL2FaceRestriction::ComputeGatherIndices()
|
||||
|
||||
@@ -326,9 +326,7 @@ public:
|
||||
@param[in] keep_nbr_block When set to true the SparseMatrix will
|
||||
include the rows (in addition to the columns)
|
||||
corresponding to face-neighbor dofs. The
|
||||
default behavior is to disregard those rows.
|
||||
|
||||
@warning This method is not implemented yet. */
|
||||
default behavior is to disregard those rows. */
|
||||
void FillI(SparseMatrix &mat,
|
||||
const bool keep_nbr_block = false) const override;
|
||||
|
||||
@@ -364,9 +362,7 @@ public:
|
||||
@param[in] keep_nbr_block When set to true the SparseMatrix will
|
||||
include the rows (in addition to the columns)
|
||||
corresponding to face-neighbor dofs. The
|
||||
default behavior is to disregard those rows.
|
||||
|
||||
@warning This method is not implemented yet. */
|
||||
default behavior is to disregard those rows. */
|
||||
void FillJAndData(const Vector &fea_data,
|
||||
SparseMatrix &mat,
|
||||
const bool keep_nbr_block = false) const override;
|
||||
|
||||
+7
-1
@@ -50,7 +50,13 @@ QuadratureInterpolator::DetKernelType
|
||||
QuadratureInterpolator::DetKernels::Fallback(
|
||||
int DIM, int SDIM, int D1D, int Q1D)
|
||||
{
|
||||
if (DIM == 1) { return internal::quadrature_interpolator::Det1D; }
|
||||
if (DIM == 1)
|
||||
{
|
||||
if (SDIM == 1) { return internal::quadrature_interpolator::Det1D; }
|
||||
else if (SDIM == 2) { return internal::quadrature_interpolator::Det1DSurface<0,0,2>; }
|
||||
else if (SDIM == 3) { return internal::quadrature_interpolator::Det1DSurface<0,0,3>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
}
|
||||
else if (DIM == 2 && SDIM == 2) { return internal::quadrature_interpolator::Det2D; }
|
||||
else if (DIM == 2 && SDIM == 3) { return internal::quadrature_interpolator::Det2DSurface; }
|
||||
else if (DIM == 3)
|
||||
|
||||
+51
-1
@@ -56,6 +56,50 @@ inline void Det1D(const int NE,
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_SDIM = 3>
|
||||
inline void Det1DSurface(const int NE,
|
||||
const real_t *b,
|
||||
const real_t *g,
|
||||
const real_t *x,
|
||||
real_t *y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0,
|
||||
Vector *d_buff = nullptr)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(b);
|
||||
MFEM_CONTRACT_VAR(d_buff);
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto G = Reshape(g, Q1D, D1D);
|
||||
const auto X = Reshape(x, D1D, T_SDIM, NE);
|
||||
auto Y = Reshape(y, Q1D, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
for (int q = 0; q < Q1D; q++)
|
||||
{
|
||||
real_t grad[T_SDIM];
|
||||
for (int s = 0; s < T_SDIM; s++) { grad[s] = 0.0; }
|
||||
for (int d = 0; d < D1D; d++)
|
||||
{
|
||||
const real_t gval = G(q, d);
|
||||
for (int s = 0; s < T_SDIM; s++)
|
||||
{
|
||||
grad[s] += gval * X(d, s, e);
|
||||
}
|
||||
}
|
||||
real_t norm2 = 0.0;
|
||||
for (int s = 0; s < T_SDIM; s++)
|
||||
{
|
||||
norm2 += grad[s] * grad[s];
|
||||
}
|
||||
Y(q, e) = std::sqrt(norm2);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void Det2D(const int NE,
|
||||
const real_t *b,
|
||||
@@ -290,7 +334,13 @@ template<int DIM, int SDIM, int D1D, int Q1D>
|
||||
QuadratureInterpolator::DetKernelType
|
||||
QuadratureInterpolator::DetKernels::Kernel()
|
||||
{
|
||||
if (DIM == 1) { return internal::quadrature_interpolator::Det1D; }
|
||||
if (DIM == 1)
|
||||
{
|
||||
if (SDIM == 1) { return internal::quadrature_interpolator::Det1D; }
|
||||
else if (SDIM == 2) { return internal::quadrature_interpolator::Det1DSurface<D1D, Q1D, 2>; }
|
||||
else if (SDIM == 3) { return internal::quadrature_interpolator::Det1DSurface<D1D, Q1D, 3>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
}
|
||||
else if (DIM == 2 && SDIM == 2) { return internal::quadrature_interpolator::Det2D<D1D, Q1D>; }
|
||||
else if (DIM == 2 && SDIM == 3) { return internal::quadrature_interpolator::Det2DSurface<D1D, Q1D>; }
|
||||
else if (DIM == 3) { return internal::quadrature_interpolator::Det3D<D1D, Q1D>; }
|
||||
|
||||
@@ -542,7 +542,8 @@ void QuadratureInterpolator::Mult(const Vector &e_vec,
|
||||
}
|
||||
|
||||
MFEM_ASSERT(!(eval_flags & DETERMINANTS) || dim == vdim ||
|
||||
(dim == 2 && vdim == 3), "Invalid dimensions for determinants.");
|
||||
(dim == 2 && vdim == 3) || (dim == 1 && vdim == 2) ||
|
||||
(dim == 1 && vdim == 3), "Invalid dimensions for determinants.");
|
||||
MFEM_ASSERT(fespace->GetMesh()->GetNumGeometries(
|
||||
fespace->GetMesh()->Dimension()) == 1,
|
||||
"mixed meshes are not supported");
|
||||
|
||||
+118
-42
@@ -1506,12 +1506,12 @@ void L2FaceRestriction::EnsureNormalDerivativeRestriction() const
|
||||
}
|
||||
}
|
||||
|
||||
InterpolationManager::InterpolationManager(const FiniteElementSpace &fes,
|
||||
ElementDofOrdering ordering,
|
||||
InterpolationManager::InterpolationManager(const FiniteElementSpace &fes_,
|
||||
ElementDofOrdering ordering_,
|
||||
FaceType type)
|
||||
: fes(fes),
|
||||
ordering(ordering),
|
||||
interp_config( fes.GetNFbyType(type) ),
|
||||
: fes(fes_),
|
||||
ordering(ordering_),
|
||||
interp_config(fes.GetNFbyType(type)),
|
||||
nc_cpt(0)
|
||||
{ }
|
||||
|
||||
@@ -1536,7 +1536,8 @@ void InterpolationManager::RegisterFaceCoarseToFineInterpolation(
|
||||
face.element[0].local_face_id +
|
||||
6*face.element[1].local_face_id +
|
||||
36*face.element[1].orientation ;
|
||||
// Unfortunately we can't trust unicity of the ptMat to identify the transformation.
|
||||
// Unfortunately we can't trust uniqueness of the ptMat to identify the
|
||||
// transformation.
|
||||
Key key(ptMat, face_key);
|
||||
auto itr = interp_map.find(key);
|
||||
if ( itr == interp_map.end() )
|
||||
@@ -1583,17 +1584,27 @@ const DenseMatrix* InterpolationManager::GetCoarseToFineInterpolation(
|
||||
IsoparametricTransformation isotr;
|
||||
isotr.SetIdentityTransformation(trace_fe->GetGeomType());
|
||||
isotr.SetPointMat(*ptMat);
|
||||
DenseMatrix& trans_pt_mat = isotr.GetPointMat();
|
||||
// PointMatrix needs to be flipped in 2D
|
||||
if ( trace_fe->GetGeomType()==Geometry::SEGMENT && !is_ghost_slave )
|
||||
{
|
||||
std::swap(trans_pt_mat(0,0),trans_pt_mat(0,1));
|
||||
}
|
||||
DenseMatrix native_interpolator(face_dofs,face_dofs);
|
||||
trace_fe->GetLocalInterpolation(isotr, native_interpolator);
|
||||
|
||||
if (trace_fe->GetMapType() == FiniteElement::INTEGRAL)
|
||||
{
|
||||
// Handle potentially inverted Jacobian matrix
|
||||
isotr.SetIntPoint(&Geometries.GetCenter(trace_fe->GetGeomType()));
|
||||
native_interpolator *= (isotr.Weight() >= 0) ? 1.0 : -1.0;
|
||||
}
|
||||
|
||||
const int dim = trace_fe->GetDim()+1;
|
||||
const int dof1d = trace_fe->GetOrder()+1;
|
||||
const int orientation = face.element[1].orientation;
|
||||
int orientation_i = face.element[1].orientation;
|
||||
const int orientation_j = face.element[1].orientation;
|
||||
|
||||
// In 2D, need to flip orientation of the segments`
|
||||
if (trace_fe->GetGeomType() == Geometry::SEGMENT && !is_ghost_slave)
|
||||
{
|
||||
orientation_i = 1;
|
||||
}
|
||||
|
||||
for (int i = 0; i < face_dofs; i++)
|
||||
{
|
||||
const int ni = (dof_map.Size()==0) ? i : dof_map[i];
|
||||
@@ -1602,7 +1613,7 @@ const DenseMatrix* InterpolationManager::GetCoarseToFineInterpolation(
|
||||
{
|
||||
// master side is elem 2, so we permute to order dofs as elem 1.
|
||||
li = PermuteFaceL2(dim, face_id2, face_id1,
|
||||
orientation, dof1d, li);
|
||||
orientation_i, dof1d, li);
|
||||
}
|
||||
for (int j = 0; j < face_dofs; j++)
|
||||
{
|
||||
@@ -1611,7 +1622,7 @@ const DenseMatrix* InterpolationManager::GetCoarseToFineInterpolation(
|
||||
{
|
||||
// master side is elem 2, so we permute to order dofs as elem 1.
|
||||
lj = PermuteFaceL2(dim, face_id2, face_id1,
|
||||
orientation, dof1d, lj);
|
||||
orientation_j, dof1d, lj);
|
||||
}
|
||||
const int nj = (dof_map.Size()==0) ? j : dof_map[j];
|
||||
(*interpolator)(li,lj) = native_interpolator(ni,nj);
|
||||
@@ -1676,7 +1687,7 @@ NCL2FaceRestriction::NCL2FaceRestriction(const FiniteElementSpace &fes,
|
||||
const L2FaceValues m,
|
||||
bool build)
|
||||
: L2FaceRestriction(fes, f_ordering, type, m, false),
|
||||
interpolations(fes, f_ordering, type)
|
||||
interpolations(fes.GetInterpolationManager(ordering, type))
|
||||
{
|
||||
if (!build) { return; }
|
||||
x_interp.UseDevice(true);
|
||||
@@ -2202,14 +2213,6 @@ void NCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
|
||||
{
|
||||
PermuteAndSetFaceDofsScatterIndices2(face,f_ind);
|
||||
}
|
||||
if ( face.IsConforming() )
|
||||
{
|
||||
interpolations.RegisterFaceConformingInterpolation(face,f_ind);
|
||||
}
|
||||
else // Non-conforming face
|
||||
{
|
||||
interpolations.RegisterFaceCoarseToFineInterpolation(face,f_ind);
|
||||
}
|
||||
f_ind++;
|
||||
}
|
||||
else if ( type==FaceType::Boundary && face.IsBoundary() )
|
||||
@@ -2219,7 +2222,6 @@ void NCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
|
||||
{
|
||||
SetBoundaryDofsScatterIndices2(face,f_ind);
|
||||
}
|
||||
interpolations.RegisterFaceConformingInterpolation(face,f_ind);
|
||||
f_ind++;
|
||||
}
|
||||
}
|
||||
@@ -2232,10 +2234,6 @@ void NCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
|
||||
{
|
||||
gather_offsets[i] += gather_offsets[i - 1];
|
||||
}
|
||||
|
||||
// Transform the interpolation matrix map into a contiguous memory structure.
|
||||
interpolations.LinearizeInterpolatorMapIntoVector();
|
||||
interpolations.InitializeNCInterpConfig();
|
||||
}
|
||||
|
||||
void NCL2FaceRestriction::ComputeGatherIndices()
|
||||
@@ -2278,6 +2276,18 @@ void NCL2FaceRestriction::ComputeGatherIndices()
|
||||
gather_offsets[0] = 0;
|
||||
}
|
||||
|
||||
static int GetSharedVSize(const FiniteElementSpace &fes)
|
||||
{
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (auto pfes = dynamic_cast<const ParFiniteElementSpace*>(&fes))
|
||||
{
|
||||
const_cast<ParFiniteElementSpace*>(pfes)->ExchangeFaceNbrData();
|
||||
return pfes->GetFaceNbrVSize();
|
||||
}
|
||||
#endif
|
||||
return 0;
|
||||
}
|
||||
|
||||
L2InterfaceFaceRestriction::L2InterfaceFaceRestriction(
|
||||
const FiniteElementSpace& fes_,
|
||||
const ElementDofOrdering ordering_,
|
||||
@@ -2288,25 +2298,54 @@ L2InterfaceFaceRestriction::L2InterfaceFaceRestriction(
|
||||
nfaces(fes.GetNFbyType(type)),
|
||||
vdim(fes.GetVDim()),
|
||||
byvdim(fes.GetOrdering() == Ordering::byVDIM),
|
||||
face_dofs(nfaces > 0 ? fes.GetFaceElement(0)->GetDof() : 0),
|
||||
face_dofs(fes.GetTypicalTraceElement()->GetDof()),
|
||||
nfdofs(face_dofs*nfaces),
|
||||
ndofs(fes.GetNDofs())
|
||||
ndofs(fes.GetNDofs()),
|
||||
nsdofs(GetSharedVSize(fes))
|
||||
{
|
||||
height = nfdofs;
|
||||
width = ndofs;
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
auto pfes = dynamic_cast<const ParFiniteElementSpace*>(&fes);
|
||||
#endif
|
||||
|
||||
const Table &face2dof = fes.GetFaceToDofTable();
|
||||
|
||||
const Mesh &mesh = *fes.GetMesh();
|
||||
int face_idx = 0;
|
||||
gather_map.SetSize(nfdofs);
|
||||
for (int f = 0; f < mesh.GetNumFaces(); ++f)
|
||||
scatter_map.SetSize(nfdofs);
|
||||
gather_map.SetSize(ndofs + nsdofs);
|
||||
gather_map = -1;
|
||||
|
||||
Array<int> dofs;
|
||||
for (int f = 0; f < mesh.GetNumFacesWithGhost(); ++f)
|
||||
{
|
||||
Mesh::FaceInformation face = mesh.GetFaceInformation(f);
|
||||
if (!face.IsOfFaceType(type)) { continue; }
|
||||
for (int i = 0; i < face_dofs; ++i)
|
||||
if (!face.IsOfFaceType(type) || face.IsNonconformingCoarse()) { continue; }
|
||||
|
||||
if (f < mesh.GetNumFaces())
|
||||
{
|
||||
gather_map[i + face_idx*face_dofs] = face2dof.GetJ()[i + f*face_dofs];
|
||||
// Local face
|
||||
face2dof.GetRow(f, dofs);
|
||||
for (int i = 0; i < face_dofs; ++i)
|
||||
{
|
||||
scatter_map[i + face_idx*face_dofs] = dofs[i];
|
||||
gather_map[dofs[i]] = i + face_idx*face_dofs;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// Shared (non-conforming) ghost face
|
||||
#ifdef MFEM_USE_MPI
|
||||
MFEM_ASSERT(pfes != nullptr, "");
|
||||
pfes->GetFaceNbrFaceVDofs(f, dofs);
|
||||
for (int i = 0; i < face_dofs; ++i)
|
||||
{
|
||||
scatter_map[i + face_idx*face_dofs] = ndofs + dofs[i];
|
||||
gather_map[ndofs + dofs[i]] = i + face_idx*face_dofs;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
++face_idx;
|
||||
}
|
||||
@@ -2314,13 +2353,19 @@ L2InterfaceFaceRestriction::L2InterfaceFaceRestriction(
|
||||
|
||||
void L2InterfaceFaceRestriction::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
const int NDOFS = ndofs;
|
||||
const int nd = face_dofs;
|
||||
const int nf = nfaces;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
const int *map = gather_map.Read();
|
||||
const int *map = scatter_map.Read();
|
||||
|
||||
Vector face_nbr_data = GetLVectorFaceNbrData(fes, x, type);
|
||||
MFEM_ASSERT(face_nbr_data.Size() / vd == nsdofs, "");
|
||||
|
||||
const auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
|
||||
const auto d_x_shared = Reshape(face_nbr_data.Read(),
|
||||
t?vd:nsdofs, t?nsdofs:vd);
|
||||
auto d_y = Reshape(y.Write(), nd, vd, nf);
|
||||
|
||||
mfem::forall(nd*nf, [=] MFEM_HOST_DEVICE (int i)
|
||||
@@ -2328,7 +2373,8 @@ void L2InterfaceFaceRestriction::Mult(const Vector &x, Vector &y) const
|
||||
const int j = map[i];
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(i % nd, c, i / nd) = d_x(t?c:j, t?j:c);
|
||||
if (j < NDOFS) { d_y(i % nd, c, i / nd) = d_x(t?c:j, t?j:c); }
|
||||
else { d_y(i % nd, c, i / nd) = d_x_shared(t?c:(j-NDOFS), t?(j-NDOFS):c); }
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -2343,15 +2389,39 @@ void L2InterfaceFaceRestriction::AddMultTranspose(
|
||||
const int *map = gather_map.Read();
|
||||
|
||||
const auto d_x = Reshape(x.Read(), nd, vd, nf);
|
||||
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
|
||||
auto d_y = Reshape(y.ReadWrite(), t?vd:ndofs, t?ndofs:vd);
|
||||
|
||||
mfem::forall(ndofs, [=] MFEM_HOST_DEVICE (int i) { d_y[i] = 0.0; });
|
||||
mfem::forall(nd*nf, [=] MFEM_HOST_DEVICE (int i)
|
||||
mfem::forall(ndofs, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
const int j = map[i];
|
||||
if (j < 0) { return; }
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(t?c:j, t?j:c) = d_x(i % nd, c, i / nd);
|
||||
d_y(t?c:i, t?i:c) += a*d_x(j % nd, c, j / nd);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void L2InterfaceFaceRestriction::MultTransposeShared(
|
||||
const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nd = face_dofs;
|
||||
const int nf = nfaces;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
const int *map = gather_map.Read();
|
||||
|
||||
const auto d_x = Reshape(x.Read(), nd, vd, nf);
|
||||
auto d_y = Reshape(y.Write(), t?vd:(ndofs+nsdofs), t?(ndofs+nsdofs):vd);
|
||||
y = 0.0;
|
||||
|
||||
mfem::forall(ndofs + nsdofs, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
const int j = map[i];
|
||||
if (j < 0) { return; }
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(t?c:i, t?i:c) = d_x(j % nd, c, j / nd);
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -2361,6 +2431,11 @@ const Array<int> &L2InterfaceFaceRestriction::GatherMap() const
|
||||
return gather_map;
|
||||
}
|
||||
|
||||
const Array<int> &L2InterfaceFaceRestriction::ScatterMap() const
|
||||
{
|
||||
return scatter_map;
|
||||
}
|
||||
|
||||
Vector GetLVectorFaceNbrData(
|
||||
const FiniteElementSpace &fes, const Vector &x, FaceType ftype)
|
||||
{
|
||||
@@ -2382,6 +2457,7 @@ Vector GetLVectorFaceNbrData(
|
||||
{
|
||||
ParGridFunction gf(pfes, const_cast<Vector&>(x));
|
||||
gf.ExchangeFaceNbrData();
|
||||
x.SyncMemory(gf);
|
||||
return std::move(gf.FaceNbrData());
|
||||
}
|
||||
}
|
||||
|
||||
+26
-14
@@ -812,13 +812,12 @@ protected:
|
||||
PointMatrix and a local face identifier. */
|
||||
using Key = std::pair<const DenseMatrix*,int>;
|
||||
/// The temporary map used to store the different interpolators.
|
||||
using Map = std::map<Key, std::pair<int,const DenseMatrix*>>;
|
||||
using Map =
|
||||
std::unordered_map<Key, std::pair<int,const DenseMatrix*>, PairHasher>;
|
||||
Map interp_map; // The temporary map that stores the interpolators.
|
||||
|
||||
public:
|
||||
InterpolationManager() = delete;
|
||||
|
||||
/** @brief main constructor.
|
||||
/** @brief Constructor.
|
||||
|
||||
@param[in] fes The FiniteElementSpace on which this operates
|
||||
@param[in] ordering Request a specific element ordering.
|
||||
@@ -909,7 +908,7 @@ private:
|
||||
class NCL2FaceRestriction : virtual public L2FaceRestriction
|
||||
{
|
||||
protected:
|
||||
InterpolationManager interpolations;
|
||||
const InterpolationManager &interpolations;
|
||||
mutable Vector x_interp;
|
||||
|
||||
/** @brief Constructs an NCL2FaceRestriction, this is a specialization of a
|
||||
@@ -996,9 +995,7 @@ public:
|
||||
@param[in] keep_nbr_block When set to true the SparseMatrix will
|
||||
include the rows (in addition to the columns)
|
||||
corresponding to face-neighbor dofs. The
|
||||
default behavior is to disregard those rows.
|
||||
|
||||
@warning This method is not implemented yet. */
|
||||
default behavior is to disregard those rows. */
|
||||
void FillI(SparseMatrix &mat,
|
||||
const bool keep_nbr_block = false) const override;
|
||||
|
||||
@@ -1016,9 +1013,7 @@ public:
|
||||
@param[in] keep_nbr_block When set to true the SparseMatrix will
|
||||
include the rows (in addition to the columns)
|
||||
corresponding to face-neighbor dofs. The
|
||||
default behavior is to disregard those rows.
|
||||
|
||||
@warning This method is not implemented yet. */
|
||||
default behavior is to disregard those rows. */
|
||||
void FillJAndData(const Vector &fea_data,
|
||||
SparseMatrix &mat,
|
||||
const bool keep_nbr_block = false) const override;
|
||||
@@ -1036,9 +1031,7 @@ public:
|
||||
added the face contributions.
|
||||
The format is: dofs x dofs x ne, where dofs is the
|
||||
number of dofs per element and ne the number of
|
||||
elements.
|
||||
|
||||
@warning This method is not implemented yet. */
|
||||
elements. */
|
||||
void AddFaceMatricesToElementMatrices(const Vector &fea_data,
|
||||
Vector &ea_data) const override;
|
||||
|
||||
@@ -1130,7 +1123,9 @@ protected:
|
||||
const int face_dofs; ///< Number of dofs on each face
|
||||
const int nfdofs; ///< Total number of dofs on the faces (E-vector size)
|
||||
const int ndofs; ///< Number of dofs in the space (L-vector size)
|
||||
const int nsdofs; ///< Number of shared face neighbor (ghost) dofs
|
||||
Array<int> gather_map; ///< Gather map
|
||||
Array<int> scatter_map; ///< Scatter map
|
||||
|
||||
public:
|
||||
/** @brief Constructs an L2InterfaceFaceRestriction.
|
||||
@@ -1168,7 +1163,24 @@ public:
|
||||
void AddMultTranspose(const Vector &x, Vector &y,
|
||||
const real_t a = 1.0) const override;
|
||||
|
||||
/// @brief Gather degrees of freedom, from face E-vector to L-vector and
|
||||
/// shared (ghost) DOFs.
|
||||
///
|
||||
/// @param[in] x The face E-Vector degrees of freedom with size
|
||||
/// (face_dofs, vdim, nf), where nf is the number of
|
||||
/// interior or boundary faces requested by @a type in the
|
||||
/// constructor. The face_dofs should be ordered according
|
||||
/// to the given ElementDofOrdering
|
||||
/// @param[out] y Vector of length vsize + face neighbor vsize
|
||||
void MultTransposeShared(const Vector &x, Vector &y) const;
|
||||
|
||||
const Array<int> &GatherMap() const override;
|
||||
|
||||
/// @brief Return the low-level mapping from L-dofs to E-dofs.
|
||||
///
|
||||
/// L-dofs that do not correspond to an E-dof (e.g. that lie on a face of a
|
||||
/// different type) are given index -1.
|
||||
const Array<int> &ScatterMap() const;
|
||||
};
|
||||
|
||||
/** @brief Convert a dof face index from Native ordering to lexicographic
|
||||
|
||||
+31
-12
@@ -333,6 +333,12 @@ void L2ProjectionGridTransfer::L2Projection::MixedMassEA(
|
||||
int nel_ho = mesh_ho->GetNE();
|
||||
int nel_lor = mesh_lor->GetNE();
|
||||
|
||||
if (nel_ho == 0)
|
||||
{
|
||||
M_LH.SetSize(0);
|
||||
return;
|
||||
}
|
||||
|
||||
const CoarseFineTransformations& cf_tr = mesh_lor->GetRefinementTransforms();
|
||||
|
||||
int nref_max = 0;
|
||||
@@ -831,11 +837,17 @@ void L2ProjectionGridTransfer::L2ProjectionL2Space::Mult(
|
||||
void L2ProjectionGridTransfer::L2ProjectionL2Space::EAMult(
|
||||
const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nel_ho = fes_ho.GetMesh()->GetNE();
|
||||
|
||||
if (nel_ho == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
const int iho = 0;
|
||||
const int nref = ho2lor.RowSize(iho);
|
||||
const int ndof_ho = fes_ho.GetFE(iho)->GetDof();
|
||||
const int ndof_lor = fes_lor.GetFE(ho2lor.GetRow(iho)[0])->GetDof();
|
||||
const int nel_ho = fes_ho.GetMesh()->GetNE();
|
||||
|
||||
DenseTensor R_dt;
|
||||
R_dt.NewMemoryAndSize(R.GetMemory(), ndof_lor*nref, ndof_ho, nel_ho, false);
|
||||
@@ -887,11 +899,17 @@ void L2ProjectionGridTransfer::L2ProjectionL2Space::MultTranspose(
|
||||
void L2ProjectionGridTransfer::L2ProjectionL2Space::EAMultTranspose(
|
||||
const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nel_ho = fes_ho.GetMesh()->GetNE();
|
||||
|
||||
if (nel_ho == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
const int iho = 0;
|
||||
const int nref = ho2lor.RowSize(iho);
|
||||
const int ndof_ho = fes_ho.GetFE(iho)->GetDof();
|
||||
const int ndof_lor = fes_lor.GetFE(ho2lor.GetRow(iho)[0])->GetDof();
|
||||
const int nel_ho = fes_ho.GetMesh()->GetNE();
|
||||
|
||||
DenseTensor R_dt;
|
||||
R_dt.NewMemoryAndSize(R.GetMemory(), ndof_lor*nref, ndof_ho, nel_ho, false);
|
||||
@@ -901,7 +919,6 @@ void L2ProjectionGridTransfer::L2ProjectionL2Space::EAMultTranspose(
|
||||
void L2ProjectionGridTransfer::L2ProjectionL2Space::Prolongate(
|
||||
const Vector &x, Vector &y) const
|
||||
{
|
||||
|
||||
if (fes_ho.GetNE() == 0) { return; }
|
||||
|
||||
if (use_ea)
|
||||
@@ -960,14 +977,13 @@ void L2ProjectionGridTransfer::L2ProjectionL2Space::EAProlongate(
|
||||
void L2ProjectionGridTransfer::L2ProjectionL2Space::ProlongateTranspose(
|
||||
const Vector &x, Vector &y) const
|
||||
{
|
||||
if (fes_ho.GetNE() == 0) { return; }
|
||||
|
||||
if (use_ea)
|
||||
{
|
||||
return EAProlongateTranspose(x,y);
|
||||
}
|
||||
|
||||
|
||||
if (fes_ho.GetNE() == 0) { return; }
|
||||
MFEM_VERIFY(P.Size() > 0, "Prolongation not supported for these spaces.")
|
||||
int vdim = fes_ho.GetVDim();
|
||||
Array<int> vdofs;
|
||||
@@ -1244,13 +1260,6 @@ void L2ProjectionGridTransfer::L2ProjectionH1Space::EAL2ProjectionH1Space
|
||||
int ndof_ho = pfes_ho.GetNDofs();
|
||||
int ndof_lor = pfes_lor.GetNDofs();
|
||||
|
||||
|
||||
// If the local mesh is empty, skip all computations
|
||||
if (nel_ho == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
const CoarseFineTransformations& cf_tr = mesh_lor->GetRefinementTransforms();
|
||||
|
||||
int nref_max = 0;
|
||||
@@ -1860,6 +1869,11 @@ L2ProjectionGridTransfer::H1SpaceMixedMassOperator::H1SpaceMixedMassOperator(
|
||||
void L2ProjectionGridTransfer::H1SpaceMixedMassOperator::Mult(const Vector &x,
|
||||
Vector &y) const
|
||||
{
|
||||
if (fes_ho->GetNE() == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
const Operator* elem_restrict_ho = fes_ho->GetElementRestriction(
|
||||
ElementDofOrdering::NATIVE);
|
||||
const Operator* elem_restrict_lor = fes_lor->GetElementRestriction(
|
||||
@@ -1906,6 +1920,11 @@ void L2ProjectionGridTransfer::H1SpaceMixedMassOperator::Mult(const Vector &x,
|
||||
void L2ProjectionGridTransfer::H1SpaceMixedMassOperator::MultTranspose(
|
||||
const Vector &x, Vector &y) const
|
||||
{
|
||||
if (fes_ho->GetNE() == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
const Operator* elem_restrict_ho = fes_ho->GetElementRestriction(
|
||||
ElementDofOrdering::NATIVE);
|
||||
const Operator* elem_restrict_lor = fes_lor->GetElementRestriction(
|
||||
|
||||
@@ -18,6 +18,7 @@ list(APPEND SRCS
|
||||
gecko.cpp
|
||||
globals.cpp
|
||||
hash.cpp
|
||||
hash_util.cpp
|
||||
isockstream.cpp
|
||||
mem_manager.cpp
|
||||
occa.cpp
|
||||
@@ -46,6 +47,7 @@ list(APPEND HDRS
|
||||
globals.hpp
|
||||
zstr.hpp
|
||||
hash.hpp
|
||||
hash_util.hpp
|
||||
isockstream.hpp
|
||||
kdtree.hpp
|
||||
mem_alloc.hpp
|
||||
|
||||
@@ -44,6 +44,7 @@
|
||||
#endif
|
||||
|
||||
#if !defined(MFEM_USE_CUDA_OR_HIP)
|
||||
constexpr bool mfem_use_gpu = false;
|
||||
#define MFEM_DEVICE
|
||||
#define MFEM_HOST
|
||||
#define MFEM_LAMBDA
|
||||
@@ -52,6 +53,7 @@
|
||||
#define MFEM_DEVICE_SYNC
|
||||
// MFEM_STREAM_SYNC is used for UVM and MPI GPU-Aware kernels
|
||||
#define MFEM_STREAM_SYNC
|
||||
#define MFEM_LAUNCH_BOUNDS(...)
|
||||
#endif
|
||||
|
||||
#if !((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
|
||||
|
||||
@@ -20,9 +20,11 @@
|
||||
|
||||
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
|
||||
#define MFEM_USE_CUDA_OR_HIP
|
||||
constexpr bool mfem_use_gpu = true;
|
||||
#define MFEM_DEVICE __device__
|
||||
#define MFEM_HOST __host__
|
||||
#define MFEM_LAMBDA __host__
|
||||
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
|
||||
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
|
||||
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(cudaDeviceSynchronize())
|
||||
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(cudaStreamSynchronize(0))
|
||||
|
||||
+207
-44
@@ -295,11 +295,12 @@ using hip_threads_z =
|
||||
#endif
|
||||
|
||||
#if defined(MFEM_USE_RAJA) && defined(RAJA_ENABLE_CUDA) && defined(__CUDACC__)
|
||||
template <const int BLOCKS = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
void RajaCuWrap1D(const int N, DBODY &&d_body)
|
||||
{
|
||||
//true denotes asynchronous kernel
|
||||
RAJA::forall<RAJA::cuda_exec<BLOCKS,true>>(RAJA::RangeSegment(0,N),d_body);
|
||||
RAJA::forall<RAJA::cuda_exec<MFEM_CUDA_BLOCKS,true>>(RAJA::RangeSegment(0,N),
|
||||
d_body);
|
||||
}
|
||||
|
||||
template <typename DBODY>
|
||||
@@ -362,18 +363,18 @@ struct RajaCuWrap;
|
||||
template <>
|
||||
struct RajaCuWrap<1>
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
RajaCuWrap1D<BLCK>(N, d_body);
|
||||
RajaCuWrap1D(N, d_body);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct RajaCuWrap<2>
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
@@ -384,7 +385,7 @@ struct RajaCuWrap<2>
|
||||
template <>
|
||||
struct RajaCuWrap<3>
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
@@ -395,11 +396,12 @@ struct RajaCuWrap<3>
|
||||
#endif
|
||||
|
||||
#if defined(MFEM_USE_RAJA) && defined(RAJA_ENABLE_HIP) && defined(__HIP__)
|
||||
template <const int BLOCKS = MFEM_HIP_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
void RajaHipWrap1D(const int N, DBODY &&d_body)
|
||||
{
|
||||
//true denotes asynchronous kernel
|
||||
RAJA::forall<RAJA::hip_exec<BLOCKS,true>>(RAJA::RangeSegment(0,N),d_body);
|
||||
RAJA::forall<RAJA::hip_exec<MFEM_HIP_BLOCKS,true>>(RAJA::RangeSegment(0,N),
|
||||
d_body);
|
||||
}
|
||||
|
||||
template <typename DBODY>
|
||||
@@ -462,18 +464,18 @@ struct RajaHipWrap;
|
||||
template <>
|
||||
struct RajaHipWrap<1>
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
RajaHipWrap1D<BLCK>(N, d_body);
|
||||
RajaHipWrap1D(N, d_body);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct RajaHipWrap<2>
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
@@ -484,7 +486,7 @@ struct RajaHipWrap<2>
|
||||
template <>
|
||||
struct RajaHipWrap<3>
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
@@ -584,12 +586,31 @@ void CuKernel2D(const int N, BODY body)
|
||||
body(k);
|
||||
}
|
||||
|
||||
// __launch_bounds__ second argument is omitted to get the default behavior
|
||||
template <int MAX_THREADS_PER_BLOCK, typename BODY>
|
||||
__global__
|
||||
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
|
||||
static void CuKernel2DLaunchBounds(const int N, BODY body)
|
||||
{
|
||||
const int k = blockIdx.x*blockDim.z + threadIdx.z;
|
||||
if (k >= N) { return; }
|
||||
body(k);
|
||||
}
|
||||
|
||||
template <typename BODY> __global__ static
|
||||
void CuKernel3D(const int N, BODY body)
|
||||
{
|
||||
for (int k = blockIdx.x; k < N; k += gridDim.x) { body(k); }
|
||||
}
|
||||
|
||||
template <int MAX_THREADS_PER_BLOCK, typename BODY>
|
||||
__global__
|
||||
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
|
||||
static void CuKernel3DLaunchBounds(const int N, BODY body)
|
||||
{
|
||||
for (int k = blockIdx.x; k < N; k += gridDim.x) { body(k); }
|
||||
}
|
||||
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
void CuWrap1D(const int N, DBODY &&d_body)
|
||||
{
|
||||
@@ -604,6 +625,8 @@ void CuWrap2D(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int BZ)
|
||||
{
|
||||
if (N==0) { return; }
|
||||
// required for optimized GCC/NVCC builds to prevent runtime
|
||||
// ODR/linkage violations of inlined templated kernel helpers
|
||||
MFEM_VERIFY(BZ>0, "");
|
||||
const int GRID = (N+BZ-1)/BZ;
|
||||
const dim3 BLCK(X,Y,BZ);
|
||||
@@ -611,6 +634,19 @@ void CuWrap2D(const int N, DBODY &&d_body,
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
}
|
||||
|
||||
template <int MAX_THREADS_PER_BLOCK, typename DBODY>
|
||||
void CuWrap2DLaunchBounds(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int BZ)
|
||||
{
|
||||
if (N==0) { return; }
|
||||
MFEM_VERIFY(BZ>0, "");
|
||||
const int GRID = (N+BZ-1)/BZ;
|
||||
const dim3 BLCK(X,Y,BZ);
|
||||
static_assert(MAX_THREADS_PER_BLOCK > 0);
|
||||
CuKernel2DLaunchBounds<MAX_THREADS_PER_BLOCK><<<GRID,BLCK>>>(N, d_body);
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
}
|
||||
|
||||
template <typename DBODY>
|
||||
void CuWrap3D(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
@@ -622,24 +658,35 @@ void CuWrap3D(const int N, DBODY &&d_body,
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
}
|
||||
|
||||
template <int Dim>
|
||||
struct CuWrap;
|
||||
|
||||
template <>
|
||||
struct CuWrap<1>
|
||||
template <int MAX_THREADS_PER_BLOCK, typename DBODY>
|
||||
void CuWrap3DLaunchBounds(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
if (N==0) { return; }
|
||||
const int GRID = G == 0 ? N : G;
|
||||
const dim3 BLCK(X,Y,Z);
|
||||
static_assert(MAX_THREADS_PER_BLOCK > 0);
|
||||
CuKernel3DLaunchBounds<MAX_THREADS_PER_BLOCK><<<GRID, BLCK>>>(N, d_body);
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
}
|
||||
|
||||
template <int Dim, int MAX_THREADS_PER_BLOCK> struct CuWrap;
|
||||
|
||||
template <int MAX_THREADS_PER_BLOCK>
|
||||
struct CuWrap<1, MAX_THREADS_PER_BLOCK>
|
||||
{
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
CuWrap1D<BLCK>(N, d_body);
|
||||
CuWrap1D<MFEM_CUDA_BLOCKS>(N, d_body);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct CuWrap<2>
|
||||
struct CuWrap<2, 0>
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
@@ -647,10 +694,22 @@ struct CuWrap<2>
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct CuWrap<3>
|
||||
template <int MAX_THREADS_PER_BLOCK>
|
||||
struct CuWrap<2, MAX_THREADS_PER_BLOCK>
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
static_assert(MAX_THREADS_PER_BLOCK > 0);
|
||||
CuWrap2DLaunchBounds<MAX_THREADS_PER_BLOCK>(N, d_body, X, Y, Z);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct CuWrap<3, 0>
|
||||
{
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
@@ -658,6 +717,17 @@ struct CuWrap<3>
|
||||
}
|
||||
};
|
||||
|
||||
template <int MAX_THREADS_PER_BLOCK>
|
||||
struct CuWrap<3, MAX_THREADS_PER_BLOCK>
|
||||
{
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
CuWrap3DLaunchBounds<MAX_THREADS_PER_BLOCK>(N, d_body, X, Y, Z, G);
|
||||
}
|
||||
};
|
||||
|
||||
#endif // defined(MFEM_USE_CUDA) && defined(__CUDACC__)
|
||||
|
||||
|
||||
@@ -680,13 +750,31 @@ void HipKernel2D(const int N, BODY body)
|
||||
body(k);
|
||||
}
|
||||
|
||||
template <int MAX_THREADS_PER_BLOCK, typename BODY>
|
||||
__global__
|
||||
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
|
||||
static void HipKernel2DLaunchBounds(const int N, BODY body)
|
||||
{
|
||||
const int k = hipBlockIdx_x*hipBlockDim_z + hipThreadIdx_z;
|
||||
if (k >= N) { return; }
|
||||
body(k);
|
||||
}
|
||||
|
||||
template <typename BODY> __global__ static
|
||||
void HipKernel3D(const int N, BODY body)
|
||||
{
|
||||
for (int k = hipBlockIdx_x; k < N; k += hipGridDim_x) { body(k); }
|
||||
}
|
||||
|
||||
template <const int BLCK = MFEM_HIP_BLOCKS, typename DBODY>
|
||||
template <int MAX_THREADS_PER_BLOCK, typename BODY>
|
||||
__global__
|
||||
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
|
||||
static void HipKernel3DLaunchBounds(const int N, BODY body)
|
||||
{
|
||||
for (int k = hipBlockIdx_x; k < N; k += hipGridDim_x) { body(k); }
|
||||
}
|
||||
|
||||
template <int BLCK = MFEM_HIP_BLOCKS, typename DBODY>
|
||||
void HipWrap1D(const int N, DBODY &&d_body)
|
||||
{
|
||||
if (N==0) { return; }
|
||||
@@ -700,12 +788,27 @@ void HipWrap2D(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int BZ)
|
||||
{
|
||||
if (N==0) { return; }
|
||||
MFEM_VERIFY(BZ>0, "");
|
||||
const int GRID = (N+BZ-1)/BZ;
|
||||
const dim3 BLCK(X,Y,BZ);
|
||||
hipLaunchKernelGGL(HipKernel2D,GRID,BLCK,0,nullptr,N,d_body);
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
}
|
||||
|
||||
template <int MAX_THREADS_PER_BLOCK, typename DBODY>
|
||||
void HipWrap2DLaunchBounds(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int BZ)
|
||||
{
|
||||
if (N==0) { return; }
|
||||
MFEM_VERIFY(BZ>0, "");
|
||||
const int GRID = (N+BZ-1)/BZ;
|
||||
const dim3 BLCK(X,Y,BZ);
|
||||
static_assert(MAX_THREADS_PER_BLOCK > 0);
|
||||
HipKernel2DLaunchBounds<MAX_THREADS_PER_BLOCK><<<dim3(GRID), dim3(BLCK), 0, 0>>>
|
||||
(N, d_body);
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
}
|
||||
|
||||
template <typename DBODY>
|
||||
void HipWrap3D(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
@@ -717,24 +820,36 @@ void HipWrap3D(const int N, DBODY &&d_body,
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
}
|
||||
|
||||
template <int Dim>
|
||||
struct HipWrap;
|
||||
|
||||
template <>
|
||||
struct HipWrap<1>
|
||||
template <int MAX_THREADS_PER_BLOCK, typename DBODY>
|
||||
void HipWrap3DLaunchBounds(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
if (N==0) { return; }
|
||||
const int GRID = G == 0 ? N : G;
|
||||
const dim3 BLCK(X,Y,Z);
|
||||
static_assert(MAX_THREADS_PER_BLOCK > 0);
|
||||
HipKernel3DLaunchBounds<MAX_THREADS_PER_BLOCK><<<dim3(GRID), dim3(BLCK), 0, 0>>>
|
||||
(N, d_body);
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
}
|
||||
|
||||
template <int Dim, int MAX_THREADS_PER_BLOCK> struct HipWrap;
|
||||
|
||||
template <int MAX_THREADS_PER_BLOCK>
|
||||
struct HipWrap<1, MAX_THREADS_PER_BLOCK>
|
||||
{
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
HipWrap1D<BLCK>(N, d_body);
|
||||
HipWrap1D<MFEM_HIP_BLOCKS>(N, d_body);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct HipWrap<2>
|
||||
struct HipWrap<2, 0>
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
@@ -742,10 +857,21 @@ struct HipWrap<2>
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct HipWrap<3>
|
||||
template <int MAX_THREADS_PER_BLOCK>
|
||||
struct HipWrap<2, MAX_THREADS_PER_BLOCK>
|
||||
{
|
||||
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
HipWrap2DLaunchBounds<MAX_THREADS_PER_BLOCK>(N, d_body, X, Y, Z);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct HipWrap<3, 0>
|
||||
{
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
@@ -753,11 +879,24 @@ struct HipWrap<3>
|
||||
}
|
||||
};
|
||||
|
||||
template <int MAX_THREADS_PER_BLOCK>
|
||||
struct HipWrap<3, MAX_THREADS_PER_BLOCK>
|
||||
{
|
||||
template <typename DBODY>
|
||||
static void run(const int N, DBODY &&d_body,
|
||||
const int X, const int Y, const int Z, const int G)
|
||||
{
|
||||
HipWrap3DLaunchBounds<MAX_THREADS_PER_BLOCK>(N, d_body, X, Y, Z, G);
|
||||
}
|
||||
};
|
||||
|
||||
#endif // defined(MFEM_USE_HIP) && defined(__HIP__)
|
||||
|
||||
|
||||
/// The forall kernel body wrapper
|
||||
template <const int DIM, typename d_lambda, typename h_lambda>
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// Forall host & device kernel dispatch
|
||||
template <int DIM, int MAX_THREADS_PER_BLOCK = 0,
|
||||
typename d_lambda, typename h_lambda>
|
||||
inline void ForallWrap(const bool use_dev, const int N,
|
||||
d_lambda &&d_body, h_lambda &&h_body,
|
||||
const int X=0, const int Y=0, const int Z=0,
|
||||
@@ -790,7 +929,7 @@ inline void ForallWrap(const bool use_dev, const int N,
|
||||
// If Backend::CUDA is allowed, use it
|
||||
if (Device::Allows(Backend::CUDA))
|
||||
{
|
||||
return CuWrap<DIM>::run(N, d_body, X, Y, Z, G);
|
||||
return CuWrap<DIM, MAX_THREADS_PER_BLOCK>::run(N, d_body, X, Y, Z, G);
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -798,7 +937,7 @@ inline void ForallWrap(const bool use_dev, const int N,
|
||||
// If Backend::HIP is allowed, use it
|
||||
if (Device::Allows(Backend::HIP))
|
||||
{
|
||||
return HipWrap<DIM>::run(N, d_body, X, Y, Z, G);
|
||||
return HipWrap<DIM, MAX_THREADS_PER_BLOCK>::run(N, d_body, X, Y, Z, G);
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -827,7 +966,9 @@ backend_cpu:
|
||||
for (int k = 0; k < N; k++) { h_body(k); }
|
||||
}
|
||||
|
||||
template <const int DIM, typename lambda>
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// Forall host & device kernel wrappers
|
||||
template <int DIM, typename lambda>
|
||||
inline void ForallWrap(const bool use_dev, const int N, lambda &&body,
|
||||
const int X=0, const int Y=0, const int Z=0,
|
||||
const int G=0)
|
||||
@@ -835,6 +976,16 @@ inline void ForallWrap(const bool use_dev, const int N, lambda &&body,
|
||||
ForallWrap<DIM>(use_dev, N, body, body, X, Y, Z, G);
|
||||
}
|
||||
|
||||
template <int DIM, int MAX_THREADS_PER_BLOCK, typename lambda>
|
||||
inline void ForallWrap(const bool use_dev, const int N, lambda &&body,
|
||||
const int X=0, const int Y=0, const int Z=0,
|
||||
const int G=0)
|
||||
{
|
||||
ForallWrap<DIM, MAX_THREADS_PER_BLOCK>(use_dev, N, body, body, X, Y, Z, G);
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// forall interfaces
|
||||
template<typename lambda>
|
||||
inline void forall(int N, lambda &&body) { ForallWrap<1>(true, N, body); }
|
||||
|
||||
@@ -843,7 +994,7 @@ inline void forall(int Nx, int Ny, lambda &&body)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
forall(Nx * Ny, [=] MFEM_HOST_DEVICE(int idx)
|
||||
mfem::forall(Nx * Ny, [=] MFEM_HOST_DEVICE(int idx)
|
||||
{
|
||||
int j = idx / Nx;
|
||||
int i = idx % Nx;
|
||||
@@ -879,7 +1030,7 @@ inline void forall(int Nx, int Ny, int Nz, lambda &&body)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
forall(Nx * Ny * Nz, [=] MFEM_HOST_DEVICE(int idx)
|
||||
mfem::forall(Nx * Ny * Nz, [=] MFEM_HOST_DEVICE(int idx)
|
||||
{
|
||||
int i = idx % Nx;
|
||||
int j = idx / Nx;
|
||||
@@ -927,6 +1078,12 @@ inline void forall_2D(int N, int X, int Y, lambda &&body)
|
||||
ForallWrap<2>(true, N, body, X, Y, 1);
|
||||
}
|
||||
|
||||
template<int MAX_THREADS_PER_BLOCK, typename lambda>
|
||||
inline void forall_2D(int N, int X, int Y, lambda &&body)
|
||||
{
|
||||
ForallWrap<2, MAX_THREADS_PER_BLOCK>(true, N, body, X, Y, 1);
|
||||
}
|
||||
|
||||
template<typename lambda>
|
||||
inline void forall_2D_batch(int N, int X, int Y, int BZ, lambda &&body)
|
||||
{
|
||||
@@ -939,6 +1096,12 @@ inline void forall_3D(int N, int X, int Y, int Z, lambda &&body)
|
||||
ForallWrap<3>(true, N, body, X, Y, Z, 0);
|
||||
}
|
||||
|
||||
template<int MAX_THREADS_PER_BLOCK, typename lambda>
|
||||
inline void forall_3D(int N, int X, int Y, int Z, lambda &&body)
|
||||
{
|
||||
ForallWrap<3, MAX_THREADS_PER_BLOCK>(true, N, body, X, Y, Z, 0);
|
||||
}
|
||||
|
||||
template<typename lambda>
|
||||
inline void forall_3D_grid(int N, int X, int Y, int Z, int G, lambda &&body)
|
||||
{
|
||||
|
||||
@@ -80,159 +80,4 @@ std::string HashFunction::GetHash() const
|
||||
return hash;
|
||||
}
|
||||
|
||||
constexpr static uint64_t rotl64(uint64_t x, int r)
|
||||
{
|
||||
return (x << r) | (x >> (64 - r));
|
||||
}
|
||||
|
||||
void Hasher::init(uint64_t seed)
|
||||
{
|
||||
data[0] = seed;
|
||||
data[1] = seed;
|
||||
nbytes = 0;
|
||||
}
|
||||
|
||||
void Hasher::add_block(uint64_t k1, uint64_t k2)
|
||||
{
|
||||
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
|
||||
constexpr uint64_t c2 = 0x4cf5ad432745937full;
|
||||
|
||||
k1 *= c1;
|
||||
k1 = rotl64(k1, 31);
|
||||
k1 *= c2;
|
||||
data[0] ^= k1;
|
||||
|
||||
data[0] = rotl64(data[0], 27);
|
||||
data[0] += data[1];
|
||||
data[0] = data[0] * 5 + 0x52dce729ull;
|
||||
|
||||
k2 *= c2;
|
||||
k2 = rotl64(k2, 33);
|
||||
k2 *= c1;
|
||||
data[1] ^= k2;
|
||||
|
||||
data[1] = rotl64(data[1], 31);
|
||||
data[1] += data[0];
|
||||
data[1] = data[1] * 5 + 0x38495ab5ull;
|
||||
}
|
||||
|
||||
static uint64_t fmix64(uint64_t k)
|
||||
{
|
||||
// http://zimbry.blogspot.com/2011/09/better-bit-mixing-improving-on.html
|
||||
// mix13
|
||||
k ^= k >> 30;
|
||||
k *= 0xbf58476d1ce4e5b9ull;
|
||||
k ^= k >> 27;
|
||||
k *= 0x94d049bb133111ebull;
|
||||
k ^= k >> 31;
|
||||
return k;
|
||||
}
|
||||
|
||||
void Hasher::append(const uint8_t *vs, uint64_t bytes)
|
||||
{
|
||||
if (bytes == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
auto rem = nbytes % 16;
|
||||
nbytes += bytes;
|
||||
uint8_t *tmp = reinterpret_cast<uint8_t *>(buf_);
|
||||
while (true)
|
||||
{
|
||||
if (bytes + rem >= 16)
|
||||
{
|
||||
std::copy(vs, vs + 16 - rem, tmp + rem);
|
||||
add_block(buf_[0], buf_[1]);
|
||||
vs += (16 - rem);
|
||||
bytes -= (16 - rem);
|
||||
rem = 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
std::copy(vs, vs + bytes, tmp + rem);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Hasher::finalize()
|
||||
{
|
||||
auto rem = nbytes % 16;
|
||||
if (rem > 0)
|
||||
{
|
||||
nbytes -= rem;
|
||||
if (rem <= 8)
|
||||
{
|
||||
finalize(buf_[0], rem);
|
||||
}
|
||||
else
|
||||
{
|
||||
finalize(buf_[0], buf_[1], rem);
|
||||
}
|
||||
return;
|
||||
}
|
||||
data[0] ^= nbytes;
|
||||
data[1] ^= nbytes;
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
|
||||
data[0] = fmix64(data[0]);
|
||||
data[1] = fmix64(data[1]);
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
}
|
||||
|
||||
void Hasher::finalize(uint64_t k1, int num)
|
||||
{
|
||||
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
|
||||
constexpr uint64_t c2 = 0x4cf5ad432745937full;
|
||||
nbytes += num;
|
||||
k1 *= c1;
|
||||
k1 = rotl64(k1, 31);
|
||||
k1 *= c2;
|
||||
data[0] ^= k1;
|
||||
|
||||
data[0] ^= nbytes;
|
||||
data[1] ^= nbytes;
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
|
||||
data[0] = fmix64(data[0]);
|
||||
data[1] = fmix64(data[1]);
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
}
|
||||
|
||||
void Hasher::finalize(uint64_t k1, uint64_t k2, int num)
|
||||
{
|
||||
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
|
||||
constexpr uint64_t c2 = 0x4cf5ad432745937full;
|
||||
nbytes += num;
|
||||
k2 *= c2;
|
||||
k2 = rotl64(k2, 33);
|
||||
k2 *= c1;
|
||||
data[1] ^= k2;
|
||||
|
||||
k1 *= c1;
|
||||
k1 = rotl64(k1, 31);
|
||||
k1 *= c2;
|
||||
data[0] ^= k1;
|
||||
|
||||
data[0] ^= nbytes;
|
||||
data[1] ^= nbytes;
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
|
||||
data[0] = fmix64(data[0]);
|
||||
data[1] = fmix64(data[1]);
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
+1
-70
@@ -15,8 +15,8 @@
|
||||
#include "../config/config.hpp"
|
||||
#include "array.hpp"
|
||||
#include "globals.hpp"
|
||||
#include "hash_util.hpp"
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
@@ -457,75 +457,6 @@ protected:
|
||||
int BinSize(int idx) const;
|
||||
};
|
||||
|
||||
///
|
||||
/// @brief streaming implementation for murmurhash3 128 (x64).
|
||||
/// Constructs the hash in 3 stages: init, append, finalize.
|
||||
///
|
||||
struct Hasher
|
||||
{
|
||||
/// where the final hash result is stored after finalize. Use data[1] when
|
||||
/// only 64 bits are required.
|
||||
uint64_t data[2] = {0, 0};
|
||||
|
||||
private:
|
||||
uint64_t nbytes = 0;
|
||||
|
||||
uint64_t buf_[2] = {0, 0};
|
||||
|
||||
public:
|
||||
|
||||
/// resets this hasher back to an initial seed
|
||||
void init(uint64_t seed = 0);
|
||||
void append(const uint8_t *vs, uint64_t bytes);
|
||||
|
||||
void finalize();
|
||||
|
||||
private:
|
||||
// add 16 bytes
|
||||
void add_block(uint64_t k1, uint64_t k2);
|
||||
|
||||
// add [1-8] more bytes, then finalize
|
||||
void finalize(uint64_t k1, int num);
|
||||
|
||||
// add [1-15] more bytes, then finalize
|
||||
// 0 < num < 16
|
||||
void finalize(uint64_t k1, uint64_t k2, int num);
|
||||
};
|
||||
|
||||
/// Helper class for hashing std::pair. Usable in place of std::hash<std::pair<T,U>>
|
||||
struct PairHasher
|
||||
{
|
||||
template <class T, class V>
|
||||
size_t operator()(const std::pair<T, V> &v) const noexcept
|
||||
{
|
||||
Hasher hash;
|
||||
// chosen randomly with a 2^64-sided dice
|
||||
hash.init(0xfebd1fe69813c14full);
|
||||
hash.append(reinterpret_cast<const uint8_t *>(&v.first), sizeof(T));
|
||||
hash.append(reinterpret_cast<const uint8_t *>(&v.second), sizeof(V));
|
||||
hash.finalize();
|
||||
return hash.data[1];
|
||||
}
|
||||
};
|
||||
|
||||
/// Helper class for hashing std::array. Usable in place of std::hash<std::array<T,N>>
|
||||
struct ArrayHasher
|
||||
{
|
||||
template <class T, size_t N>
|
||||
size_t operator()(const std::array<T, N> &v) const noexcept
|
||||
{
|
||||
Hasher hash;
|
||||
// chosen randomly with a 2^64-sided dice
|
||||
hash.init(0xfebd1fe69813c14full);
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
hash.append(reinterpret_cast<const uint8_t *>(&v[i]), sizeof(T));
|
||||
}
|
||||
hash.finalize();
|
||||
return hash.data[1];
|
||||
}
|
||||
};
|
||||
|
||||
/// Hash function for data sequences.
|
||||
/** Depends on GnuTLS for SHA-256 hashing. */
|
||||
class HashFunction
|
||||
|
||||
@@ -0,0 +1,172 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "hash_util.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
constexpr static uint64_t rotl64(uint64_t x, int r)
|
||||
{
|
||||
return (x << r) | (x >> (64 - r));
|
||||
}
|
||||
|
||||
void Hasher::init(uint64_t seed)
|
||||
{
|
||||
data[0] = seed;
|
||||
data[1] = seed;
|
||||
nbytes = 0;
|
||||
}
|
||||
|
||||
void Hasher::add_block(uint64_t k1, uint64_t k2)
|
||||
{
|
||||
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
|
||||
constexpr uint64_t c2 = 0x4cf5ad432745937full;
|
||||
|
||||
k1 *= c1;
|
||||
k1 = rotl64(k1, 31);
|
||||
k1 *= c2;
|
||||
data[0] ^= k1;
|
||||
|
||||
data[0] = rotl64(data[0], 27);
|
||||
data[0] += data[1];
|
||||
data[0] = data[0] * 5 + 0x52dce729ull;
|
||||
|
||||
k2 *= c2;
|
||||
k2 = rotl64(k2, 33);
|
||||
k2 *= c1;
|
||||
data[1] ^= k2;
|
||||
|
||||
data[1] = rotl64(data[1], 31);
|
||||
data[1] += data[0];
|
||||
data[1] = data[1] * 5 + 0x38495ab5ull;
|
||||
}
|
||||
|
||||
static uint64_t fmix64(uint64_t k)
|
||||
{
|
||||
// http://zimbry.blogspot.com/2011/09/better-bit-mixing-improving-on.html
|
||||
// mix13
|
||||
k ^= k >> 30;
|
||||
k *= 0xbf58476d1ce4e5b9ull;
|
||||
k ^= k >> 27;
|
||||
k *= 0x94d049bb133111ebull;
|
||||
k ^= k >> 31;
|
||||
return k;
|
||||
}
|
||||
|
||||
void Hasher::append(const std::byte *vs, uint64_t bytes)
|
||||
{
|
||||
if (bytes == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
auto rem = nbytes % 16;
|
||||
nbytes += bytes;
|
||||
std::byte *tmp = reinterpret_cast<std::byte *>(buf_);
|
||||
while (true)
|
||||
{
|
||||
if (bytes + rem >= 16)
|
||||
{
|
||||
std::copy(vs, vs + 16 - rem, tmp + rem);
|
||||
add_block(buf_[0], buf_[1]);
|
||||
vs += (16 - rem);
|
||||
bytes -= (16 - rem);
|
||||
rem = 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
std::copy(vs, vs + bytes, tmp + rem);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Hasher::finalize()
|
||||
{
|
||||
auto rem = nbytes % 16;
|
||||
if (rem > 0)
|
||||
{
|
||||
nbytes -= rem;
|
||||
if (rem <= 8)
|
||||
{
|
||||
finalize(buf_[0], rem);
|
||||
}
|
||||
else
|
||||
{
|
||||
finalize(buf_[0], buf_[1], rem);
|
||||
}
|
||||
return;
|
||||
}
|
||||
data[0] ^= nbytes;
|
||||
data[1] ^= nbytes;
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
|
||||
data[0] = fmix64(data[0]);
|
||||
data[1] = fmix64(data[1]);
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
}
|
||||
|
||||
void Hasher::finalize(uint64_t k1, int num)
|
||||
{
|
||||
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
|
||||
constexpr uint64_t c2 = 0x4cf5ad432745937full;
|
||||
nbytes += num;
|
||||
k1 *= c1;
|
||||
k1 = rotl64(k1, 31);
|
||||
k1 *= c2;
|
||||
data[0] ^= k1;
|
||||
|
||||
data[0] ^= nbytes;
|
||||
data[1] ^= nbytes;
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
|
||||
data[0] = fmix64(data[0]);
|
||||
data[1] = fmix64(data[1]);
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
}
|
||||
|
||||
void Hasher::finalize(uint64_t k1, uint64_t k2, int num)
|
||||
{
|
||||
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
|
||||
constexpr uint64_t c2 = 0x4cf5ad432745937full;
|
||||
nbytes += num;
|
||||
k2 *= c2;
|
||||
k2 = rotl64(k2, 33);
|
||||
k2 *= c1;
|
||||
data[1] ^= k2;
|
||||
|
||||
k1 *= c1;
|
||||
k1 = rotl64(k1, 31);
|
||||
k1 *= c2;
|
||||
data[0] ^= k1;
|
||||
|
||||
data[0] ^= nbytes;
|
||||
data[1] ^= nbytes;
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
|
||||
data[0] = fmix64(data[0]);
|
||||
data[1] = fmix64(data[1]);
|
||||
|
||||
data[0] += data[1];
|
||||
data[1] += data[0];
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,172 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_HASH_UTIL_HPP
|
||||
#define MFEM_HASH_UTIL_HPP
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <tuple>
|
||||
#include <functional>
|
||||
#include <utility>
|
||||
#include <cstdint>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// @brief streaming implementation for murmurhash3 128 (x64).
|
||||
///
|
||||
/// Constructs the hash in 3 stages: init, append, finalize.
|
||||
struct Hasher
|
||||
{
|
||||
/// @brief Storage for the final hash result after finalize() is called.
|
||||
///
|
||||
/// Use data[1] when only 64 bits are required.
|
||||
uint64_t data[2] = {0, 0};
|
||||
|
||||
private:
|
||||
uint64_t nbytes = 0;
|
||||
uint64_t buf_[2] = {0, 0};
|
||||
|
||||
public:
|
||||
|
||||
/// Resets the Hasher back to an initial seed
|
||||
void init(uint64_t seed = 0);
|
||||
|
||||
/// Append data @a vs of size @a bytes.
|
||||
void append(const std::byte *vs, uint64_t bytes);
|
||||
|
||||
void finalize();
|
||||
|
||||
private:
|
||||
/// Add a block of 16 bytes.
|
||||
void add_block(uint64_t k1, uint64_t k2);
|
||||
|
||||
/// @brief Add [1-8] more bytes, then finalize.
|
||||
///
|
||||
/// @a num must satisfy 0 < num < 9.
|
||||
void finalize(uint64_t k1, int num);
|
||||
|
||||
/// @brief Add [1-15] more bytes, then finalize.
|
||||
///
|
||||
/// @a num must satisfy 0 < num < 16.
|
||||
void finalize(uint64_t k1, uint64_t k2, int num);
|
||||
};
|
||||
|
||||
template <class T> struct ChainedHasher
|
||||
{
|
||||
static void Append(Hasher &hasher, const T &value)
|
||||
{
|
||||
if constexpr (std::is_fundamental_v<T> || std::is_pointer_v<T>)
|
||||
{
|
||||
hasher.append(reinterpret_cast<const std::byte *>(&value), sizeof(T));
|
||||
}
|
||||
else
|
||||
{
|
||||
std::hash<T> h;
|
||||
auto v = h(value);
|
||||
hasher.append(reinterpret_cast<std::byte *>(&v), sizeof(v));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, class V> struct ChainedHasher<std::pair<T, V>>
|
||||
{
|
||||
static void Append(Hasher &hasher, const std::pair<T, V> &value)
|
||||
{
|
||||
ChainedHasher<T>::Append(hasher, value.first);
|
||||
ChainedHasher<V>::Append(hasher, value.second);
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, size_t N> struct ChainedHasher<std::array<T, N>>
|
||||
{
|
||||
static void Append(Hasher &hasher, const std::array<T, N> &value)
|
||||
{
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
ChainedHasher<T>::Append(hasher, value[i]);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template<class... Ts> struct ChainedHasher<std::tuple<Ts...>>
|
||||
{
|
||||
private:
|
||||
template <size_t N>
|
||||
static void AppendImpl(Hasher &hasher, const std::tuple<Ts...> &value)
|
||||
{
|
||||
ChainedHasher<std::decay_t<decltype(std::get<N>(value))>>::Append(
|
||||
hasher, std::get<N>(value));
|
||||
if constexpr (N + 1 < sizeof...(Ts))
|
||||
{
|
||||
AppendImpl<N + 1>(hasher, value);
|
||||
}
|
||||
}
|
||||
|
||||
public:
|
||||
static void Append(Hasher &hasher, const std::tuple<Ts...> &value)
|
||||
{
|
||||
if constexpr (sizeof...(Ts))
|
||||
{
|
||||
AppendImpl<0>(hasher, value);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/// Helper class for hashing std::pair of hashable types.
|
||||
struct PairHasher
|
||||
{
|
||||
template <class T, class V>
|
||||
size_t operator()(const std::pair<T, V> &v) const noexcept
|
||||
{
|
||||
Hasher hash;
|
||||
// chosen randomly with a 2^64-sided dice
|
||||
hash.init(0xfebd1fe69813c14full);
|
||||
ChainedHasher<std::pair<T, V>>::Append(hash, v);
|
||||
hash.finalize();
|
||||
return hash.data[1];
|
||||
}
|
||||
};
|
||||
|
||||
/// Helper class for hashing std::array of a hashable type.
|
||||
struct ArrayHasher
|
||||
{
|
||||
template <class T, size_t N>
|
||||
size_t operator()(const std::array<T, N> &v) const noexcept
|
||||
{
|
||||
Hasher hash;
|
||||
// chosen randomly with a 2^64-sided dice
|
||||
hash.init(0xfebd1fe69813c14full);
|
||||
ChainedHasher<std::array<T, N>>::Append(hash, v);
|
||||
hash.finalize();
|
||||
return hash.data[1];
|
||||
}
|
||||
};
|
||||
|
||||
/// Helper class for hashing std::tuple of hashable types.
|
||||
struct TupleHasher
|
||||
{
|
||||
template <class T>
|
||||
size_t operator()(const T &v) const noexcept
|
||||
{
|
||||
Hasher hash;
|
||||
// chosen randomly with a 2^64-sided dice
|
||||
hash.init(0xfebd1fe69813c14full);
|
||||
ChainedHasher<T>::Append(hash, v);
|
||||
hash.finalize();
|
||||
return hash.data[1];
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
@@ -20,9 +20,11 @@
|
||||
|
||||
#if defined(MFEM_USE_HIP) && defined(__HIP__)
|
||||
#define MFEM_USE_CUDA_OR_HIP
|
||||
constexpr bool mfem_use_gpu = true;
|
||||
#define MFEM_DEVICE __device__
|
||||
#define MFEM_HOST __host__
|
||||
#define MFEM_LAMBDA __host__ __device__
|
||||
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
|
||||
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
|
||||
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(hipDeviceSynchronize())
|
||||
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(hipStreamSynchronize(0))
|
||||
|
||||
@@ -55,6 +55,7 @@ list(APPEND HDRS
|
||||
dinvariants.hpp
|
||||
dtensor.hpp
|
||||
dual.hpp
|
||||
eigensolver.hpp
|
||||
filteredsolver.hpp
|
||||
handle.hpp
|
||||
invariants.hpp
|
||||
|
||||
@@ -0,0 +1,203 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
/**
|
||||
* @file eigensolver.hpp
|
||||
*
|
||||
* @brief This file contains a common interface for all eigensolver classes
|
||||
*/
|
||||
|
||||
#ifndef MFEM_EIGENSOLVER
|
||||
#define MFEM_EIGENSOLVER
|
||||
|
||||
#ifdef MFEM_HYPRE
|
||||
#include "hypre.hpp"
|
||||
#endif
|
||||
|
||||
#ifdef MFEM_SLEPC
|
||||
#include "slepc.hpp"
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
enum class EigenSolverType
|
||||
{
|
||||
HYPRE,
|
||||
SLEPC,
|
||||
INVALID_TYPE
|
||||
};
|
||||
|
||||
/// Provides base class for MFEM Eigensolvers
|
||||
class EigenSolverBase
|
||||
{
|
||||
public:
|
||||
EigenSolverBase() {}
|
||||
|
||||
/// Destructor
|
||||
virtual ~EigenSolverBase() = default;
|
||||
|
||||
/// Solves the eigenvalue problem
|
||||
virtual void Solve() = 0;
|
||||
|
||||
/// Set the required number of modes
|
||||
virtual void SetNumModes(int num_Modes)
|
||||
{
|
||||
numModes=num_Modes;
|
||||
}
|
||||
|
||||
/// @brief Set the operator to the eigenvalue problem
|
||||
/// @param A - operator
|
||||
virtual void SetOperator(Operator& A) = 0;
|
||||
|
||||
/// @brief Sets operators for the generalized eigenvalue problem
|
||||
/// @param A - operator
|
||||
/// @param M - mass matrix
|
||||
virtual void SetOperator(Operator& A, Operator& M)
|
||||
{
|
||||
MFEM_ABORT("Generalized eigensolver is not supported!");
|
||||
}
|
||||
|
||||
/// Optional method - sets preconditioner for the
|
||||
/// eigenvalue solver.
|
||||
virtual void SetPreconditioner(Solver& precond)
|
||||
{
|
||||
MFEM_ABORT("Preconditioner is not supported!");
|
||||
}
|
||||
|
||||
/// Returns the converged eigenvalues
|
||||
virtual void GetEigenvalues(Array<real_t>& eigen_vals) = 0;
|
||||
|
||||
/// Returns the vec_index eigenvector.
|
||||
virtual void GetEigenvector(int vec_index, Vector& vector) = 0;
|
||||
|
||||
/// Returns the eigensolver type.
|
||||
EigenSolverType GetSolverType() { return eigSolverType; }
|
||||
|
||||
protected:
|
||||
int numModes = 0;
|
||||
EigenSolverType eigSolverType = EigenSolverType::INVALID_TYPE;
|
||||
};
|
||||
|
||||
#ifdef MFEM_HYPRE
|
||||
class EigenSolverHypreLOBPCG : public EigenSolverBase
|
||||
{
|
||||
public:
|
||||
EigenSolverHypreLOBPCG(MPI_Comm comm)
|
||||
{
|
||||
eigenSolver = std::make_unique<HypreLOBPCG>(comm);
|
||||
eigSolverType = EigenSolverType::HYPRE;
|
||||
}
|
||||
|
||||
~EigenSolverHypreLOBPCG() {}
|
||||
|
||||
void Solve() override { eigenSolver->Solve(); }
|
||||
void SetNumModes(int num_Modes) override
|
||||
{
|
||||
eigenSolver->SetNumModes(num_Modes);
|
||||
numModes = num_Modes;
|
||||
}
|
||||
|
||||
void SetOperator(Operator& A) override { eigenSolver->SetOperator(A); }
|
||||
|
||||
void SetOperator(Operator& A, Operator& M) override
|
||||
{
|
||||
eigenSolver->SetOperator(A);
|
||||
eigenSolver->SetMassMatrix(M);
|
||||
}
|
||||
|
||||
void SetPreconditioner(Solver& precond) override { eigenSolver->SetPreconditioner(precond); }
|
||||
void GetEigenvalues(Array<real_t>& eigen_vals) override { eigenSolver->GetEigenvalues(eigen_vals); }
|
||||
void GetEigenvector(int vec_index, Vector& vector) override
|
||||
{
|
||||
const HypreParVector& eigenvec = eigenSolver->GetEigenvector(vec_index);
|
||||
vector = eigenvec;
|
||||
}
|
||||
|
||||
void SetTol(real_t tol) { eigenSolver->SetTol(tol); }
|
||||
void SetRelTol(real_t rel_tol) { eigenSolver->SetRelTol(rel_tol); }
|
||||
void SetMaxIter(int max_iter) { eigenSolver->SetMaxIter(max_iter); }
|
||||
void SetPrintLevel(int logging) { eigenSolver->SetPrintLevel(logging); }
|
||||
void SetRandomSeed(int seed) { eigenSolver->SetRandomSeed(seed); }
|
||||
void SetPrecondUsageMode(int usage_mode) { eigenSolver->SetPrecondUsageMode(usage_mode); }
|
||||
|
||||
private:
|
||||
std::unique_ptr<HypreLOBPCG> eigenSolver = nullptr;
|
||||
};
|
||||
#endif
|
||||
|
||||
#ifdef MFEM_SLEPC
|
||||
class EigenSolverSlepc : public EigenSolverBase
|
||||
{
|
||||
public:
|
||||
EigenSolverSlepc(MPI_Comm comm)
|
||||
{
|
||||
eigSolverType = EigenSolverType::SLEPC;
|
||||
eigenSolver = std::make_unique<SlepcEigenSolver>(comm);
|
||||
|
||||
eigenSolver->SetWhichEigenpairs(SlepcEigenSolver::TARGET_REAL);
|
||||
eigenSolver->SetTarget(0.0);
|
||||
eigenSolver->SetSpectralTransformation(SlepcEigenSolver::SHIFT_INVERT);
|
||||
}
|
||||
|
||||
~EigenSolverSlepc() {}
|
||||
|
||||
void Solve() override { eigenSolver->Solve(); }
|
||||
void SetNumModes(int num_Modes) override
|
||||
{
|
||||
eigenSolver->SetNumModes(num_Modes);
|
||||
numModes = num_Modes;
|
||||
}
|
||||
/// @brief Set the operator to the slepc eigenvalue problem. This method deep copies data to create a PetscParMatrix
|
||||
/// @param A - operator, must be of type HypreParMatrix.
|
||||
void SetOperator(Operator& A) override
|
||||
{
|
||||
petscMatA = std::make_unique<PetscParMatrix>
|
||||
(dynamic_cast<HypreParMatrix*>(&A));
|
||||
eigenSolver->SetOperator(*petscMatA);
|
||||
}
|
||||
/// @brief Set the operators to the slepc eigenvalue problem. This method deep copies data to create a PetscParMatrix
|
||||
/// @param A - operator, must be of type HypreParMatrix.
|
||||
/// @param M - operator, must be of type HypreParMatrix.
|
||||
void SetOperator(Operator& A, Operator& M) override
|
||||
{
|
||||
petscMatA = std::make_unique<PetscParMatrix>
|
||||
(dynamic_cast<const HypreParMatrix*>(&A));
|
||||
petscMatM = std::make_unique<PetscParMatrix>
|
||||
(dynamic_cast<const HypreParMatrix*>(&M));
|
||||
|
||||
eigenSolver->SetOperators(*petscMatA, *petscMatM);
|
||||
}
|
||||
void SetPreconditioner([[maybe_unused]] Solver& precond) override {}
|
||||
void GetEigenvalues(Array<real_t>& eigen_vals) override
|
||||
{
|
||||
eigen_vals.SetSize(numModes);
|
||||
for (int ik = 0; ik < numModes; ik++)
|
||||
{
|
||||
eigenSolver->GetEigenvalue(static_cast<unsigned int>(ik), eigen_vals[ik]);
|
||||
}
|
||||
}
|
||||
void GetEigenvector( int vec_index, Vector& vector) override
|
||||
{ eigenSolver->GetEigenvector(vec_index, vector); }
|
||||
|
||||
void SetTol(real_t tol) { eigenSolver->SetTol(tol); }
|
||||
void SetMaxIter(int max_iter) { eigenSolver->SetMaxIter(max_iter); }
|
||||
|
||||
private:
|
||||
std::unique_ptr<SlepcEigenSolver> eigenSolver = nullptr;
|
||||
std::unique_ptr<PetscParMatrix> petscMatA = nullptr;
|
||||
std::unique_ptr<PetscParMatrix> petscMatM = nullptr;
|
||||
};
|
||||
#endif
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
@@ -3634,12 +3634,25 @@ void HypreSmoother::SetType(HypreSmoother::Type type_, int relax_times_)
|
||||
relax_times = relax_times_;
|
||||
}
|
||||
|
||||
void HypreSmoother::GetType(HypreSmoother::Type &type_, int &relax_times_) const
|
||||
{
|
||||
type_ = static_cast<HypreSmoother::Type>(type);
|
||||
relax_times_ = relax_times;
|
||||
}
|
||||
|
||||
void HypreSmoother::SetSOROptions(real_t relax_weight_, real_t omega_)
|
||||
{
|
||||
relax_weight = relax_weight_;
|
||||
omega = omega_;
|
||||
}
|
||||
|
||||
void HypreSmoother::GetSOROptions(real_t &relax_weight_, real_t &omega_) const
|
||||
{
|
||||
// TODO: are these used for all smoother types?
|
||||
relax_weight_ = relax_weight;
|
||||
omega_ = omega;
|
||||
}
|
||||
|
||||
void HypreSmoother::SetPolyOptions(int poly_order_, real_t poly_fraction_,
|
||||
int eig_est_cg_iter_)
|
||||
{
|
||||
@@ -3648,6 +3661,15 @@ void HypreSmoother::SetPolyOptions(int poly_order_, real_t poly_fraction_,
|
||||
eig_est_cg_iter = eig_est_cg_iter_;
|
||||
}
|
||||
|
||||
void HypreSmoother::GetPolyOptions(int &poly_order_, real_t &poly_fraction_,
|
||||
int &eig_est_cg_iter_) const
|
||||
{
|
||||
// TODO: are these used for all smoother types?
|
||||
poly_order_ = poly_order;
|
||||
poly_fraction_ = poly_fraction;
|
||||
eig_est_cg_iter_ = eig_est_cg_iter;
|
||||
}
|
||||
|
||||
void HypreSmoother::SetTaubinOptions(real_t lambda_, real_t mu_,
|
||||
int taubin_iter_)
|
||||
{
|
||||
@@ -3656,6 +3678,14 @@ void HypreSmoother::SetTaubinOptions(real_t lambda_, real_t mu_,
|
||||
taubin_iter = taubin_iter_;
|
||||
}
|
||||
|
||||
void HypreSmoother::GetTaubinOptions(real_t &lambda_, real_t &mu_,
|
||||
int &taubin_iter_) const
|
||||
{
|
||||
lambda_ = lambda;
|
||||
mu_ = mu;
|
||||
taubin_iter_ = taubin_iter;
|
||||
}
|
||||
|
||||
void HypreSmoother::SetWindowByName(const char* name)
|
||||
{
|
||||
real_t a = -1, b, c;
|
||||
@@ -3678,6 +3708,13 @@ void HypreSmoother::SetWindowParameters(real_t a, real_t b, real_t c)
|
||||
window_params[2] = c;
|
||||
}
|
||||
|
||||
void HypreSmoother::GetWindowParameters(real_t &a, real_t &b, real_t &c) const
|
||||
{
|
||||
a = window_params[0];
|
||||
b = window_params[1];
|
||||
c = window_params[2];
|
||||
}
|
||||
|
||||
void HypreSmoother::SetOperator(const Operator &op)
|
||||
{
|
||||
A = const_cast<HypreParMatrix *>(dynamic_cast<const HypreParMatrix *>(&op));
|
||||
@@ -4173,12 +4210,20 @@ HypreSolver::~HypreSolver()
|
||||
auxX.Delete();
|
||||
}
|
||||
|
||||
void HyprePCG::SetDefaultOptions()
|
||||
{
|
||||
// Explicitly set just in case past/future versions of hypre change the
|
||||
// defaults
|
||||
SetTol(1e-6);
|
||||
SetMaxIter(1000);
|
||||
}
|
||||
|
||||
HyprePCG::HyprePCG(MPI_Comm comm) : precond(NULL)
|
||||
{
|
||||
iterative_mode = true;
|
||||
|
||||
HYPRE_ParCSRPCGCreate(comm, &pcg_solver);
|
||||
SetDefaultOptions();
|
||||
}
|
||||
|
||||
HyprePCG::HyprePCG(const HypreParMatrix &A_) : HypreSolver(&A_), precond(NULL)
|
||||
@@ -4190,6 +4235,7 @@ HyprePCG::HyprePCG(const HypreParMatrix &A_) : HypreSolver(&A_), precond(NULL)
|
||||
HYPRE_ParCSRMatrixGetComm(*A, &comm);
|
||||
|
||||
HYPRE_ParCSRPCGCreate(comm, &pcg_solver);
|
||||
SetDefaultOptions();
|
||||
}
|
||||
|
||||
void HyprePCG::SetOperator(const Operator &op)
|
||||
@@ -4214,21 +4260,54 @@ void HyprePCG::SetOperator(const Operator &op)
|
||||
auxX.Delete(); auxX.Reset();
|
||||
}
|
||||
|
||||
void HyprePCG::SetUseTwoNorm(bool val)
|
||||
{
|
||||
HYPRE_PCGSetTwoNorm(pcg_solver, val);
|
||||
}
|
||||
|
||||
bool HyprePCG::GetUseTwoNorm() const
|
||||
{
|
||||
HYPRE_Int val;
|
||||
HYPRE_PCGGetTwoNorm(pcg_solver, &val);
|
||||
return val != 0;
|
||||
}
|
||||
|
||||
void HyprePCG::SetTol(real_t tol)
|
||||
{
|
||||
HYPRE_PCGSetTol(pcg_solver, tol);
|
||||
}
|
||||
|
||||
real_t HyprePCG::GetTol() const
|
||||
{
|
||||
HYPRE_Real tol;
|
||||
HYPRE_PCGGetTol(pcg_solver, &tol);
|
||||
return tol;
|
||||
}
|
||||
|
||||
void HyprePCG::SetAbsTol(real_t atol)
|
||||
{
|
||||
HYPRE_PCGSetAbsoluteTol(pcg_solver, atol);
|
||||
}
|
||||
|
||||
real_t HyprePCG::GetAbsTol() const
|
||||
{
|
||||
HYPRE_Real atol;
|
||||
hypre_PCGGetAbsoluteTol(pcg_solver, &atol);
|
||||
return atol;
|
||||
}
|
||||
|
||||
void HyprePCG::SetMaxIter(int max_iter)
|
||||
{
|
||||
HYPRE_PCGSetMaxIter(pcg_solver, max_iter);
|
||||
}
|
||||
|
||||
int HyprePCG::GetMaxIter() const
|
||||
{
|
||||
HYPRE_Int max_iter;
|
||||
HYPRE_PCGGetMaxIter(pcg_solver, &max_iter);
|
||||
return max_iter;
|
||||
}
|
||||
|
||||
void HyprePCG::SetLogging(int logging)
|
||||
{
|
||||
HYPRE_PCGSetLogging(pcg_solver, logging);
|
||||
@@ -4344,6 +4423,20 @@ HyprePCG::~HyprePCG()
|
||||
HYPRE_ParCSRPCGDestroy(pcg_solver);
|
||||
}
|
||||
|
||||
#if MFEM_HYPRE_VERSION >= 21500
|
||||
HypreParVector HyprePCG::GetResiduals() const
|
||||
{
|
||||
HYPRE_ParVector r;
|
||||
HYPRE_ParCSRPCGGetResidual(pcg_solver, &r);
|
||||
return HypreParVector(r);
|
||||
}
|
||||
|
||||
void HyprePCG::GetFinalAbsResidualNorm(real_t &final_res_norm, real_t p) const
|
||||
{
|
||||
auto r = GetResiduals();
|
||||
ParNormlp(r, p, r.GetComm());
|
||||
}
|
||||
#endif
|
||||
|
||||
HypreGMRES::HypreGMRES(MPI_Comm comm) : precond(NULL)
|
||||
{
|
||||
@@ -4399,26 +4492,69 @@ void HypreGMRES::SetOperator(const Operator &op)
|
||||
auxX.Delete(); auxX.Reset();
|
||||
}
|
||||
|
||||
#if MFEM_HYPRE_VERSION >= 21500
|
||||
HypreParVector HypreGMRES::GetResiduals() const
|
||||
{
|
||||
HYPRE_ParVector r;
|
||||
HYPRE_ParCSRGMRESGetResidual(gmres_solver, &r);
|
||||
return HypreParVector(r);
|
||||
}
|
||||
|
||||
void HypreGMRES::GetFinalAbsResidualNorm(real_t &final_res_norm, real_t p) const
|
||||
{
|
||||
auto r = GetResiduals();
|
||||
ParNormlp(r, p, r.GetComm());
|
||||
}
|
||||
#endif
|
||||
|
||||
void HypreGMRES::SetTol(real_t tol)
|
||||
{
|
||||
HYPRE_GMRESSetTol(gmres_solver, tol);
|
||||
}
|
||||
|
||||
real_t HypreGMRES::GetTol()const
|
||||
{
|
||||
HYPRE_Real tol;
|
||||
HYPRE_GMRESGetTol(gmres_solver, &tol);
|
||||
return tol;
|
||||
}
|
||||
|
||||
void HypreGMRES::SetAbsTol(real_t tol)
|
||||
{
|
||||
HYPRE_GMRESSetAbsoluteTol(gmres_solver, tol);
|
||||
}
|
||||
|
||||
real_t HypreGMRES::GetAbsTol() const
|
||||
{
|
||||
HYPRE_Real atol;
|
||||
HYPRE_GMRESGetAbsoluteTol(gmres_solver, &atol);
|
||||
return atol;
|
||||
}
|
||||
|
||||
void HypreGMRES::SetMaxIter(int max_iter)
|
||||
{
|
||||
HYPRE_GMRESSetMaxIter(gmres_solver, max_iter);
|
||||
}
|
||||
|
||||
int HypreGMRES::GetMaxIter() const
|
||||
{
|
||||
HYPRE_Int max_iter;
|
||||
HYPRE_GMRESGetMaxIter(gmres_solver, &max_iter);
|
||||
return max_iter;
|
||||
}
|
||||
|
||||
void HypreGMRES::SetKDim(int k_dim)
|
||||
{
|
||||
HYPRE_GMRESSetKDim(gmres_solver, k_dim);
|
||||
}
|
||||
|
||||
int HypreGMRES::GetKDim() const
|
||||
{
|
||||
HYPRE_Int k_dim;
|
||||
HYPRE_GMRESGetKDim(gmres_solver, &k_dim);
|
||||
return k_dim;
|
||||
}
|
||||
|
||||
void HypreGMRES::SetLogging(int logging)
|
||||
{
|
||||
HYPRE_GMRESSetLogging(gmres_solver, logging);
|
||||
@@ -4576,16 +4712,37 @@ void HypreFGMRES::SetTol(real_t tol)
|
||||
HYPRE_ParCSRFlexGMRESSetTol(fgmres_solver, tol);
|
||||
}
|
||||
|
||||
real_t HypreFGMRES::GetTol() const
|
||||
{
|
||||
HYPRE_Real tol;
|
||||
HYPRE_FlexGMRESGetTol(fgmres_solver, &tol);
|
||||
return tol;
|
||||
}
|
||||
|
||||
void HypreFGMRES::SetMaxIter(int max_iter)
|
||||
{
|
||||
HYPRE_ParCSRFlexGMRESSetMaxIter(fgmres_solver, max_iter);
|
||||
}
|
||||
|
||||
int HypreFGMRES::GetMaxIter() const
|
||||
{
|
||||
HYPRE_Int max_iter;
|
||||
HYPRE_FlexGMRESGetMaxIter(fgmres_solver, &max_iter);
|
||||
return max_iter;
|
||||
}
|
||||
|
||||
void HypreFGMRES::SetKDim(int k_dim)
|
||||
{
|
||||
HYPRE_ParCSRFlexGMRESSetKDim(fgmres_solver, k_dim);
|
||||
}
|
||||
|
||||
int HypreFGMRES::GetKDim() const
|
||||
{
|
||||
HYPRE_Int k_dim;
|
||||
HYPRE_FlexGMRESGetKDim(fgmres_solver, &k_dim);
|
||||
return k_dim;
|
||||
}
|
||||
|
||||
void HypreFGMRES::SetLogging(int logging)
|
||||
{
|
||||
HYPRE_ParCSRFlexGMRESSetLogging(fgmres_solver, logging);
|
||||
@@ -4682,6 +4839,21 @@ HypreFGMRES::~HypreFGMRES()
|
||||
HYPRE_ParCSRFlexGMRESDestroy(fgmres_solver);
|
||||
}
|
||||
|
||||
#if MFEM_HYPRE_VERSION >= 21500
|
||||
HypreParVector HypreFGMRES::GetResiduals() const
|
||||
{
|
||||
HYPRE_ParVector r;
|
||||
HYPRE_ParCSRFlexGMRESGetResidual(fgmres_solver, &r);
|
||||
return HypreParVector(r);
|
||||
}
|
||||
|
||||
void HypreFGMRES::GetFinalAbsResidualNorm(real_t &final_res_norm,
|
||||
real_t p) const
|
||||
{
|
||||
auto r = GetResiduals();
|
||||
ParNormlp(r, p, r.GetComm());
|
||||
}
|
||||
#endif
|
||||
|
||||
void HypreDiagScale::SetOperator(const Operator &op)
|
||||
{
|
||||
@@ -5170,6 +5342,13 @@ void HypreBoomerAMG::ResetAMGPrecond()
|
||||
}
|
||||
}
|
||||
|
||||
int HypreBoomerAMG::GetMaxIter() const
|
||||
{
|
||||
HYPRE_Int max_iter;
|
||||
HYPRE_BoomerAMGGetMaxIter(amg_precond, &max_iter);
|
||||
return max_iter;
|
||||
}
|
||||
|
||||
void HypreBoomerAMG::SetOperator(const Operator &op)
|
||||
{
|
||||
const HypreParMatrix *new_A = dynamic_cast<const HypreParMatrix *>(&op);
|
||||
|
||||
+97
-7
@@ -1160,6 +1160,15 @@ public:
|
||||
return HypreUsingGPU() ? l1Jacobi : l1GS;
|
||||
}
|
||||
|
||||
/// Default solver settings:
|
||||
/// type = DefaultType()
|
||||
/// relax_times = 1
|
||||
/// omega = 1.0
|
||||
/// poly_order = 2
|
||||
/// poly_fraction = 0.3
|
||||
/// lambda = 0.5
|
||||
/// mu = -0.5
|
||||
/// taubin_iter = 40
|
||||
HypreSmoother();
|
||||
|
||||
HypreSmoother(const HypreParMatrix &A_, int type = DefaultType(),
|
||||
@@ -1169,20 +1178,28 @@ public:
|
||||
|
||||
/// Set the relaxation type and number of sweeps
|
||||
void SetType(HypreSmoother::Type type, int relax_times = 1);
|
||||
using Operator::GetType;
|
||||
void GetType(HypreSmoother::Type &type, int &relax_times) const;
|
||||
/// Set SOR-related parameters
|
||||
void SetSOROptions(real_t relax_weight, real_t omega);
|
||||
void GetSOROptions(real_t &relax_weight, real_t &omega) const;
|
||||
|
||||
/// Set parameters for polynomial smoothing
|
||||
/** By default, 10 iterations of CG are used to estimate the eigenvalues.
|
||||
Setting eig_est_cg_iter = 0 uses hypre's hypre_ParCSRMaxEigEstimate() instead. */
|
||||
void SetPolyOptions(int poly_order, real_t poly_fraction,
|
||||
int eig_est_cg_iter = 10);
|
||||
void GetPolyOptions(int &poly_order, real_t &poly_fraction,
|
||||
int &eig_est_cg_iter) const;
|
||||
/// Set parameters for Taubin's lambda-mu method
|
||||
void SetTaubinOptions(real_t lambda, real_t mu, int iter);
|
||||
void GetTaubinOptions(real_t &lambda, real_t &mu, int &iter) const;
|
||||
|
||||
/// Convenience function for setting canonical windowing parameters
|
||||
void SetWindowByName(const char* window_name);
|
||||
/// Set parameters for windowing function for FIR smoother.
|
||||
void SetWindowParameters(real_t a, real_t b, real_t c);
|
||||
void GetWindowParameters(real_t &a, real_t &b, real_t &c) const;
|
||||
/// Compute window and Chebyshev coefficients for given polynomial order.
|
||||
void SetFIRCoefficients(real_t max_eig);
|
||||
|
||||
@@ -1190,12 +1207,15 @@ public:
|
||||
/** By default, the l1-norms take their sign from the corresponding diagonal
|
||||
entries in the associated matrix. */
|
||||
void SetPositiveDiagonal(bool pos = true) { pos_l1_norms = pos; }
|
||||
bool IsPositiveDiagonal() const { return pos_l1_norms; };
|
||||
|
||||
/** Explicitly indicate whether the linear system matrix A is symmetric. If A
|
||||
is symmetric, the smoother will also be symmetric. In this case, calling
|
||||
MultTranspose will be redirected to Mult. (This is also done if the
|
||||
smoother is diagonal.) By default, A is assumed to be nonsymmetric. */
|
||||
void SetOperatorSymmetry(bool is_sym) { A_is_symmetric = is_sym; }
|
||||
/// @return true if the smoother assumes A is symmetric, false otherwise
|
||||
bool IsOperatorSymmetric() const { return A_is_symmetric; }
|
||||
|
||||
/** Set/update the associated operator. Must be called after setting the
|
||||
HypreSmoother type and options. */
|
||||
@@ -1327,6 +1347,7 @@ public:
|
||||
#endif
|
||||
|
||||
/// PCG solver in hypre
|
||||
/// Defaults to (relative) tol=1e-6, atol=0, max_iter=1000
|
||||
class HyprePCG : public HypreSolver
|
||||
{
|
||||
private:
|
||||
@@ -1334,6 +1355,9 @@ private:
|
||||
|
||||
HypreSolver * precond;
|
||||
|
||||
/// Default PCG options
|
||||
void SetDefaultOptions();
|
||||
|
||||
public:
|
||||
HyprePCG(MPI_Comm comm);
|
||||
|
||||
@@ -1342,8 +1366,11 @@ public:
|
||||
void SetOperator(const Operator &op) override;
|
||||
|
||||
void SetTol(real_t tol);
|
||||
real_t GetTol() const;
|
||||
void SetAbsTol(real_t atol);
|
||||
real_t GetAbsTol() const;
|
||||
void SetMaxIter(int max_iter);
|
||||
int GetMaxIter() const;
|
||||
void SetLogging(int logging);
|
||||
void SetPrintLevel(int print_lvl);
|
||||
|
||||
@@ -1368,12 +1395,32 @@ public:
|
||||
num_iterations = internal::to_int(num_it);
|
||||
}
|
||||
|
||||
/// Gets the relative residual norm
|
||||
void GetFinalResidualNorm(real_t &final_res_norm) const
|
||||
{
|
||||
HYPRE_ParCSRPCGGetFinalRelativeResidualNorm(pcg_solver,
|
||||
&final_res_norm);
|
||||
}
|
||||
|
||||
/// @param[in] use
|
||||
/// Convergence criterion:
|
||||
/// - when true: (r, r) < max(r_tol^2 (b, b), a_tol^2)
|
||||
/// - when false: (r, A r) < max(r_tol^2 (b, A b), a_tol^2)
|
||||
/// @sa HYPRE_PCGSetTwoNorm
|
||||
void SetUseTwoNorm(bool use);
|
||||
|
||||
/// @sa HYPRE_PCGGetTwoNorm
|
||||
bool GetUseTwoNorm() const;
|
||||
|
||||
#if MFEM_HYPRE_VERSION >= 21500
|
||||
/// Gets the internal Hypre solver residual vector.
|
||||
/// @sa HYPRE_ParCSRPCGGetResidual
|
||||
HypreParVector GetResiduals() const;
|
||||
|
||||
/// Computes the absolute residual p-norm.
|
||||
void GetFinalAbsResidualNorm(real_t &final_res_norm, real_t p = 2) const;
|
||||
#endif
|
||||
|
||||
/// The typecast to HYPRE_Solver returns the internal pcg_solver
|
||||
operator HYPRE_Solver() const override { return pcg_solver; }
|
||||
|
||||
@@ -1391,7 +1438,8 @@ public:
|
||||
virtual ~HyprePCG();
|
||||
};
|
||||
|
||||
/// GMRES solver in hypre
|
||||
/// GMRES solver in hypre.
|
||||
/// Defaults to k=50, (relative) tol=1e-6, atol=0, max_iter=100.
|
||||
class HypreGMRES : public HypreSolver
|
||||
{
|
||||
private:
|
||||
@@ -1410,9 +1458,13 @@ public:
|
||||
void SetOperator(const Operator &op) override;
|
||||
|
||||
void SetTol(real_t tol);
|
||||
real_t GetTol() const;
|
||||
void SetAbsTol(real_t tol);
|
||||
real_t GetAbsTol() const;
|
||||
void SetMaxIter(int max_iter);
|
||||
int GetMaxIter() const;
|
||||
void SetKDim(int dim);
|
||||
int GetKDim() const;
|
||||
void SetLogging(int logging);
|
||||
void SetPrintLevel(int print_lvl);
|
||||
|
||||
@@ -1432,12 +1484,22 @@ public:
|
||||
num_iterations = internal::to_int(num_it);
|
||||
}
|
||||
|
||||
/// Gets the relative residual norm
|
||||
void GetFinalResidualNorm(real_t &final_res_norm) const
|
||||
{
|
||||
HYPRE_ParCSRGMRESGetFinalRelativeResidualNorm(gmres_solver,
|
||||
&final_res_norm);
|
||||
}
|
||||
|
||||
#if MFEM_HYPRE_VERSION >= 21500
|
||||
/// Gets the internal Hypre solver residual vector.
|
||||
/// @sa HYPRE_ParCSRGMRESGetResidual
|
||||
HypreParVector GetResiduals() const;
|
||||
|
||||
/// Computes the absolute residual p-norm.
|
||||
void GetFinalAbsResidualNorm(real_t &final_res_norm, real_t p = 2) const;
|
||||
#endif
|
||||
|
||||
/// The typecast to HYPRE_Solver returns the internal gmres_solver
|
||||
operator HYPRE_Solver() const override { return gmres_solver; }
|
||||
|
||||
@@ -1455,7 +1517,8 @@ public:
|
||||
virtual ~HypreGMRES();
|
||||
};
|
||||
|
||||
/// Flexible GMRES solver in hypre
|
||||
/// Flexible GMRES solver in hypre.
|
||||
/// Defaults to k=50, (relative) tol=1e-6, max_iter=100.
|
||||
class HypreFGMRES : public HypreSolver
|
||||
{
|
||||
private:
|
||||
@@ -1474,8 +1537,11 @@ public:
|
||||
void SetOperator(const Operator &op) override;
|
||||
|
||||
void SetTol(real_t tol);
|
||||
real_t GetTol() const;
|
||||
void SetMaxIter(int max_iter);
|
||||
int GetMaxIter() const;
|
||||
void SetKDim(int dim);
|
||||
int GetKDim() const;
|
||||
void SetLogging(int logging);
|
||||
void SetPrintLevel(int print_lvl);
|
||||
|
||||
@@ -1495,12 +1561,22 @@ public:
|
||||
num_iterations = internal::to_int(num_it);
|
||||
}
|
||||
|
||||
/// Gets the relative residual norm
|
||||
void GetFinalResidualNorm(real_t &final_res_norm) const
|
||||
{
|
||||
HYPRE_ParCSRFlexGMRESGetFinalRelativeResidualNorm(fgmres_solver,
|
||||
&final_res_norm);
|
||||
}
|
||||
|
||||
#if MFEM_HYPRE_VERSION >= 21500
|
||||
/// Gets the internal Hypre solver residual vector.
|
||||
/// @sa HYPRE_ParCSRFlexGMRESGetResidual
|
||||
HypreParVector GetResiduals() const;
|
||||
|
||||
/// Computes the absolute residual p-norm.
|
||||
void GetFinalAbsResidualNorm(real_t &final_res_norm, real_t p = 2) const;
|
||||
#endif
|
||||
|
||||
/// The typecast to HYPRE_Solver returns the internal fgmres_solver
|
||||
operator HYPRE_Solver() const override { return fgmres_solver; }
|
||||
|
||||
@@ -1556,7 +1632,8 @@ public:
|
||||
virtual ~HypreDiagScale() { }
|
||||
};
|
||||
|
||||
/// The ParaSails preconditioner in hypre
|
||||
/// The ParaSails preconditioner in hypre.
|
||||
/// See SetDefaultOptions() for default solver options.
|
||||
class HypreParaSails : public HypreSolver
|
||||
{
|
||||
private:
|
||||
@@ -1685,10 +1762,14 @@ public:
|
||||
/**
|
||||
@brief Wrapper for Hypre's native parallel ILU preconditioner.
|
||||
|
||||
The default ILU factorization type is ILU(k). If you need to change this, or
|
||||
any other option, you can use the HYPRE_Solver method to cast the object for use
|
||||
with Hypre's native functions. For example, if want to use natural ordering
|
||||
rather than RCM reordering, you can use the following approach:
|
||||
Default parameters: ILU(k) factorization type, tol=0.0 (for use as a
|
||||
preconditioner), fill level = 1 (for ILU(k)), reverse Cuthill-McKee (RCM)
|
||||
re-ordering.
|
||||
|
||||
If you need to change this, or any other option, you can use the HYPRE_Solver
|
||||
method to cast the object for use with Hypre's native functions. For example, if
|
||||
want to use natural ordering rather than RCM reordering, you can use the
|
||||
following approach:
|
||||
|
||||
@code
|
||||
mfem::HypreILU ilu();
|
||||
@@ -1829,6 +1910,7 @@ public:
|
||||
|
||||
void SetMaxIter(int max_iter)
|
||||
{ HYPRE_BoomerAMGSetMaxIter(amg_precond, max_iter); }
|
||||
int GetMaxIter() const;
|
||||
|
||||
/// Expert option - consult hypre documentation/team
|
||||
void SetMaxLevels(int max_levels)
|
||||
@@ -1853,6 +1935,8 @@ public:
|
||||
/// Expert option - consult hypre documentation/team
|
||||
void SetRelaxType(int relax_type)
|
||||
{ HYPRE_BoomerAMGSetRelaxType(amg_precond, relax_type); }
|
||||
// not implemented in hypre
|
||||
// int GetRelaxType() const;
|
||||
|
||||
/// Expert option - consult hypre documentation/team
|
||||
void SetCycleType(int cycle_type)
|
||||
@@ -2153,8 +2237,14 @@ public:
|
||||
~HypreLOBPCG();
|
||||
|
||||
void SetTol(real_t tol);
|
||||
// not implemented in HYPRE
|
||||
// real_t GetTol() const;
|
||||
void SetRelTol(real_t rel_tol);
|
||||
// not implemented in HYPRE
|
||||
// real_t GetRelTol() const;
|
||||
void SetMaxIter(int max_iter);
|
||||
// not implemented in HYPRE
|
||||
// int GetMaxIter() const;
|
||||
void SetPrintLevel(int logging);
|
||||
void SetNumModes(int num_eigs) { nev = num_eigs; }
|
||||
void SetPrecondUsageMode(int pcg_mode);
|
||||
|
||||
@@ -3639,12 +3639,20 @@ void PetscBDDCSolver::BDDCSolverConstructor(const PetscBDDCSolverParams &opts)
|
||||
// make sure ess/nat_dof have been collectively set
|
||||
PetscBool lpr = PETSC_FALSE,pr;
|
||||
if (opts.ess_dof) { lpr = PETSC_TRUE; }
|
||||
#if PETSC_VERSION_LT(3,24,0)
|
||||
mpiierr = MPI_Allreduce(&lpr,&pr,1,MPIU_BOOL,MPI_LOR,comm);
|
||||
#else
|
||||
mpiierr = MPI_Allreduce(&lpr,&pr,1,MPI_C_BOOL,MPI_LOR,comm);
|
||||
#endif
|
||||
CCHKERRQ(comm,mpiierr);
|
||||
MFEM_VERIFY(lpr == pr,"ess_dof should be collectively set");
|
||||
lpr = PETSC_FALSE;
|
||||
if (opts.nat_dof) { lpr = PETSC_TRUE; }
|
||||
#if PETSC_VERSION_LT(3,24,0)
|
||||
mpiierr = MPI_Allreduce(&lpr,&pr,1,MPIU_BOOL,MPI_LOR,comm);
|
||||
#else
|
||||
mpiierr = MPI_Allreduce(&lpr,&pr,1,MPI_C_BOOL,MPI_LOR,comm);
|
||||
#endif
|
||||
CCHKERRQ(comm,mpiierr);
|
||||
MFEM_VERIFY(lpr == pr,"nat_dof should be collectively set");
|
||||
// make sure fields have been collectively set
|
||||
@@ -4058,8 +4066,13 @@ void PetscNonlinearSolver::SetOperator(const Operator &op)
|
||||
ls = (PetscBool)(height == op.Height() && width == op.Width() &&
|
||||
(void*)&op == fctx &&
|
||||
(void*)&op == jctx);
|
||||
#if PETSC_VERSION_LT(3,24,0)
|
||||
mpiierr = MPI_Allreduce(&ls,&gs,1,MPIU_BOOL,MPI_LAND,
|
||||
PetscObjectComm((PetscObject)snes));
|
||||
#else
|
||||
mpiierr = MPI_Allreduce(&ls,&gs,1,MPI_C_BOOL,MPI_LAND,
|
||||
PetscObjectComm((PetscObject)snes));
|
||||
#endif
|
||||
CCHKERRQ(PetscObjectComm((PetscObject)snes),mpiierr);
|
||||
if (!gs)
|
||||
{
|
||||
|
||||
@@ -1066,6 +1066,11 @@ void SparseMatrix::BooleanMultTranspose(const Array<int> &x,
|
||||
y.SetSize(Width());
|
||||
y = 0;
|
||||
|
||||
HostReadI();
|
||||
HostReadJ();
|
||||
x.HostRead();
|
||||
y.HostReadWrite();
|
||||
|
||||
for (int i = 0; i < Height(); i++)
|
||||
{
|
||||
if (x[i])
|
||||
|
||||
+12
-1
@@ -363,14 +363,19 @@ void SuperLUSolver::Init(MPI_Comm comm)
|
||||
// Set default options:
|
||||
// options.Fact = DOFACT;
|
||||
// options.Equil = YES;
|
||||
// options.ParSymbFact = NO;
|
||||
// options.ColPerm = METIS_AT_PLUS_A;
|
||||
// options.RowPerm = LargeDiag_MC64;
|
||||
// options.ReplaceTinyPivot = NO;
|
||||
// options.Trans = NOTRANS;
|
||||
// options.IterRefine = SLU_DOUBLE;
|
||||
// options.Trans = NOTRANS;
|
||||
// options.SolveInitialized = NO;
|
||||
// options.RefineInitialized = NO;
|
||||
// options.PrintStat = YES;
|
||||
// options.lookahead_etree = NO;
|
||||
// options.num_lookaheads = 10;
|
||||
// options.superlu_acc_offload = 1;
|
||||
// options.SymPattern = NO;
|
||||
superlu_dist_options_t *options = (superlu_dist_options_t *)optionsPtr_;
|
||||
set_default_options_dist(options);
|
||||
#if SUPERLU_DIST_MAJOR_VERSION > 7 || \
|
||||
@@ -472,6 +477,12 @@ void SuperLUSolver::SetFact(superlu::Fact fact)
|
||||
options->Fact = opt;
|
||||
}
|
||||
|
||||
void SuperLUSolver::SetDeviceOffload(bool offload)
|
||||
{
|
||||
superlu_dist_options_t *options = (superlu_dist_options_t *)optionsPtr_;
|
||||
options->superlu_acc_offload = offload;
|
||||
}
|
||||
|
||||
void SuperLUSolver::SetOperator(const Operator &op)
|
||||
{
|
||||
// Verify that we have a compatible operator
|
||||
|
||||
+6
-1
@@ -250,7 +250,8 @@ public:
|
||||
work (default false) */
|
||||
void SetSymmetricPattern(bool sym);
|
||||
|
||||
/** @brief Specify whether to perform parallel symbolic factorization.
|
||||
/** @brief Specify whether to perform parallel symbolic factorization
|
||||
(default false)
|
||||
@note If true SuperLU will use superlu::PARMETIS for the Column
|
||||
Permutation regardless of the setting */
|
||||
void SetParSymbFact(bool par);
|
||||
@@ -263,6 +264,10 @@ public:
|
||||
superlu::FACTORED*/
|
||||
void SetFact(superlu::Fact fact);
|
||||
|
||||
/** @brief Specify whether to offload numerical factorization onto the device
|
||||
(default true if SuperLU_DIST has been compiled with GPU support) */
|
||||
void SetDeviceOffload(bool offload);
|
||||
|
||||
// Processor grid for SuperLU_DIST.
|
||||
const int nprow_, npcol_, npdep_;
|
||||
|
||||
|
||||
@@ -794,7 +794,6 @@ status info:
|
||||
$(info MFEM_MPI_NP = $(MFEM_MPI_NP))
|
||||
@true
|
||||
|
||||
ASTYLE_BIN = astyle
|
||||
ASTYLE = $(ASTYLE_BIN) --options=$(SRC)config/mfem.astylerc
|
||||
ASTYLE_VER = "Artistic Style Version 3.1"
|
||||
FORMAT_FILES = $(foreach dir,$(DIRS) $(EM_DIRS) config,$(dir)/*.?pp)
|
||||
|
||||
+15
-10
@@ -2078,12 +2078,13 @@ public:
|
||||
contrary to the ones obtained through Mesh::GetFacesElements and can
|
||||
directly be used, e.g., Elem1 and Elem2 indices.
|
||||
Likewise the orientations for Elem1 and Elem2 already take into account
|
||||
special cases and can be used as is.
|
||||
*/
|
||||
special cases and can be used as is. */
|
||||
struct FaceInformation
|
||||
{
|
||||
/// The face topology (boundary, conforming, or nonconforming).
|
||||
FaceTopology topology;
|
||||
|
||||
/// Information about the adjacent elements.
|
||||
struct
|
||||
{
|
||||
ElementLocation location;
|
||||
@@ -2093,8 +2094,13 @@ public:
|
||||
int orientation;
|
||||
} element[2];
|
||||
|
||||
/// Detailed face information (see FaceInfoTag).
|
||||
FaceInfoTag tag;
|
||||
|
||||
/// If the face is nonconforming, the index of the NC face. -1 otherwise.
|
||||
int ncface;
|
||||
|
||||
/// The point matrix for nonconforming faces.
|
||||
const DenseMatrix* point_matrix;
|
||||
|
||||
/** @brief Return true if the face is a local interior face which is NOT
|
||||
@@ -2113,21 +2119,20 @@ public:
|
||||
|
||||
/** @brief return true if the face is an interior face to the computation
|
||||
domain, either a local or shared interior face (not a boundary face)
|
||||
which is NOT a master nonconforming face.
|
||||
*/
|
||||
which is NOT a master nonconforming face. */
|
||||
bool IsInterior() const
|
||||
{
|
||||
return topology == FaceTopology::Conforming ||
|
||||
topology == FaceTopology::Nonconforming;
|
||||
}
|
||||
|
||||
/** @brief Return true if the face is a boundary face. */
|
||||
/// Return true if the face is a boundary face.
|
||||
bool IsBoundary() const
|
||||
{
|
||||
return topology == FaceTopology::Boundary;
|
||||
}
|
||||
|
||||
/// @brief Return true if the face is of the same type as @a type.
|
||||
/// Return true if the face is of the same type as @a type.
|
||||
bool IsOfFaceType(FaceType type) const
|
||||
{
|
||||
switch (type)
|
||||
@@ -2141,13 +2146,13 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
/// @brief Return true if the face is a conforming face.
|
||||
/// Return true if the face is a conforming face.
|
||||
bool IsConforming() const
|
||||
{
|
||||
return topology == FaceTopology::Conforming;
|
||||
}
|
||||
|
||||
/// @brief Return true if the face is a nonconforming fine face.
|
||||
/// Return true if the face is a nonconforming fine face.
|
||||
bool IsNonconformingFine() const
|
||||
{
|
||||
return topology == FaceTopology::Nonconforming &&
|
||||
@@ -2155,7 +2160,7 @@ public:
|
||||
element[1].conformity == ElementConformity::Superset);
|
||||
}
|
||||
|
||||
/// @brief Return true if the face is a nonconforming coarse face.
|
||||
/// Return true if the face is a nonconforming coarse face.
|
||||
/** Note that ghost nonconforming master faces cannot be clearly
|
||||
identified as such with the currently available information, so this
|
||||
method will return false for such faces. */
|
||||
@@ -2165,7 +2170,7 @@ public:
|
||||
element[1].conformity == ElementConformity::Subset;
|
||||
}
|
||||
|
||||
/// @brief cast operator from FaceInformation to FaceInfo.
|
||||
/// cast operator from FaceInformation to FaceInfo.
|
||||
operator Mesh::FaceInfo() const;
|
||||
};
|
||||
|
||||
|
||||
+20
-3
@@ -43,13 +43,30 @@ KnotVector::KnotVector(istream &input)
|
||||
|
||||
KnotVector::KnotVector(int order, int NCP)
|
||||
{
|
||||
if (NCP == -1)
|
||||
{
|
||||
NumOfControlPoints = order + 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
NumOfControlPoints = NCP;
|
||||
}
|
||||
Order = order;
|
||||
NumOfControlPoints = NCP;
|
||||
knot.SetSize(NumOfControlPoints + Order + 1);
|
||||
NumOfElements = 0;
|
||||
coarse = false;
|
||||
|
||||
knot = -1.;
|
||||
if (NCP == -1)
|
||||
{
|
||||
for (int i = 0 ; i < Order + 1; i++)
|
||||
{
|
||||
knot[i] = 0.0;
|
||||
knot[i + Order + 1] = 1.0;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
knot = -1.;
|
||||
}
|
||||
}
|
||||
|
||||
KnotVector::KnotVector(int order, const Vector &k)
|
||||
|
||||
+8
-6
@@ -74,9 +74,13 @@ public:
|
||||
integers are read, for order and number of control points. */
|
||||
KnotVector(std::istream &input);
|
||||
|
||||
/** @brief Create a KnotVector with undefined knots (initialized to -1) of
|
||||
order @a order and number of control points @a NCP. */
|
||||
KnotVector(int order, int NCP);
|
||||
/** @brief Create a KnotVector with order @a order.
|
||||
When @a NCP is not provided the number of control points is set to
|
||||
@a order + 1, and the first @a order + 1 knots are set to 0 and last
|
||||
@a order + 1 knots are set to 1.
|
||||
When @a NCP is given number of control points is @a NCP and
|
||||
the knots are initialized to -1) */
|
||||
KnotVector(int order, int NCP = -1);
|
||||
|
||||
/** @brief Create a KnotVector with order @a order and knots @a knot.
|
||||
If @a k has the correct number of repeated knots at the begin and end,
|
||||
@@ -88,12 +92,10 @@ public:
|
||||
|
||||
/** @brief Create a KnotVector by passing in a degree, a Vector of interval
|
||||
lengths of length n, and a list of continuity of length n + 1.
|
||||
|
||||
The intervals refer to spans between unique knot values (not counting
|
||||
zero-size intervals at repeated knots), and the continuity values should
|
||||
be >= -1 (discontinuous) and <= order-1 (maximally-smooth for the given
|
||||
polynomial degree). Periodicity is not supported.
|
||||
*/
|
||||
polynomial degree). Periodicity is not supported.*/
|
||||
KnotVector(int order, const Vector& intervals,
|
||||
const Array<int>& continuity);
|
||||
|
||||
|
||||
@@ -1211,7 +1211,6 @@ void ParNCMesh::GetFaceNeighbors(ParMesh &pmesh)
|
||||
if (Dim <= 2) { nfaces = NEdges, nghosts = NGhostEdges; }
|
||||
|
||||
// enlarge Mesh::faces_info for ghost slaves
|
||||
MFEM_ASSERT(pmesh.faces_info.Size() == nfaces, "");
|
||||
MFEM_ASSERT(pmesh.GetNumFaces() == nfaces, "");
|
||||
pmesh.faces_info.SetSize(nfaces + nghosts);
|
||||
for (int i = nfaces; i < pmesh.faces_info.Size(); i++)
|
||||
@@ -1312,7 +1311,6 @@ void ParNCMesh::GetFaceNeighbors(ParMesh &pmesh)
|
||||
// Mesh::ApplyLocalSlaveTransformation.
|
||||
}
|
||||
|
||||
MFEM_ASSERT(fi.NCFace < 0, "fi.NCFace = " << fi.NCFace);
|
||||
fi.NCFace = pmesh.nc_faces_info.Size();
|
||||
pmesh.nc_faces_info.Append(Mesh::NCFaceInfo(true, sf.master, pm));
|
||||
}
|
||||
|
||||
@@ -174,7 +174,6 @@ ParticleTrajectories::ParticleTrajectories(const ParticleSet &particles,
|
||||
|
||||
void ParticleTrajectories::AddSegmentStart()
|
||||
{
|
||||
if (!pset.GetNParticles()) { return; }
|
||||
// Create a new mesh for all particle segments for this timestep
|
||||
segment_meshes.emplace_front(1, pset.GetNParticles()*2,
|
||||
pset.GetNParticles(),
|
||||
@@ -200,11 +199,10 @@ void ParticleTrajectories::AddSegmentStart()
|
||||
|
||||
void ParticleTrajectories::SetSegmentEnd()
|
||||
{
|
||||
if (segment_meshes.empty()) { return; } // no segments to end
|
||||
|
||||
const Array<ParticleSet::IDType> &end_ids = pset.GetIDs();
|
||||
|
||||
// Add all endpoint vertices + segments for all particles
|
||||
// Add all endpoint vertices + segments for all particles that were in
|
||||
// SetSegmentStart
|
||||
int num_start = segment_ids.front().Size();
|
||||
for (int i = 0; i < num_start; i++)
|
||||
{
|
||||
@@ -230,11 +228,6 @@ void ParticleTrajectories::SetSegmentEnd()
|
||||
void ParticleTrajectories::Visualize()
|
||||
{
|
||||
SetSegmentEnd();
|
||||
if (segment_meshes.empty() && !mesh)
|
||||
{
|
||||
AddSegmentStart();
|
||||
return;
|
||||
}
|
||||
|
||||
// Create a mesh of all the trajectory segments
|
||||
std::vector<Mesh*> all_meshes;
|
||||
@@ -246,8 +239,23 @@ void ParticleTrajectories::Visualize()
|
||||
{
|
||||
all_meshes.push_back(mesh);
|
||||
}
|
||||
if (mesh_bb)
|
||||
{
|
||||
all_meshes.push_back(mesh_bb);
|
||||
}
|
||||
|
||||
Mesh trajectories(all_meshes.data(), all_meshes.size());
|
||||
bool vis = trajectories.GetNE() > 0;
|
||||
#ifdef MFEM_USE_MPI
|
||||
MPI_Allreduce(MPI_IN_PLACE, &vis, 1, MFEM_MPI_CXX_BOOL,
|
||||
MPI_LOR, pset.GetComm());
|
||||
#endif // MFEM_USE_MPI
|
||||
if (!vis) // if all rank have 0 elements, skip visualization
|
||||
{
|
||||
AddSegmentStart();
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
VisualizeMesh(sock, vishost, visport, trajectories, comm,
|
||||
@@ -260,5 +268,97 @@ void ParticleTrajectories::Visualize()
|
||||
AddSegmentStart();
|
||||
}
|
||||
|
||||
void ParticleTrajectories::SetVisualizationBoundingBox(const Vector &xmin,
|
||||
const Vector &xmax)
|
||||
{
|
||||
MFEM_VERIFY(xmin.Size() == pset.GetDim() &&
|
||||
xmax.Size() == pset.GetDim(),
|
||||
"Bounding box dimension must match ParticleSet dimension.");
|
||||
|
||||
// Create a box mesh for visualization
|
||||
if (mesh_bb)
|
||||
{
|
||||
delete mesh_bb;
|
||||
mesh_bb = nullptr;
|
||||
}
|
||||
|
||||
if (pset.GetDim() == 2)
|
||||
{
|
||||
int dim = 2;
|
||||
int nvert = 4;
|
||||
int nelem = 4;
|
||||
mesh_bb = new Mesh(1, nvert, nelem, 0, dim);
|
||||
Vector v0(dim), v1(dim), v2(dim), v3(dim);
|
||||
v0 = xmin;
|
||||
v1 = xmax;
|
||||
v2[0] = xmax[0]; v2[1] = xmin[1];
|
||||
v3[0] = xmin[0]; v3[1] = xmax[1];
|
||||
|
||||
mesh_bb->AddVertex(v0);
|
||||
mesh_bb->AddVertex(v1);
|
||||
mesh_bb->AddVertex(v2);
|
||||
mesh_bb->AddVertex(v3);
|
||||
|
||||
int vi[2] = {0,1};
|
||||
mesh_bb->AddSegment(vi);
|
||||
vi[0] = 1; vi[1] = 2;
|
||||
mesh_bb->AddSegment(vi);
|
||||
vi[0] = 2; vi[1] = 3;
|
||||
mesh_bb->AddSegment(vi);
|
||||
vi[0] = 3; vi[1] = 0;
|
||||
mesh_bb->AddSegment(vi);
|
||||
mesh_bb->FinalizeMesh();
|
||||
}
|
||||
else // dim == 3
|
||||
{
|
||||
int dim = 3;
|
||||
int nvert = 8;
|
||||
int nelem = 12;
|
||||
mesh_bb = new Mesh(1, nvert, nelem, 0, dim);
|
||||
Vector v(dim);
|
||||
|
||||
// Vertices
|
||||
v[0] = xmin[0]; v[1] = xmin[1]; v[2] = xmin[2];
|
||||
mesh_bb->AddVertex(v); // 0: 000
|
||||
v[0] = xmax[0]; v[1] = xmin[1]; v[2] = xmin[2];
|
||||
mesh_bb->AddVertex(v); // 1: 100
|
||||
v[0] = xmax[0]; v[1] = xmax[1]; v[2] = xmin[2];
|
||||
mesh_bb->AddVertex(v); // 2: 110
|
||||
v[0] = xmin[0]; v[1] = xmax[1]; v[2] = xmin[2];
|
||||
mesh_bb->AddVertex(v); // 3: 010
|
||||
|
||||
v[0] = xmin[0]; v[1] = xmin[1]; v[2] = xmax[2];
|
||||
mesh_bb->AddVertex(v); // 4: 001
|
||||
v[0] = xmax[0]; v[1] = xmin[1]; v[2] = xmax[2];
|
||||
mesh_bb->AddVertex(v); // 5: 101
|
||||
v[0] = xmax[0]; v[1] = xmax[1]; v[2] = xmax[2];
|
||||
mesh_bb->AddVertex(v); // 6: 111
|
||||
v[0] = xmin[0]; v[1] = xmax[1]; v[2] = xmax[2];
|
||||
mesh_bb->AddVertex(v); // 7: 011
|
||||
|
||||
// Segments
|
||||
int vi[2];
|
||||
// Bottom face
|
||||
vi[0] = 0; vi[1] = 1; mesh_bb->AddSegment(vi);
|
||||
vi[0] = 1; vi[1] = 2; mesh_bb->AddSegment(vi);
|
||||
vi[0] = 2; vi[1] = 3; mesh_bb->AddSegment(vi);
|
||||
vi[0] = 3; vi[1] = 0; mesh_bb->AddSegment(vi);
|
||||
|
||||
// Top face
|
||||
vi[0] = 4; vi[1] = 5; mesh_bb->AddSegment(vi);
|
||||
vi[0] = 5; vi[1] = 6; mesh_bb->AddSegment(vi);
|
||||
vi[0] = 6; vi[1] = 7; mesh_bb->AddSegment(vi);
|
||||
vi[0] = 7; vi[1] = 4; mesh_bb->AddSegment(vi);
|
||||
|
||||
// Vertical edges
|
||||
vi[0] = 0; vi[1] = 4; mesh_bb->AddSegment(vi);
|
||||
vi[0] = 1; vi[1] = 5; mesh_bb->AddSegment(vi);
|
||||
vi[0] = 2; vi[1] = 6; mesh_bb->AddSegment(vi);
|
||||
vi[0] = 3; vi[1] = 7; mesh_bb->AddSegment(vi);
|
||||
|
||||
mesh_bb->FinalizeMesh();
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace common
|
||||
} // namespace mfem
|
||||
|
||||
@@ -46,7 +46,8 @@ class ParticleTrajectories
|
||||
{
|
||||
protected:
|
||||
const ParticleSet &pset;
|
||||
Mesh *mesh = nullptr;
|
||||
Mesh *mesh = nullptr; // optional edge mesh to visualize along with particles
|
||||
Mesh *mesh_bb = nullptr; // optional bounding box mesh for visualization
|
||||
|
||||
socketstream sock;
|
||||
/// Track particle IDs that exist at the segment start.
|
||||
@@ -90,10 +91,24 @@ public:
|
||||
const char *keys_=nullptr);
|
||||
|
||||
/// Add a mesh to be visualized along with the particle trajectories.
|
||||
void AddMeshForVisualization(Mesh *mesh_) { mesh = mesh_; }
|
||||
void AddMeshForVisualization(Mesh *mesh_)
|
||||
{
|
||||
MFEM_VERIFY(mesh_->Dimension() == 1,
|
||||
"Mesh dimension must be 1 to match the particle trajectory.");
|
||||
mesh = mesh_;
|
||||
}
|
||||
|
||||
/// Visualize the particle trajectories (and mesh if provided).
|
||||
void Visualize();
|
||||
|
||||
/// Set the bounding box for visualization.
|
||||
void SetVisualizationBoundingBox(const Vector &xmin, const Vector &xmax);
|
||||
|
||||
/// Destructor
|
||||
~ParticleTrajectories()
|
||||
{
|
||||
delete mesh_bb;
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
|
||||
@@ -34,11 +34,13 @@ if (MFEM_USE_MPI)
|
||||
EXTRA_HEADERS maxwell_solver.hpp ${MFEM_MINIAPPS_COMMON_HEADERS}
|
||||
LIBRARIES mfem-common)
|
||||
|
||||
add_mfem_miniapp(lorentz
|
||||
MAIN lorentz.cpp
|
||||
EXTRA_HEADERS ${MFEM_MINIAPPS_COMMON_HEADERS}
|
||||
LIBRARIES mfem-common)
|
||||
|
||||
if (MFEM_USE_GSLIB)
|
||||
add_mfem_miniapp(lorentz
|
||||
MAIN lorentz.cpp
|
||||
EXTRA_HEADERS ${MFEM_MINIAPPS_COMMON_HEADERS}
|
||||
LIBRARIES mfem-common)
|
||||
endif()
|
||||
|
||||
# Add the corresponding tests to the "test" target
|
||||
if (MFEM_ENABLE_TESTING)
|
||||
add_test(NAME tesla_np=4
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -21,7 +21,10 @@ MFEM_LIB_FILE = mfem_is_not_built
|
||||
-include $(CONFIG_MK)
|
||||
|
||||
SEQ_MINIAPPS =
|
||||
PAR_MINIAPPS = volta tesla maxwell joule lorentz
|
||||
PAR_MINIAPPS = volta tesla maxwell joule
|
||||
ifeq ($(MFEM_USE_GSLIB), YES)
|
||||
PAR_MINIAPPS += lorentz
|
||||
endif
|
||||
ifeq ($(MFEM_USE_MPI),NO)
|
||||
MINIAPPS = $(SEQ_MINIAPPS)
|
||||
else
|
||||
@@ -51,9 +54,11 @@ all: $(MINIAPPS)
|
||||
$(MFEM_CXX) $(MFEM_LINK_FLAGS) -o $@ $@.o $@_solver.o $(COMMON_LIB) \
|
||||
$(MFEM_LIBS)
|
||||
|
||||
ifeq ($(MFEM_USE_MPI),YES)
|
||||
lorentz: %: $(SRC)%.cpp $(MFEM_LIB_FILE) $(CONFIG_MK) | lib-common
|
||||
$(MFEM_CXX) $(MFEM_FLAGS) -c $(<)
|
||||
$(MFEM_CXX) $(MFEM_LINK_FLAGS) -o $@ $@.o $(COMMON_LIB) $(MFEM_LIBS)
|
||||
endif
|
||||
|
||||
# Rules for compiling miniapp dependencies
|
||||
$(addsuffix _solver.o,$(MINIAPPS)): \
|
||||
@@ -112,10 +117,10 @@ joule-test-par: joule
|
||||
lorentz-test-par: lorentz-test-1 lorentz-test-2
|
||||
lorentz-test-1: lorentz volta-test-3
|
||||
@$(call mfem-test,$<, $(RUN_MPI), Electromagnetic miniapp,\
|
||||
-er Volta-AMR-Parallel -ec 2 -x0 '0.5 0.5 0.9' -p0 '1 0 0')
|
||||
-er Volta-AMR-Parallel -ec 2 -npt 100 -xmin '0.0 0.0 0.0' -xmax '1.0 1.0 1.0' -pmin '1 0 0' -pmax '1 0 0' -rdf 0 -vt 0 -nt 100')
|
||||
lorentz-test-2: lorentz tesla-test-2
|
||||
@$(call mfem-test,$<, $(RUN_MPI), Electromagnetic miniapp,\
|
||||
-br Tesla-AMR-Parallel -bc 2 -x0 '0.1 0.5 0.1' -p0 '0 0.4 0.1' -tf 9)
|
||||
-br Tesla-AMR-Parallel -bc 2 -br Tesla-AMR-Parallel -npt 10 -xmin '0.0 0.0 0.0' -xmax '1.0 1.0 1.0' -pmin '0 0.1 0.05' -pmax '0 0.4 0.1' -nt 1000 -rdf 0 -vt 0)
|
||||
|
||||
# Testing: "test" target and mfem-test* variables are defined in config/test.mk
|
||||
|
||||
|
||||
@@ -421,22 +421,18 @@ void NavierParticles::Step(const real_t dt, const ParGridFunction &u_gf,
|
||||
void NavierParticles::InterpolateUW(const ParGridFunction &u_gf,
|
||||
const ParGridFunction &w_gf)
|
||||
{
|
||||
finder.FindPoints(X(), X().GetOrdering());
|
||||
finder.FindPoints(X());
|
||||
|
||||
finder.Interpolate(u_gf, U());
|
||||
Ordering::Reorder(U(), U().GetVDim(), u_gf.ParFESpace()->GetOrdering(),
|
||||
U().GetOrdering());
|
||||
finder.Interpolate(u_gf, U(), U().GetOrdering());
|
||||
|
||||
finder.Interpolate(w_gf, W());
|
||||
Ordering::Reorder(W(), W().GetVDim(), w_gf.ParFESpace()->GetOrdering(),
|
||||
W().GetOrdering());
|
||||
finder.Interpolate(w_gf, W(), W().GetOrdering());
|
||||
}
|
||||
|
||||
void NavierParticles::DeactivateLostParticles(bool findpts)
|
||||
{
|
||||
if (findpts)
|
||||
{
|
||||
finder.FindPoints(X(), X().GetOrdering());
|
||||
finder.FindPoints(X());
|
||||
}
|
||||
|
||||
const Array<unsigned int> lost_idxs = finder.GetPointsNotFoundIndices();
|
||||
|
||||
@@ -0,0 +1,489 @@
|
||||
MFEM NURBS mesh v1.0
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see fem/geom.hpp):
|
||||
#
|
||||
# SEGMENT = 1
|
||||
# SQUARE = 3
|
||||
# CUBE = 5
|
||||
#
|
||||
|
||||
dimension
|
||||
3
|
||||
|
||||
elements
|
||||
1
|
||||
1 5 0 1 2 3 4 5 6 7
|
||||
|
||||
boundary
|
||||
6
|
||||
1 3 2 1 0 3
|
||||
1 3 4 5 6 7
|
||||
1 3 0 1 5 4
|
||||
1 3 1 2 6 5
|
||||
1 3 2 3 7 6
|
||||
1 3 3 0 4 7
|
||||
|
||||
edges
|
||||
12
|
||||
0 0 1
|
||||
0 3 2
|
||||
0 4 5
|
||||
0 7 6
|
||||
1 0 3
|
||||
1 1 2
|
||||
1 4 7
|
||||
1 5 6
|
||||
2 0 4
|
||||
2 1 5
|
||||
2 2 6
|
||||
2 3 7
|
||||
|
||||
vertices
|
||||
8
|
||||
|
||||
knotvectors
|
||||
3
|
||||
2 6 0 0 0 0.25 0.5 0.75 1 1 1
|
||||
2 6 0 0 0 0.25 0.5 0.75 1 1 1
|
||||
2 6 0 0 0 0.25 0.5 0.75 1 1 1
|
||||
|
||||
weights
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
|
||||
FiniteElementSpace
|
||||
FiniteElementCollection: NURBS2
|
||||
VDim: 3
|
||||
Ordering: 1
|
||||
|
||||
0.0116849 0.100677 0.107741
|
||||
0.700841 0.44147 0.344065
|
||||
0.437989 1.28285 0.264685
|
||||
-0.303721 0.806601 -0.11583
|
||||
-0.539413 0.0200674 0.908587
|
||||
0.395933 0.494981 1.39817
|
||||
0.0117652 1.43411 1.08871
|
||||
-0.759897 0.891295 0.785367
|
||||
0.0812991 0.127468 0.125781
|
||||
0.255008 0.202244 0.188147
|
||||
0.448525 0.285255 0.254138
|
||||
0.625506 0.391535 0.323138
|
||||
0.34086 1.23115 0.189627
|
||||
0.135768 1.14038 0.0949877
|
||||
-0.0571106 1.04065 -0.00204767
|
||||
-0.225607 0.894706 -0.0866114
|
||||
-0.3804 0.0431261 0.932761
|
||||
-0.121171 0.133223 1.01883
|
||||
0.0987705 0.257745 1.15354
|
||||
0.299995 0.410817 1.30949
|
||||
-0.0928372 1.36692 1.04319
|
||||
-0.292608 1.22409 0.954962
|
||||
-0.47 1.08105 0.868931
|
||||
-0.663376 0.956914 0.807106
|
||||
-0.0313904 0.173208 0.0241675
|
||||
-0.116453 0.335806 -0.0797544
|
||||
-0.197708 0.522984 -0.143019
|
||||
-0.268697 0.719567 -0.147642
|
||||
0.693218 0.561638 0.333951
|
||||
0.642943 0.782839 0.330385
|
||||
0.567332 1.00697 0.306227
|
||||
0.477659 1.20209 0.287077
|
||||
-0.550848 0.145891 0.873495
|
||||
-0.617164 0.379504 0.811408
|
||||
-0.684941 0.59523 0.782838
|
||||
-0.736288 0.802334 0.780498
|
||||
0.357892 0.575016 1.31896
|
||||
0.285779 0.784367 1.21896
|
||||
0.185794 1.01755 1.15046
|
||||
0.0653891 1.28559 1.09951
|
||||
-0.0537529 0.0832559 0.179502
|
||||
-0.188121 0.0603262 0.356393
|
||||
-0.323693 0.0343845 0.566552
|
||||
-0.463087 0.0213273 0.787315
|
||||
0.675777 0.435988 0.449458
|
||||
0.610746 0.444971 0.684641
|
||||
0.542159 0.473619 0.947334
|
||||
0.451733 0.481779 1.235
|
||||
0.387085 1.30441 0.330155
|
||||
0.300155 1.32346 0.502159
|
||||
0.197236 1.33735 0.733646
|
||||
0.0778968 1.38129 0.967056
|
||||
-0.364395 0.836177 -0.0380904
|
||||
-0.499556 0.880443 0.175051
|
||||
-0.618562 0.899155 0.419916
|
||||
-0.729908 0.894583 0.658954
|
||||
-0.191762 0.792322 -0.0838227
|
||||
-0.0127656 0.922096 0.0198956
|
||||
0.169871 1.04169 0.106131
|
||||
0.374421 1.15606 0.221327
|
||||
-0.107725 0.590479 -0.0701237
|
||||
0.0797398 0.719117 0.052235
|
||||
0.248024 0.837672 0.153949
|
||||
0.4506 0.951982 0.257914
|
||||
-0.014591 0.399524 -0.0133179
|
||||
0.169119 0.524685 0.0962864
|
||||
0.349821 0.64291 0.208715
|
||||
0.547914 0.743218 0.290306
|
||||
0.0485967 0.219259 0.0598748
|
||||
0.22429 0.311269 0.147327
|
||||
0.424681 0.420235 0.239975
|
||||
0.603974 0.511425 0.314819
|
||||
0.0215703 0.115709 0.220091
|
||||
0.202775 0.192498 0.292796
|
||||
0.404824 0.300707 0.373618
|
||||
0.587761 0.396426 0.424654
|
||||
-0.0813987 0.0906892 0.402963
|
||||
0.122699 0.181167 0.495311
|
||||
0.320206 0.283771 0.586703
|
||||
0.513593 0.396941 0.659549
|
||||
-0.192037 0.0641737 0.611853
|
||||
0.0238142 0.153445 0.708339
|
||||
0.217314 0.279993 0.795557
|
||||
0.427183 0.403976 0.895618
|
||||
-0.321216 0.0526643 0.821777
|
||||
-0.0668994 0.132414 0.92639
|
||||
0.136671 0.264444 1.03878
|
||||
0.346957 0.403474 1.16556
|
||||
0.652713 0.556232 0.447968
|
||||
0.597625 0.792616 0.438187
|
||||
0.516989 1.00686 0.408826
|
||||
0.435038 1.2089 0.363187
|
||||
0.570735 0.56432 0.677065
|
||||
0.500648 0.786806 0.651401
|
||||
0.416515 1.01355 0.598304
|
||||
0.325952 1.21023 0.541362
|
||||
0.493342 0.572126 0.917974
|
||||
0.405211 0.78473 0.869844
|
||||
0.312137 1.00883 0.809381
|
||||
0.239854 1.22498 0.760211
|
||||
0.407138 0.581025 1.17298
|
||||
0.318312 0.777059 1.09777
|
||||
0.225803 1.01555 1.0397
|
||||
0.121726 1.25049 0.981676
|
||||
0.290702 1.25013 0.278809
|
||||
0.0701636 1.14755 0.189221
|
||||
-0.119108 1.04886 0.103803
|
||||
-0.289182 0.91921 0.00804675
|
||||
0.185344 1.27637 0.482987
|
||||
-0.0241763 1.17861 0.405915
|
||||
-0.228914 1.06728 0.323432
|
||||
-0.408653 0.937665 0.225694
|
||||
0.0857194 1.30362 0.704401
|
||||
-0.122501 1.20271 0.644137
|
||||
-0.324492 1.07716 0.554526
|
||||
-0.523359 0.958708 0.456627
|
||||
-0.0299893 1.32937 0.937238
|
||||
-0.231056 1.22639 0.854007
|
||||
-0.4277 1.08064 0.77448
|
||||
-0.624632 0.962435 0.693615
|
||||
-0.335332 0.743245 -0.0361732
|
||||
-0.264988 0.556883 -0.0142987
|
||||
-0.179867 0.361039 0.028428
|
||||
-0.0895202 0.170043 0.122851
|
||||
-0.46712 0.787684 0.186278
|
||||
-0.399567 0.586191 0.200671
|
||||
-0.301046 0.368151 0.251277
|
||||
-0.218868 0.158789 0.317644
|
||||
-0.59138 0.804574 0.412544
|
||||
-0.506716 0.60204 0.430758
|
||||
-0.435083 0.376564 0.464743
|
||||
-0.356472 0.147414 0.523001
|
||||
-0.689157 0.806058 0.647632
|
||||
-0.629068 0.593243 0.655094
|
||||
-0.555738 0.371114 0.695893
|
||||
-0.485726 0.146879 0.754833
|
||||
-0.417228 0.17436 0.899793
|
||||
-0.169866 0.253389 0.990733
|
||||
0.028903 0.366525 1.11301
|
||||
0.252098 0.500703 1.24433
|
||||
-0.500923 0.414148 0.856523
|
||||
-0.284378 0.495056 0.948388
|
||||
-0.0843163 0.601653 1.05541
|
||||
0.155216 0.712225 1.16544
|
||||
-0.58149 0.643576 0.830741
|
||||
-0.373319 0.748859 0.924034
|
||||
-0.18065 0.845261 1.0198
|
||||
0.0460522 0.952923 1.10623
|
||||
-0.637619 0.848756 0.803888
|
||||
-0.448156 0.969832 0.881928
|
||||
-0.260587 1.09929 0.976527
|
||||
-0.0459431 1.21968 1.05279
|
||||
-0.0137524 0.209301 0.168941
|
||||
0.169242 0.299016 0.260497
|
||||
0.377537 0.400892 0.353247
|
||||
0.570084 0.506608 0.425317
|
||||
-0.080203 0.408135 0.0918812
|
||||
0.105504 0.518748 0.204877
|
||||
0.298886 0.634349 0.305087
|
||||
0.493708 0.735607 0.396409
|
||||
-0.168586 0.619614 0.0360283
|
||||
0.0165872 0.734724 0.158097
|
||||
0.205366 0.847514 0.254686
|
||||
0.405389 0.959363 0.360348
|
||||
-0.258677 0.823089 0.0161381
|
||||
-0.0786817 0.945242 0.11815
|
||||
0.121369 1.0512 0.208493
|
||||
0.324175 1.15154 0.303941
|
||||
-0.124414 0.198947 0.367495
|
||||
0.0754223 0.285305 0.478578
|
||||
0.271456 0.388643 0.569286
|
||||
0.47429 0.500746 0.649904
|
||||
-0.214578 0.415337 0.300568
|
||||
-0.0214509 0.508707 0.419948
|
||||
0.191156 0.607363 0.514446
|
||||
0.397619 0.729213 0.61556
|
||||
-0.302858 0.639951 0.251014
|
||||
-0.103789 0.739391 0.357589
|
||||
0.110001 0.848732 0.463479
|
||||
0.318473 0.955375 0.556868
|
||||
-0.385766 0.851741 0.222241
|
||||
-0.192006 0.963536 0.318999
|
||||
0.024575 1.07461 0.420919
|
||||
0.234314 1.16838 0.513436
|
||||
-0.239144 0.176376 0.576763
|
||||
-0.0256959 0.27234 0.678898
|
||||
0.180326 0.386088 0.77861
|
||||
0.380616 0.507244 0.864894
|
||||
-0.336051 0.4068 0.522487
|
||||
-0.134268 0.49507 0.619439
|
||||
0.0937046 0.598445 0.720534
|
||||
0.295642 0.716778 0.815371
|
||||
-0.413699 0.644524 0.471466
|
||||
-0.217717 0.723955 0.573321
|
||||
0.00422339 0.839902 0.669769
|
||||
0.207017 0.943147 0.772738
|
||||
-0.496887 0.853244 0.456225
|
||||
-0.291644 0.956378 0.557542
|
||||
-0.0924407 1.0788 0.643644
|
||||
0.1296 1.17577 0.725705
|
||||
-0.366417 0.168186 0.785706
|
||||
-0.121768 0.262554 0.893007
|
||||
0.083928 0.379718 1.0059
|
||||
0.297426 0.504658 1.11483
|
||||
-0.443726 0.407516 0.735498
|
||||
-0.240641 0.495966 0.83729
|
||||
-0.0238202 0.596216 0.939116
|
||||
0.203993 0.722308 1.04178
|
||||
-0.524688 0.636263 0.710265
|
||||
-0.331783 0.741903 0.804337
|
||||
-0.117384 0.835587 0.896008
|
||||
0.103953 0.951344 0.987382
|
||||
-0.599196 0.860865 0.697438
|
||||
-0.398317 0.964829 0.780767
|
||||
-0.195503 1.08864 0.858335
|
||||
0.0177754 1.19986 0.938708
|
||||
@@ -0,0 +1,118 @@
|
||||
MFEM NURBS mesh v1.0
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see fem/geom.hpp):
|
||||
#
|
||||
# SEGMENT = 1
|
||||
# SQUARE = 3
|
||||
# CUBE = 5
|
||||
#
|
||||
|
||||
dimension
|
||||
2
|
||||
|
||||
elements
|
||||
1
|
||||
1 3 0 1 2 3
|
||||
|
||||
boundary
|
||||
4
|
||||
1 1 0 1
|
||||
2 1 2 3
|
||||
3 1 3 0
|
||||
4 1 1 2
|
||||
|
||||
edges
|
||||
4
|
||||
0 0 1
|
||||
0 3 2
|
||||
1 0 3
|
||||
1 1 2
|
||||
|
||||
vertices
|
||||
4
|
||||
|
||||
knotvectors
|
||||
2
|
||||
2 6 0 0 0 0.25 0.5 0.75 1 1 1
|
||||
2 6 0 0 0 0.25 0.5 0.75 1 1 1
|
||||
|
||||
weights
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
1
|
||||
|
||||
FiniteElementSpace
|
||||
FiniteElementCollection: NURBS2
|
||||
VDim: 2
|
||||
Ordering: 1
|
||||
|
||||
0.0163925 0.141238
|
||||
0.774637 0.626247
|
||||
-0.147699 1.3396
|
||||
-0.757759 0.550541
|
||||
0.121943 0.231571
|
||||
0.272418 0.394336
|
||||
0.420152 0.532036
|
||||
0.635666 0.624585
|
||||
-0.261202 1.30275
|
||||
-0.454309 1.14438
|
||||
-0.593397 0.942458
|
||||
-0.710473 0.706781
|
||||
-0.111803 0.190859
|
||||
-0.313132 0.306672
|
||||
-0.51706 0.436229
|
||||
-0.67826 0.509765
|
||||
0.608023 0.786507
|
||||
0.372822 1.01006
|
||||
0.159851 1.1653
|
||||
-0.0696727 1.29923
|
||||
-0.00322359 0.290759
|
||||
0.158563 0.459956
|
||||
0.321006 0.615434
|
||||
0.509901 0.715169
|
||||
-0.240232 0.408664
|
||||
-0.0626107 0.581669
|
||||
0.136422 0.738867
|
||||
0.308415 0.910906
|
||||
-0.452041 0.542364
|
||||
-0.263077 0.727566
|
||||
-0.0801052 0.906599
|
||||
0.0851199 1.07157
|
||||
-0.624784 0.659372
|
||||
-0.470927 0.866487
|
||||
-0.318204 1.05325
|
||||
-0.134756 1.23187
|
||||
@@ -18,6 +18,10 @@
|
||||
// nurbs_ex1 -m meshes/two-cubes-nurbs-rot.mesh -o 1 -r 3 -rf meshes/two-cubes.ref
|
||||
// nurbs_ex1 -m meshes/two-cubes-nurbs-autoedge.mesh -o 1 -r 3 -rf meshes/two-cubes.ref
|
||||
// nurbs_ex1 -m ../../data/segment-nurbs.mesh -r 2 -o 2 -lod 3
|
||||
// nurbs_ex1 -m meshes/square-nurbs-deformed.mesh -o 2
|
||||
// nurbs_ex1 -m meshes/square-nurbs-deformed.mesh -o 2 -no-ibp
|
||||
// nurbs_ex1 -m meshes/cube-nurbs-deformed.mesh -o 2
|
||||
// nurbs_ex1 -m meshes/cube-nurbs-deformed.mesh -o 2 -no-ibp
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
@@ -553,9 +557,18 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 14. Save data in the VisIt format
|
||||
VisItDataCollection visit_dc("Example1", mesh);
|
||||
visit_dc.RegisterField("solution", &x);
|
||||
visit_dc.Save();
|
||||
if (ibp)
|
||||
{
|
||||
VisItDataCollection visit_dc("Example1", mesh);
|
||||
visit_dc.RegisterField("solution", &x);
|
||||
visit_dc.Save();
|
||||
}
|
||||
else
|
||||
{
|
||||
VisItDataCollection visit_dc("Example1_nibp", mesh);
|
||||
visit_dc.RegisterField("solution", &x);
|
||||
visit_dc.Save();
|
||||
}
|
||||
|
||||
// 15. Free the used memory.
|
||||
delete a;
|
||||
|
||||
@@ -43,6 +43,9 @@ add_mfem_miniapp(convert-dc
|
||||
add_mfem_miniapp(lor-transfer
|
||||
MAIN lor-transfer.cpp LIBRARIES mfem)
|
||||
|
||||
add_mfem_miniapp(compare-dc
|
||||
MAIN compare-dc.cpp LIBRARIES mfem)
|
||||
|
||||
add_mfem_miniapp(tmop-check-metric
|
||||
MAIN tmop-check-metric.cpp LIBRARIES mfem)
|
||||
|
||||
|
||||
@@ -0,0 +1,166 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
//
|
||||
// -------------------------------------------------------------------
|
||||
// Compare DC Miniapp: Compare fields saved via DataCollection classes
|
||||
// -------------------------------------------------------------------
|
||||
//
|
||||
// This miniapp loads previously saved data and computes the l2 norm of the
|
||||
// difference. Currently, only the VisItDataCollection class is supported.
|
||||
//
|
||||
// Compile with: make compare-dc
|
||||
//
|
||||
// Serial sample runs:
|
||||
// > compare-dc -r0 ../../examples/Example5 -r1 ../../examples/alt/Example5
|
||||
// > compare-dc -r0 Example5 -r1 alt/Example5 -tol 1e-6
|
||||
//
|
||||
// Parallel sample runs:
|
||||
// > mpirun -np 4 compare-dc -r0 ../../examples/Example5-Parallel
|
||||
// -r1 ../../examples/alt/Example5-Parallel
|
||||
//
|
||||
// NB: when no tolerance is provided the difference is simple reported.
|
||||
// If a tolerance is provided this is compared with the symmetric
|
||||
// relative difference. An error is given if difference exceeds the tolerance.
|
||||
|
||||
#include "mfem.hpp"
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
#ifdef MFEM_USE_MPI
|
||||
Mpi::Init();
|
||||
if (!Mpi::Root()) { mfem::out.Disable(); mfem::err.Disable(); }
|
||||
Hypre::Init();
|
||||
#endif
|
||||
|
||||
// Parse command-line options.
|
||||
const char *coll_name0 = NULL;
|
||||
const char *coll_name1 = NULL;
|
||||
int cycle = 0;
|
||||
int pad_digits_cycle = 6;
|
||||
int pad_digits_rank = 6;
|
||||
real_t tol = -1;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&coll_name0, "-r0", "--root-file_0",
|
||||
"Set the VisIt data collection root file prefix.", true);
|
||||
args.AddOption(&coll_name1, "-r1", "--root-file_1",
|
||||
"Set the VisIt data collection root file prefix.", true);
|
||||
args.AddOption(&cycle, "-c", "--cycle", "Set the cycle index to read.");
|
||||
args.AddOption(&pad_digits_cycle, "-pdc", "--pad-digits-cycle",
|
||||
"Number of digits in cycle.");
|
||||
args.AddOption(&pad_digits_rank, "-pdr", "--pad-digits-rank",
|
||||
"Number of digits in MPI rank.");
|
||||
args.AddOption(&tol, "-tol", "--tolerance",
|
||||
"Tolerance for checking the results.");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
args.PrintUsage(mfem::out);
|
||||
return 1;
|
||||
}
|
||||
args.PrintOptions(mfem::out);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
VisItDataCollection dc0(MPI_COMM_WORLD, coll_name0);
|
||||
#else
|
||||
VisItDataCollection dc0(coll_name0);
|
||||
#endif
|
||||
dc0.SetPadDigitsCycle(pad_digits_cycle);
|
||||
dc0.SetPadDigitsRank(pad_digits_rank);
|
||||
dc0.Load(cycle);
|
||||
|
||||
if (dc0.Error() != DataCollection::No_Error)
|
||||
{
|
||||
mfem::out << "Error loading VisIt data collection: " << coll_name0 << endl;
|
||||
return 1;
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
VisItDataCollection dc1(MPI_COMM_WORLD, coll_name1);
|
||||
#else
|
||||
VisItDataCollection dc1(coll_name1);
|
||||
#endif
|
||||
dc1.SetPadDigitsCycle(pad_digits_cycle);
|
||||
dc1.SetPadDigitsRank(pad_digits_rank);
|
||||
dc1.Load(cycle);
|
||||
|
||||
if (dc1.Error() != DataCollection::No_Error)
|
||||
{
|
||||
mfem::out << "Error loading VisIt data collection: " << coll_name1 << endl;
|
||||
return 1;
|
||||
}
|
||||
|
||||
typedef DataCollection::FieldMapType fields_t;
|
||||
const fields_t &fields0 = dc0.GetFieldMap();
|
||||
// Print the names of all fields.
|
||||
bool error = false;
|
||||
for (fields_t::const_iterator it0 = fields0.begin();
|
||||
it0 != fields0.end() ; ++it0)
|
||||
{
|
||||
GridFunction *gf0 = dc0.GetField(it0->first);
|
||||
if (!gf0)
|
||||
{
|
||||
mfem::out << "Error loading:"<<it0->first<< endl;
|
||||
mfem::out << "From data collection: " << coll_name0 << endl;
|
||||
return 1;
|
||||
}
|
||||
|
||||
GridFunction *gf1 = dc1.GetField(it0->first);
|
||||
if (!gf1)
|
||||
{
|
||||
mfem::out << "Error loading:"<<it0->first<< endl;
|
||||
mfem::out << "From data collection: " << coll_name1 << endl;
|
||||
return 1;
|
||||
}
|
||||
if (gf0->Size() != gf1->Size())
|
||||
{
|
||||
mfem::out << "Size error for:"<<it0->first<< endl;
|
||||
mfem::out << "In data collection: " << coll_name0
|
||||
<<" size is "<<gf0->Size()<< endl;
|
||||
mfem::out << "In data collection: " << coll_name1
|
||||
<<" size is "<<gf1->Size()<< endl;
|
||||
return 1;
|
||||
}
|
||||
|
||||
// Norm of vectors
|
||||
real_t nrm0 = gf0->Norml2();
|
||||
real_t nrm1 = gf1->Norml2();
|
||||
|
||||
// Difference
|
||||
(*gf0) -= (*gf1);
|
||||
real_t nrmd = gf0->Norml2();
|
||||
real_t rel_sym = 2*nrmd/(nrm0 + nrm1);
|
||||
if (gf0->Norml2() > rel_sym) { error = true; }
|
||||
|
||||
// Report
|
||||
mfem::out <<"==========================================="<<std::endl;
|
||||
mfem::out <<"|"<<it0->first<<"_0| = "<<nrm0<<std::endl;
|
||||
mfem::out <<"|"<<it0->first<<"_1| = "<<nrm1<<std::endl;
|
||||
mfem::out <<"\n|"<<it0->first<<"_0 - "<<it0->first<<"_1| = "<<nrmd <<std::endl;
|
||||
|
||||
mfem::out <<"\n2|"<<it0->first<<"_0 - "<<it0->first<<"_1|"<<std::endl;
|
||||
mfem::out << std::setfill('-') << std::setw(15 + 2*it0->first.length())
|
||||
<<" = "<<rel_sym<<std::endl;
|
||||
mfem::out <<"(|"<<it0->first<<"_0| + |"<<it0->first<<"_1|)\n"<<std::endl;
|
||||
|
||||
}
|
||||
|
||||
if (error && tol > 0.0)
|
||||
{
|
||||
mfem::out << "Data collections: " << coll_name0
|
||||
<< " & " << coll_name1 << " are outside of the tolerance!\n";
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -21,7 +21,7 @@ MFEM_LIB_FILE = mfem_is_not_built
|
||||
-include $(CONFIG_MK)
|
||||
|
||||
SEQ_MINIAPPS = display-basis load-dc convert-dc get-values lor-transfer \
|
||||
tmop-check-metric tmop-metric-magnitude
|
||||
tmop-check-metric tmop-metric-magnitude compare-dc
|
||||
|
||||
PAR_MINIAPPS = nodal-transfer plor-transfer gridfunction-bounds
|
||||
|
||||
@@ -78,7 +78,8 @@ RUN_MPI = $(MFEM_MPIEXEC) $(MFEM_MPIEXEC_NP) $(MFEM_MPI_NP)
|
||||
# Testing: Specific execution options
|
||||
# Do not test: display-basis, load-dc, convert-dc, get-values, lor-transfer, plor-transfer
|
||||
NO_TEST_APPS = display-basis load-dc convert-dc get-values lor-transfer \
|
||||
plor-transfer tmop-check-metric tmop-metric-magnitude gridfunction-bounds
|
||||
plor-transfer tmop-check-metric tmop-metric-magnitude gridfunction-bounds \
|
||||
compare-dc
|
||||
$(foreach app,$(NO_TEST_APPS),$(app)-test-seq $(app)-test-par):
|
||||
@true
|
||||
|
||||
|
||||
@@ -31,11 +31,6 @@ function(add_benchmark name)
|
||||
set_property(SOURCE ${${NAME}_BENCH_SRCS} PROPERTY LANGUAGE CUDA)
|
||||
endif(MFEM_USE_CUDA)
|
||||
|
||||
if (MFEM_USE_HIP)
|
||||
set_property(SOURCE ${${NAME}_BENCH_SRCS} PROPERTY LANGUAGE
|
||||
HIP_SOURCE_PROPERTY_FORMAT TRUE)
|
||||
endif(MFEM_USE_HIP)
|
||||
|
||||
add_executable(bench_${name} ${${NAME}_BENCH_SRCS})
|
||||
target_link_libraries(bench_${name} mfem pthread)
|
||||
add_dependencies(${MFEM_ALL_BENCHMARKS_TARGET_NAME} bench_${name})
|
||||
|
||||
+229
-114
@@ -8,23 +8,89 @@
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
//
|
||||
//
|
||||
// This benchmark contains the implementation of the CEED's bake-off problems:
|
||||
// high-order kernels/benchmarks designed to test and compare the performance
|
||||
// of high-order codes.
|
||||
//
|
||||
// See: https://ceed.exascaleproject.org/bps
|
||||
|
||||
#include "bench.hpp"
|
||||
#include "bench.hpp" // IWYU pragma: keep
|
||||
|
||||
#ifdef MFEM_USE_BENCHMARK
|
||||
|
||||
/*
|
||||
This benchmark contains the implementation of the CEED's bake-off problems:
|
||||
high-order kernels/benchmarks designed to test and compare the performance
|
||||
of high-order codes.
|
||||
#include <cassert>
|
||||
#include <string>
|
||||
|
||||
See: ceed.exascaleproject.org/bps and github.com/CEED/benchmarks
|
||||
*/
|
||||
template <int VDIM, bool GLL>
|
||||
#include "fem/qinterp/det.hpp" // IWYU pragma: keep
|
||||
#include "fem/qinterp/grad.hpp" // IWYU pragma: keep
|
||||
#include "fem/integ/lininteg_domain_kernels.hpp" // IWYU pragma: keep
|
||||
#include "fem/integ/bilininteg_vecdiffusion_pa.hpp" // IWYU pragma: keep
|
||||
|
||||
// Custom benchmark arguments generator
|
||||
static void CustomArguments(bmi::Benchmark *b) noexcept
|
||||
{
|
||||
constexpr int MAX_NDOFS = 16 * 1024 * (mfem_use_gpu ? 1024 : 8);
|
||||
|
||||
const auto orders = { 7, 6, 5, 4, 3, 2, 1 };
|
||||
|
||||
constexpr auto ndofs = [](int n) constexpr noexcept -> int
|
||||
{
|
||||
return (n + 1) * (n + 1) * (n + 1);
|
||||
};
|
||||
|
||||
constexpr auto inc = [](int n) constexpr noexcept -> int
|
||||
{
|
||||
return n < 160 ? 4 : n < 240 ? 8 : n < 320 ? 16 : 32;
|
||||
};
|
||||
|
||||
for (auto p : orders)
|
||||
{
|
||||
for (int n = 16; ndofs(n) <= MAX_NDOFS; n += inc(n))
|
||||
{
|
||||
b->Args({p, n});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Register kernel specializations used in the benchmarks
|
||||
static void AddKernelSpecializations()
|
||||
{
|
||||
using DET = QuadratureInterpolator::DetKernels;
|
||||
DET::Specialization<3, 3, 2, 2>::Add();
|
||||
DET::Specialization<3, 3, 2, 3>::Add();
|
||||
DET::Specialization<3, 3, 2, 5>::Add();
|
||||
DET::Specialization<3, 3, 2, 6>::Add();
|
||||
DET::Specialization<3, 3, 5, 5>::Add();
|
||||
// Others might exceed memory limits
|
||||
|
||||
using GRAD = QuadratureInterpolator::GradKernels;
|
||||
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 2>::Add();
|
||||
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 7>::Add();
|
||||
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 8>::Add();
|
||||
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 9>::Add();
|
||||
|
||||
using LIN = DomainLFIntegrator::AssembleKernels;
|
||||
LIN::Specialization<3, 7, 7>::Add();
|
||||
LIN::Specialization<3, 6, 6>::Add();
|
||||
LIN::Specialization<3, 8, 8>::Add();
|
||||
|
||||
using VDIFF = VectorDiffusionIntegrator::ApplyPAKernels;
|
||||
VDIFF::Specialization<3, 3, 3, 3>::Add();
|
||||
VDIFF::Specialization<3, 3, 4, 4>::Add();
|
||||
VDIFF::Specialization<3, 3, 5, 5>::Add();
|
||||
VDIFF::Specialization<3, 3, 6, 6>::Add();
|
||||
VDIFF::Specialization<3, 3, 7, 7>::Add();
|
||||
VDIFF::Specialization<3, 3, 8, 8>::Add();
|
||||
}
|
||||
|
||||
// Bake-off base class
|
||||
template <int BFI, int VDIM, bool GLL>
|
||||
struct BakeOff
|
||||
{
|
||||
static constexpr int DIM = 3;
|
||||
const int N, p, q;
|
||||
inline static constexpr int DIM = 3;
|
||||
const int p, c, q, n, nx, ny, nz;
|
||||
Mesh mesh;
|
||||
H1_FECollection fec;
|
||||
FiniteElementSpace fes;
|
||||
@@ -38,12 +104,15 @@ struct BakeOff
|
||||
GridFunction x, y;
|
||||
BilinearForm a;
|
||||
double mdofs{};
|
||||
BilinearFormIntegrator *bfi;
|
||||
|
||||
BakeOff(int p):
|
||||
N(Device::IsEnabled() ? 32 : 4),
|
||||
p(p),
|
||||
q(2 * p + (GLL ? -1 : 3)),
|
||||
mesh(Mesh::MakeCartesian3D(N, N, N, Element::HEXAHEDRON)),
|
||||
BakeOff(int p, int side):
|
||||
p(p), c(side), q(2 * p + (GLL ? -1 : 3)),
|
||||
n((assert(c >= p), c / p)),
|
||||
nx(n + (p * (n + 1) * p * n * p * n < c * c * c ? 1 : 0)),
|
||||
ny(n + (p * (n + 1) * p * (n + 1) * p * n < c * c * c ? 1 : 0)),
|
||||
nz(n),
|
||||
mesh(Mesh::MakeCartesian3D(nx, ny, nz, Element::HEXAHEDRON)),
|
||||
fec(p, DIM, BasisType::GaussLobatto),
|
||||
fes(&mesh, &fec, VDIM, VDIM == 3 ? Ordering::byVDIM : Ordering::byNODES),
|
||||
geom_type(mesh.GetTypicalElementGeometry()),
|
||||
@@ -58,22 +127,41 @@ struct BakeOff
|
||||
a(&fes)
|
||||
{
|
||||
x = 0.0;
|
||||
if constexpr (BFI == 1)
|
||||
{
|
||||
bfi = new MassIntegrator(one, ir);
|
||||
}
|
||||
else if constexpr (BFI == 2)
|
||||
{
|
||||
bfi = new VectorMassIntegrator(one, ir);
|
||||
}
|
||||
else if constexpr (BFI == 3 || BFI == 5)
|
||||
{
|
||||
bfi = new DiffusionIntegrator(one, ir);
|
||||
}
|
||||
else if constexpr (BFI == 4 || BFI == 6)
|
||||
{
|
||||
bfi = new VectorDiffusionIntegrator(one, ir);
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(BFI >= 1 && BFI <= 6, "Invalid BilinearFormIntegrator");
|
||||
}
|
||||
a.AddDomainIntegrator(bfi);
|
||||
}
|
||||
|
||||
virtual void benchmark() = 0;
|
||||
|
||||
double SumMdofs() const { return mdofs; }
|
||||
[[nodiscard]] double SumMdofs() const noexcept { return mdofs; }
|
||||
|
||||
double MDofs() const { return 1e-6 * dofs; }
|
||||
[[nodiscard]] double MDofs() const noexcept { return 1e-6 * dofs; }
|
||||
};
|
||||
|
||||
/// Bake-off Problems (BPs)
|
||||
template <typename BFI, int VDIM, bool GLL>
|
||||
struct Problem : public BakeOff<VDIM, GLL>
|
||||
// Bake-off Problems (BPs)
|
||||
template <int BFI, int VDIM, bool GLL>
|
||||
struct BP : public BakeOff<BFI, VDIM, GLL>
|
||||
{
|
||||
const double rtol = 1e-12;
|
||||
const int max_it = 32;
|
||||
const int print_lvl = -1;
|
||||
const int max_it = 32, print_lvl = -1;
|
||||
|
||||
Array<int> ess_tdof_list;
|
||||
Array<int> ess_bdr;
|
||||
@@ -82,44 +170,56 @@ struct Problem : public BakeOff<VDIM, GLL>
|
||||
Vector B, X;
|
||||
CGSolver cg;
|
||||
|
||||
using BakeOff<VDIM, GLL>::a;
|
||||
using BakeOff<VDIM, GLL>::ir;
|
||||
using BakeOff<VDIM, GLL>::one;
|
||||
using BakeOff<VDIM, GLL>::mesh;
|
||||
using BakeOff<VDIM, GLL>::fes;
|
||||
using BakeOff<VDIM, GLL>::x;
|
||||
using BakeOff<VDIM, GLL>::y;
|
||||
using BakeOff<VDIM, GLL>::mdofs;
|
||||
using base = BakeOff<BFI, VDIM, GLL>;
|
||||
using base::a;
|
||||
using base::ir;
|
||||
using base::one;
|
||||
using base::mesh;
|
||||
using base::fes;
|
||||
using base::x;
|
||||
using base::y;
|
||||
using base::mdofs;
|
||||
using base::unit_vec;
|
||||
using base::bfi;
|
||||
|
||||
Problem(int order):
|
||||
BakeOff<VDIM, GLL>(order),
|
||||
BP(int p, int side) noexcept: base(p, side),
|
||||
ess_bdr(mesh.bdr_attributes.Max()),
|
||||
b(&fes)
|
||||
{
|
||||
ess_bdr = 1;
|
||||
fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
if (VDIM == 1)
|
||||
|
||||
if constexpr (VDIM == 1)
|
||||
{
|
||||
b.AddDomainIntegrator(new DomainLFIntegrator(this->one));
|
||||
b.AddDomainIntegrator(new DomainLFIntegrator(one));
|
||||
}
|
||||
else
|
||||
{
|
||||
b.AddDomainIntegrator(new VectorDomainLFIntegrator(this->unit_vec));
|
||||
b.AddDomainIntegrator(new VectorDomainLFIntegrator(unit_vec));
|
||||
}
|
||||
b.UseFastAssembly(true);
|
||||
b.Assemble();
|
||||
|
||||
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
a.AddDomainIntegrator(new BFI(one, ir));
|
||||
a.Assemble();
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
|
||||
cg.SetRelTol(rtol);
|
||||
cg.SetOperator(*A);
|
||||
cg.SetAbsTol(0.0);
|
||||
cg.iterative_mode = false;
|
||||
{
|
||||
cg.SetPrintLevel(-1);
|
||||
cg.SetMaxIter(1000);
|
||||
cg.SetRelTol(1e-8);
|
||||
cg.Mult(B, X);
|
||||
MFEM_VERIFY(cg.GetConverged(), "CG solver did not converge!");
|
||||
}
|
||||
cg.SetRelTol(0.0);
|
||||
cg.SetMaxIter(max_it);
|
||||
cg.SetPrintLevel(print_lvl);
|
||||
cg.iterative_mode = false;
|
||||
MFEM_DEVICE_SYNC;
|
||||
|
||||
benchmark();
|
||||
mdofs = 0.0;
|
||||
}
|
||||
|
||||
void benchmark() override
|
||||
@@ -130,104 +230,115 @@ struct Problem : public BakeOff<VDIM, GLL>
|
||||
}
|
||||
};
|
||||
|
||||
/// Bake-off Problems (BPs)
|
||||
#define BakeOff_Problem(i, Kernel, VDIM, p_eq_q) \
|
||||
static void BP##i(bm::State &state) \
|
||||
{ \
|
||||
Problem<Kernel##Integrator, VDIM, p_eq_q> ker(state.range(0)); \
|
||||
while (state.KeepRunning()) { ker.benchmark(); } \
|
||||
state.counters["MDof/s"] = \
|
||||
bm::Counter(ker.SumMdofs(), bm::Counter::kIsRate); \
|
||||
} \
|
||||
BENCHMARK(BP##i)->DenseRange(1, 6)->Unit(bm::kMillisecond);
|
||||
|
||||
/// BP1: scalar PCG with mass matrix, q=p+2
|
||||
BakeOff_Problem(1, Mass, 1, false)
|
||||
|
||||
/// BP2: vector PCG with mass matrix, q=p+2
|
||||
BakeOff_Problem(2, VectorMass, 3, false)
|
||||
|
||||
/// BP3: scalar PCG with stiffness matrix, q=p+2
|
||||
BakeOff_Problem(3, Diffusion, 1, false)
|
||||
|
||||
/// BP4: vector PCG with stiffness matrix, q=p+2
|
||||
BakeOff_Problem(4, VectorDiffusion, 3, false)
|
||||
|
||||
/// BP5: scalar PCG with stiffness matrix, q=p+1
|
||||
BakeOff_Problem(5, Diffusion, 1, true)
|
||||
|
||||
/// BP6: vector PCG with stiffness matrix, q=p+1
|
||||
BakeOff_Problem(6, VectorDiffusion, 3, true)
|
||||
|
||||
/// Bake-off Kernels (BKs)
|
||||
template <typename BFI, int VDIM, bool GLL>
|
||||
struct Kernel : public BakeOff<VDIM, GLL>
|
||||
// Bake-off Kernels (BKs)
|
||||
template <int BFI, int VDIM, bool GLL>
|
||||
struct BK : public BakeOff<BFI, VDIM, GLL>
|
||||
{
|
||||
using BakeOff<VDIM, GLL>::a;
|
||||
using BakeOff<VDIM, GLL>::ir;
|
||||
using BakeOff<VDIM, GLL>::one;
|
||||
using BakeOff<VDIM, GLL>::fes;
|
||||
using BakeOff<VDIM, GLL>::x;
|
||||
using BakeOff<VDIM, GLL>::y;
|
||||
using BakeOff<VDIM, GLL>::mdofs;
|
||||
Vector xe, ye;
|
||||
|
||||
Kernel(int order): BakeOff<VDIM, GLL>(order)
|
||||
using base = BakeOff<BFI, VDIM, GLL>;
|
||||
using base::ir;
|
||||
using base::one;
|
||||
using base::bfi;
|
||||
using base::fes;
|
||||
using base::mdofs;
|
||||
|
||||
BK(int order, int side) noexcept: base(order, side)
|
||||
{
|
||||
x.Randomize(1);
|
||||
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
a.AddDomainIntegrator(new BFI(one, ir));
|
||||
a.Assemble();
|
||||
a.Mult(x, y);
|
||||
MFEM_DEVICE_SYNC;
|
||||
bfi->AssemblePA(fes);
|
||||
|
||||
const Table &el2dof = fes.GetElementToDofTable();
|
||||
const int e_size = el2dof.Size_of_connections()*fes.GetVDim();
|
||||
const auto R = fes.GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC);
|
||||
MFEM_VERIFY(e_size == R->Height(), "Input/Output E-vector size mismatch!");
|
||||
|
||||
xe.SetSize(R->Height());
|
||||
ye.SetSize(R->Height());
|
||||
xe.UseDevice(true);
|
||||
ye.UseDevice(true);
|
||||
|
||||
xe.Randomize(1);
|
||||
xe.Read();
|
||||
ye = 0.0;
|
||||
|
||||
benchmark();
|
||||
mdofs = 0.0;
|
||||
}
|
||||
|
||||
void benchmark() override
|
||||
{
|
||||
a.Mult(x, y);
|
||||
bfi->AddMultPA(xe, ye);
|
||||
MFEM_DEVICE_SYNC;
|
||||
mdofs += this->MDofs();
|
||||
}
|
||||
};
|
||||
|
||||
/// Generic CEED BKi
|
||||
#define BakeOff_Kernel(i, KER, VDIM, GLL) \
|
||||
static void BK##i(bm::State &state) \
|
||||
{ \
|
||||
Kernel<KER##Integrator, VDIM, GLL> ker(state.range(0)); \
|
||||
while (state.KeepRunning()) { ker.benchmark(); } \
|
||||
state.counters["MDof/s"] = \
|
||||
bm::Counter(ker.SumMdofs(), bm::Counter::kIsRate); \
|
||||
} \
|
||||
BENCHMARK(BK##i)->DenseRange(1, 6)->Unit(bm::kMillisecond);
|
||||
// Benchmarks
|
||||
template <typename T>
|
||||
static void Benchmark(bm::State& state) noexcept
|
||||
{
|
||||
T run(state.range(0), state.range(1));
|
||||
while (state.KeepRunning()) { run.benchmark(); }
|
||||
state.counters["Dofs"] = bm::Counter(run.dofs);
|
||||
state.counters["MDof/s"] = bm::Counter(run.SumMdofs(), bm::Counter::kIsRate);
|
||||
state.counters["Order"] = bm::Counter(state.range(0));
|
||||
}
|
||||
|
||||
/// BK1: scalar E-vector-to-E-vector evaluation of mass matrix, q=p+2
|
||||
BakeOff_Kernel(1, Mass, 1, false)
|
||||
#define REGISTER(PK, BFI, VDIM, GLL) \
|
||||
BENCHMARK_TEMPLATE(Benchmark, PK<BFI, VDIM, GLL>) \
|
||||
->Name(#PK #BFI)->Apply(CustomArguments)->Unit(bm::kMillisecond)
|
||||
|
||||
/// BK2: vector E-vector-to-E-vector evaluation of mass matrix, q=p+2
|
||||
BakeOff_Kernel(2, VectorMass, 3, false)
|
||||
// BP1: scalar PCG with mass matrix, q=p+2
|
||||
REGISTER(BP, 1, 1, false);
|
||||
|
||||
/// BK3: scalar E-vector-to-E-vector evaluation of stiffness matrix, q=p+2
|
||||
BakeOff_Kernel(3, Diffusion, 1, false)
|
||||
// BP2: vector PCG with mass matrix, q=p+2
|
||||
REGISTER(BP, 2, 3, false);
|
||||
|
||||
/// BK4: vector E-vector-to-E-vector evaluation of stiffness matrix, q=p+2
|
||||
BakeOff_Kernel(4, VectorDiffusion, 3, false)
|
||||
// BP3: scalar PCG with stiffness matrix, q=p+2
|
||||
REGISTER(BP, 3, 1, false);
|
||||
|
||||
/// BK5: scalar E-vector-to-E-vector evaluation of stiffness matrix, q=p+1
|
||||
BakeOff_Kernel(5, Diffusion, 1, true)
|
||||
// BP4: vector PCG with stiffness matrix, q=p+2
|
||||
REGISTER(BP, 4, 3, false);
|
||||
|
||||
/// BK6: vector E-vector-to-E-vector evaluation of stiffness matrix, q=p+1
|
||||
BakeOff_Kernel(6, VectorDiffusion, 3, true)
|
||||
// BP5: scalar PCG with stiffness matrix, q=p+1
|
||||
REGISTER(BP, 5, 1, true);
|
||||
|
||||
// BP6: vector PCG with stiffness matrix, q=p+1
|
||||
REGISTER(BP, 6, 3, true);
|
||||
|
||||
// BK1: scalar E-vector-to-E-vector evaluation of mass matrix, q=p+2
|
||||
REGISTER(BK, 1, 1, false);
|
||||
|
||||
// BK2: vector E-vector-to-E-vector evaluation of mass matrix, q=p+2
|
||||
REGISTER(BK, 2, 3, false);
|
||||
|
||||
// BK3: scalar E-vector-to-E-vector evaluation of stiffness matrix, q=p+2
|
||||
REGISTER(BK, 3, 1, false);
|
||||
|
||||
// BK4: vector E-vector-to-E-vector evaluation of stiffness matrix, q=p+2
|
||||
REGISTER(BK, 4, 3, false);
|
||||
|
||||
// BK5: scalar E-vector-to-E-vector evaluation of stiffness matrix, q=p+1
|
||||
REGISTER(BK, 5, 1, true);
|
||||
|
||||
// BK6: vector E-vector-to-E-vector evaluation of stiffness matrix, q=p+1
|
||||
REGISTER(BK, 6, 3, true);
|
||||
|
||||
/**
|
||||
* @brief main entry point
|
||||
* --benchmark_filter=BK1/6
|
||||
* --benchmark_context=device=cpu
|
||||
* @brief CEED Bake-off Problems main entry point
|
||||
* Command line options:
|
||||
* --benchmark_context=device=gpu
|
||||
* --benchmark_filter=BP1
|
||||
* --benchmark_out_format=csv
|
||||
* --benchmark_out=bp1.csv
|
||||
*/
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
bm::ConsoleReporter CR;
|
||||
bm::Initialize(&argc, argv);
|
||||
|
||||
AddKernelSpecializations();
|
||||
|
||||
// Device setup, cpu by default
|
||||
std::string device_config = "cpu";
|
||||
auto global_context = bmi::GetGlobalContext();
|
||||
@@ -240,12 +351,16 @@ int main(int argc, char *argv[])
|
||||
device_config = device->second;
|
||||
}
|
||||
}
|
||||
|
||||
Device device(device_config.c_str());
|
||||
device.Print();
|
||||
|
||||
if (bm::ReportUnrecognizedArguments(argc, argv)) { return 1; }
|
||||
if (bm::ReportUnrecognizedArguments(argc, argv)) { return EXIT_FAILURE; }
|
||||
|
||||
bm::RunSpecifiedBenchmarks(&CR);
|
||||
return 0;
|
||||
bm::Shutdown();
|
||||
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
|
||||
#endif // MFEM_USE_BENCHMARK
|
||||
|
||||
@@ -101,6 +101,7 @@ set(UNIT_TESTS_SRCS
|
||||
fem/test_calcdshape.cpp
|
||||
fem/test_calcshape.cpp
|
||||
fem/test_calcvshape.cpp
|
||||
fem/test_calchessian.cpp
|
||||
fem/test_coefficient.cpp
|
||||
fem/test_col_lag_der.cpp
|
||||
fem/test_datacollection.cpp
|
||||
|
||||
@@ -320,8 +320,12 @@ TEST_CASE("NormalTraceJumpIntegrator Element Assembly", "[AssemblyLevel][GPU]")
|
||||
{
|
||||
const auto fname = GENERATE(
|
||||
"../../data/inline-quad.mesh",
|
||||
"../../data/amr-quad.mesh",
|
||||
"../../data/beam-quad-amr.mesh",
|
||||
"../../data/star-q3.mesh",
|
||||
"../../data/inline-hex.mesh",
|
||||
"../../data/amr-hex.mesh",
|
||||
"../../data/fichera-amr.mesh",
|
||||
"../../data/fichera-q3.mesh"
|
||||
);
|
||||
const int order = GENERATE(1, 2, 3);
|
||||
@@ -356,7 +360,7 @@ TEST_CASE("NormalTraceJumpIntegrator Element Assembly", "[AssemblyLevel][GPU]")
|
||||
for (int f = 0; f < mesh.GetNumFaces(); ++f)
|
||||
{
|
||||
const Mesh::FaceInformation info = mesh.GetFaceInformation(f);
|
||||
if (!info.IsInterior()) { continue; }
|
||||
if (!info.IsInterior() || info.IsNonconformingCoarse()) { continue; }
|
||||
|
||||
const int el1 = info.element[0].index;
|
||||
const int el2 = info.element[1].index;
|
||||
|
||||
@@ -0,0 +1,441 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "unit_tests.hpp"
|
||||
|
||||
#include <iostream>
|
||||
#include <cmath>
|
||||
|
||||
using namespace mfem;
|
||||
|
||||
|
||||
/**
|
||||
* Compute the error of the taylor series expansion of the shapefunctions, upto
|
||||
* and including the hessian term:
|
||||
* res = shape(xi) + dshape(xi)*eps*dx + 0.5*hessian(xi)*eps*eps*dx*dx
|
||||
* - shape(xi + eps*dx)
|
||||
*/
|
||||
real_t TaylorSeriesError(const FiniteElement* fe,
|
||||
const IntegrationPoint &ip,
|
||||
const Vector &dx,
|
||||
const real_t eps)
|
||||
{
|
||||
const int dof = fe->GetDof();
|
||||
const int dim = fe->GetDim();
|
||||
const int hdim = (dim*(dim+1))/2;
|
||||
|
||||
Vector shape(dof);
|
||||
DenseMatrix dshape(dof,dim);
|
||||
DenseMatrix hessian(dof,hdim);
|
||||
|
||||
fe->CalcShape(ip, shape);
|
||||
fe->CalcDShape(ip, dshape);
|
||||
fe->CalcHessian(ip, hessian);
|
||||
|
||||
Vector dx2(hdim);
|
||||
if (dim == 1)
|
||||
{
|
||||
dx2[0] = dx[0]*dx[0];
|
||||
}
|
||||
else if (dim == 2)
|
||||
{
|
||||
dx2[0] = dx[0]*dx[0];
|
||||
dx2[1] = 2*dx[0]*dx[1];
|
||||
dx2[2] = dx[1]*dx[1];
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
dx2[0] = dx[0]*dx[0];
|
||||
dx2[1] = 2*dx[0]*dx[1];
|
||||
dx2[2] = 2*dx[0]*dx[2];
|
||||
dx2[3] = dx[1]*dx[1];
|
||||
dx2[4] = 2*dx[1]*dx[2];
|
||||
dx2[5] = dx[2]*dx[2];
|
||||
}
|
||||
|
||||
Vector res(dof);
|
||||
res = shape;
|
||||
dshape.AddMult(dx, res, eps);
|
||||
hessian.AddMult(dx2, res, 0.5*eps*eps);
|
||||
|
||||
IntegrationPoint ip_eps;
|
||||
Vector shape_eps(dof);
|
||||
ip_eps.x = ip.x + eps*dx[0];
|
||||
if (dim >= 2 ) { ip_eps.y = ip.y + eps*dx[1]; }
|
||||
if (dim == 3 ) { ip_eps.z = ip.z + eps*dx[2]; }
|
||||
|
||||
fe->CalcShape(ip_eps, shape_eps);
|
||||
res -= shape_eps;
|
||||
return res.Norml2();
|
||||
}
|
||||
|
||||
/**
|
||||
* Check the convergence of the taylor series, of a given element @a fe at
|
||||
* a given point @a ip in a given direction @a dx.
|
||||
* For linear and quadratic elements the taylor series is exact.
|
||||
* For other elements the convergence should be third order.
|
||||
*/
|
||||
|
||||
void CheckTaylorSeries(const FiniteElement* fe,
|
||||
const IntegrationPoint &ip,
|
||||
const Vector &dx)
|
||||
{
|
||||
real_t eps = 0.1;
|
||||
constexpr real_t red = 4.0;
|
||||
constexpr int steps = 100;
|
||||
constexpr real_t tol = 1e-8;
|
||||
|
||||
real_t error = TaylorSeriesError(fe, ip, dx, eps);
|
||||
real_t order;
|
||||
int i;
|
||||
for (i = 0; i < steps; ++i)
|
||||
{
|
||||
eps /= red;
|
||||
real_t err_new = TaylorSeriesError(fe, ip, dx, eps);
|
||||
order = log(error/err_new)/log(red);
|
||||
error = err_new;
|
||||
if (error < tol) { break; }
|
||||
}
|
||||
mfem::out<<i<<" "<<error<<" "<<order<<std::endl;
|
||||
if (i == 0)
|
||||
{
|
||||
REQUIRE(error == MFEM_Approx(0));
|
||||
}
|
||||
else
|
||||
{
|
||||
REQUIRE(order > 2.98);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Test if a given element @a fe has the correct behaviour of the taylor series.
|
||||
*/
|
||||
void TestCalcHessian(const FiniteElement* fe)
|
||||
{
|
||||
const int dim = fe->GetDim();
|
||||
|
||||
constexpr int check_res = 2;
|
||||
int num_check_dirs = dim;
|
||||
|
||||
// Get a uniform grid of integration points
|
||||
RefinedGeometry* ref = GlobGeometryRefiner.Refine(fe->GetGeomType(),
|
||||
check_res);
|
||||
const IntegrationRule& intRule = ref->RefPts;
|
||||
int npoints = intRule.GetNPoints();
|
||||
Vector dx(dim);
|
||||
for (int i=0; i < npoints; ++i)
|
||||
{
|
||||
// Get the current integration point from intRule
|
||||
IntegrationPoint pt = intRule.IntPoint(i);
|
||||
|
||||
for (int j=0; j < num_check_dirs; ++j)
|
||||
{
|
||||
dx[0] = sin(2*j + 0.3);
|
||||
if (dim >= 2) { dx[1] = cos(5*j + 0.2); }
|
||||
if (dim == 3) { dx[2] = sin(3*j + 0.1); }
|
||||
CheckTaylorSeries(fe, pt, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
TEST_CASE("CalcHessian",
|
||||
"[Linear1DFiniteElement]"
|
||||
"[Linear2DFiniteElement]"
|
||||
"[Linear3DFiniteElement]"
|
||||
"[BiLinear2DFiniteElement]"
|
||||
"[TriLinear3DFiniteElement]"
|
||||
"[H1_SegmentElement]"
|
||||
"[H1_QuadrilateralElement]"
|
||||
"[H1_HexahedronElement]"
|
||||
"[H1_TriangleElement]"
|
||||
"[H1_TetrahedronElement]"
|
||||
"[NURBS1DFiniteElement]"
|
||||
"[NURBS2DFiniteElement]"
|
||||
"[NURBS3DFiniteElement]")
|
||||
{
|
||||
|
||||
// Fixed Order Elements
|
||||
SECTION("Linear1DFiniteElement")
|
||||
{
|
||||
mfem::out<<"Linear1DFiniteElement"<<std::endl;
|
||||
Linear1DFiniteElement fe;
|
||||
TestCalcHessian(&fe);
|
||||
}
|
||||
|
||||
SECTION("Linear2DFiniteElement")
|
||||
{
|
||||
mfem::out<<"Linear2DFiniteElement"<<std::endl;
|
||||
Linear2DFiniteElement fe;
|
||||
TestCalcHessian(&fe);
|
||||
}
|
||||
|
||||
SECTION("Linear3DFiniteElement")
|
||||
{
|
||||
mfem::out<<"Linear3DFiniteElement"<<std::endl;
|
||||
Linear3DFiniteElement fe;
|
||||
TestCalcHessian(&fe);
|
||||
}
|
||||
|
||||
SECTION("BiLinear2DFiniteElement")
|
||||
{
|
||||
mfem::out<<"BiLinear2DFiniteElement"<<std::endl;
|
||||
BiLinear2DFiniteElement fe;
|
||||
TestCalcHessian(&fe);
|
||||
}
|
||||
|
||||
SECTION("TriLinear3DFiniteElement")
|
||||
{
|
||||
mfem::out<<"TriLinear3DFiniteElement"<<std::endl;
|
||||
TriLinear3DFiniteElement fe;
|
||||
TestCalcHessian(&fe);
|
||||
}
|
||||
|
||||
// H1 Elements
|
||||
SECTION("H1_SegmentElement")
|
||||
{
|
||||
int order = GENERATE(1,2,3,4,5);
|
||||
mfem::out<<"H1_SegmentElement = "<<order<<std::endl;
|
||||
H1_SegmentElement fe(order);
|
||||
TestCalcHessian(&fe);
|
||||
}
|
||||
|
||||
SECTION("H1_QuadrilateralElement")
|
||||
{
|
||||
int order = GENERATE(1,2,3,4,5);
|
||||
H1_QuadrilateralElement fe(order);
|
||||
mfem::out<<"H1_QuadrilateralElement = "<<order<<std::endl;
|
||||
TestCalcHessian(&fe);
|
||||
}
|
||||
|
||||
SECTION("H1_HexahedronElement")
|
||||
{
|
||||
int order = GENERATE(1,2,3,4,5);
|
||||
mfem::out<<"H1_HexahedronElement = "<<order<<std::endl;
|
||||
H1_HexahedronElement fe(order);
|
||||
TestCalcHessian(&fe);
|
||||
}
|
||||
|
||||
SECTION("H1_TriangleElement")
|
||||
{
|
||||
int order = GENERATE(1,2,3,4,5);
|
||||
mfem::out<<"H1_TriangleElement = "<<order<<std::endl;
|
||||
H1_TriangleElement fe(order);
|
||||
TestCalcHessian(&fe);
|
||||
}
|
||||
|
||||
SECTION("H1_TetrahedronElement")
|
||||
{
|
||||
int order = GENERATE(1,2,3,4,5);
|
||||
mfem::out<<"H1_TetrahedronElement = "<<order<<std::endl;
|
||||
H1_TetrahedronElement fe(order);
|
||||
TestCalcHessian(&fe);
|
||||
}
|
||||
|
||||
// NURBS Elements
|
||||
SECTION("NURBS1DFiniteElement")
|
||||
{
|
||||
int order = GENERATE(1,2,3,4,5);
|
||||
mfem::out<<"NURBS1DFiniteElement = "<<order<<std::endl;
|
||||
NURBS1DFiniteElement fe(order);
|
||||
Array <const KnotVector*> kv(1);
|
||||
kv[0] = new KnotVector(order);
|
||||
fe.KnotVectors() = kv;
|
||||
int IJK[1];
|
||||
IJK[0] = 0;
|
||||
fe.SetIJK(IJK);
|
||||
fe.SetOrder();
|
||||
fe.Weights() = 1.0;
|
||||
TestCalcHessian(&fe);
|
||||
delete kv[0];
|
||||
}
|
||||
|
||||
SECTION("NURBS2DFiniteElement")
|
||||
{
|
||||
int order = GENERATE(1,2,3,4,5);
|
||||
mfem::out<<"NURBS2DFiniteElement = "<<order<<std::endl;
|
||||
NURBS2DFiniteElement fe(order);
|
||||
Array <const KnotVector*> kv(2);
|
||||
kv[0] = new KnotVector(order);
|
||||
kv[1] = new KnotVector(order);
|
||||
fe.KnotVectors() = kv;
|
||||
int IJK[2];
|
||||
IJK[0] = IJK[1] = 0;
|
||||
fe.SetIJK(IJK);
|
||||
fe.SetOrder();
|
||||
fe.Weights() = 1.0;
|
||||
TestCalcHessian(&fe);
|
||||
delete kv[0];
|
||||
delete kv[1];
|
||||
}
|
||||
|
||||
SECTION("NURBS3DFiniteElement")
|
||||
{
|
||||
int order = GENERATE(1,2,3,4,5);
|
||||
mfem::out<<"NURBS3DFiniteElement = "<<order<<std::endl;
|
||||
NURBS3DFiniteElement fe(order);
|
||||
Array <const KnotVector*> kv(3);
|
||||
kv[0] = new KnotVector(order);
|
||||
kv[1] = new KnotVector(order);
|
||||
kv[2] = new KnotVector(order);
|
||||
fe.KnotVectors() = kv;
|
||||
int IJK[3];
|
||||
IJK[0] = IJK[1] = IJK[2] = 0;
|
||||
fe.SetIJK(IJK);
|
||||
fe.SetOrder();
|
||||
fe.Weights() = 1.0;
|
||||
TestCalcHessian(&fe);
|
||||
delete kv[0];
|
||||
delete kv[1];
|
||||
delete kv[2];
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("Laplacian",
|
||||
"[NURBS2DFiniteElement]"
|
||||
"[NURBS3DFiniteElement]")
|
||||
{
|
||||
int order = 4;
|
||||
std::string meshName = GENERATE("square-nurbs.mesh",
|
||||
"cube-nurbs.mesh");
|
||||
mfem::out<<"\nCheck laplacian for "<< meshName <<std::endl;
|
||||
bool deformed = GENERATE(false,true);
|
||||
if (deformed) { mfem::out<<"Mesh is deformed"<<std::endl; }
|
||||
bool NURBS = GENERATE(false,true);
|
||||
if (NURBS) { mfem::out<<"Using NURBS"<<std::endl; }
|
||||
|
||||
Mesh mesh("../../data/" + meshName, 1, 1);
|
||||
const int dim = mesh.Dimension();
|
||||
|
||||
// Rotate mesh
|
||||
DenseMatrix Rotate(dim);
|
||||
if (dim == 2)
|
||||
{
|
||||
NURBSPatch::Get2DRotationMatrix(M_PI/7, Rotate);
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
real_t n[] = {0.0,0.0,1.0};
|
||||
NURBSPatch::Get3DRotationMatrix(n, M_PI/7,M_PI/7, Rotate);
|
||||
}
|
||||
|
||||
Vector x0(dim), x1(dim);
|
||||
for (int i = 0; i <mesh.GetNodes()->Size()/dim; i++)
|
||||
{
|
||||
mesh.GetNode(i, x0.GetData());
|
||||
Rotate.Mult(x0, x1);
|
||||
mesh.SetNode(i, x1.GetData());
|
||||
}
|
||||
|
||||
// Distort mesh
|
||||
real_t distort_scale = 0.05;
|
||||
if (deformed)
|
||||
{
|
||||
Vector dx(mesh.GetNodes()->Size());
|
||||
dx.Randomize(1234);
|
||||
dx *= 2.0; dx -= 1.0; dx *= distort_scale;
|
||||
mesh.MoveNodes(dx);
|
||||
}
|
||||
|
||||
if (NURBS)
|
||||
{
|
||||
// We need a C1 smooth mesh
|
||||
mesh.DegreeElevate(1);
|
||||
|
||||
// Refine mesh
|
||||
mesh.UniformRefinement();
|
||||
|
||||
// Distort mesh
|
||||
distort_scale = 0.01;
|
||||
if (deformed)
|
||||
{
|
||||
Vector dx(mesh.GetNodes()->Size());
|
||||
dx.Randomize(1234);
|
||||
dx *= 2.0; dx -= 1.0; dx *= distort_scale;
|
||||
mesh.MoveNodes(dx);
|
||||
}
|
||||
}
|
||||
|
||||
// Create Space
|
||||
FiniteElementCollection *fe_coll = nullptr;
|
||||
NURBSExtension *ext = nullptr;
|
||||
if (NURBS)
|
||||
{
|
||||
fe_coll = new NURBSFECollection (order);
|
||||
ext = new NURBSExtension(mesh.NURBSext, order);
|
||||
}
|
||||
else
|
||||
{
|
||||
fe_coll = new H1_FECollection (order);
|
||||
}
|
||||
FiniteElementSpace fes(&mesh, ext, fe_coll);
|
||||
|
||||
// Compute (grad w, grad phi) + (w, laplace phi) = 0
|
||||
SparseMatrix gmat(fes.GetNDofs());
|
||||
Vector shape, lshape;
|
||||
DenseMatrix dshape, elmat;
|
||||
|
||||
DofTransformation doftrans;
|
||||
ElementTransformation *eltrans;
|
||||
Array<int> vdofs;
|
||||
for (int e = 0; e < fes.GetNE(); e++)
|
||||
{
|
||||
const int dof = fes.GetFE(e)->GetDof();
|
||||
shape.SetSize(dof);
|
||||
dshape.SetSize(dof,dim);
|
||||
lshape.SetSize(dof);
|
||||
|
||||
elmat.SetSize(dof);
|
||||
elmat = 0.0;
|
||||
eltrans = fes.GetElementTransformation (e);
|
||||
|
||||
// Integrand involves non-polynomial mapping
|
||||
const int intorder = 3*fes.GetFE(e)->GetOrder();
|
||||
const IntegrationRule *ir = &IntRules.Get(fes.GetFE(e)->GetGeomType(),
|
||||
intorder);
|
||||
|
||||
elmat = 0.0;
|
||||
for (int i = 0; i < ir->GetNPoints(); i++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(i);
|
||||
eltrans->SetIntPoint(&ip);
|
||||
const real_t w = ip.weight * eltrans->Weight();
|
||||
|
||||
fes.GetFE(e)->CalcShape(ip, shape);
|
||||
fes.GetFE(e)->CalcPhysLaplacian(*eltrans, lshape);
|
||||
fes.GetFE(e)->CalcPhysDShape(*eltrans, dshape);
|
||||
|
||||
// Check Laplacian
|
||||
AddMult_a_AAt (w, dshape, elmat);
|
||||
AddMult_a_VWt (w, shape, lshape, elmat);
|
||||
}
|
||||
|
||||
// Add to global matrix
|
||||
fes.GetElementVDofs (e, vdofs);
|
||||
gmat.AddSubMatrix (vdofs, vdofs, elmat, 1);
|
||||
}
|
||||
|
||||
// Apply homogeneous essential boundary conditions on entire boundary
|
||||
Array<int> ess_dofs;
|
||||
fes.GetBoundaryTrueDofs(ess_dofs);
|
||||
for (int i=0; i<ess_dofs.Size(); i++)
|
||||
{
|
||||
gmat.EliminateRowCol(ess_dofs[i], Operator::DiagonalPolicy::DIAG_ZERO);
|
||||
}
|
||||
gmat.Finalize (1);
|
||||
mfem::out<<"Difference between matrices = "<< gmat.MaxNorm() <<std::endl;
|
||||
// Tolerance can be tighter if intorder is increased
|
||||
REQUIRE(gmat.MaxNorm() == MFEM_Approx(0.0, 1e-8));
|
||||
|
||||
delete fe_coll;
|
||||
}
|
||||
|
||||
@@ -47,13 +47,14 @@ TEST_CASE("GSLIBInterpolate", "[GSLIBInterpolate][GSLIB]")
|
||||
int point_ordering = GENERATE(0, 1);
|
||||
int ncomp = GENERATE(1, 2);
|
||||
int gf_ordering = GENERATE(0, 1);
|
||||
int func_out_ordering = GENERATE(0, 1);
|
||||
bool href = GENERATE(true, false);
|
||||
bool pref = GENERATE(true, false);
|
||||
|
||||
int ne = 4;
|
||||
|
||||
CAPTURE(space, simplex, dim, func_order, mesh_order, mesh_node_ordering,
|
||||
point_ordering, ncomp, gf_ordering, href, pref);
|
||||
point_ordering, ncomp, gf_ordering, func_out_ordering, href, pref);
|
||||
|
||||
if (ncomp == 1 && gf_ordering == 1)
|
||||
{
|
||||
@@ -145,7 +146,8 @@ TEST_CASE("GSLIBInterpolate", "[GSLIBInterpolate][GSLIB]")
|
||||
FindPointsGSLIB finder;
|
||||
finder.Setup(mesh);
|
||||
finder.SetL2AvgType(FindPointsGSLIB::NONE);
|
||||
finder.Interpolate(vxyz, field_vals, interp_vals, point_ordering);
|
||||
finder.Interpolate(vxyz, field_vals, interp_vals, point_ordering,
|
||||
func_out_ordering);
|
||||
Array<unsigned int> code_out = finder.GetCode();
|
||||
Vector dist_p_out = finder.GetDist();
|
||||
|
||||
@@ -168,7 +170,7 @@ TEST_CASE("GSLIBInterpolate", "[GSLIBInterpolate][GSLIB]")
|
||||
{
|
||||
if (code_out[i] < 2)
|
||||
{
|
||||
err = gf_ordering == Ordering::byNODES ?
|
||||
err = func_out_ordering == Ordering::byNODES ?
|
||||
fabs(exact_val(j) - interp_vals[i + j*pts_cnt]) :
|
||||
fabs(exact_val(j) - interp_vals[i*ncomp + j]);
|
||||
max_err = std::max(max_err, err);
|
||||
|
||||
@@ -464,4 +464,80 @@ TEST_CASE("QuadratureInterpolator", "[QuadratureInterpolator][GPU]")
|
||||
REQUIRE(rel_error_norm == MFEM_Approx(0.0));
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("Surface Determinants: 1D surface in 2D/3D and 2D surface in 3D")
|
||||
{
|
||||
const auto mesh_fname = GENERATE(
|
||||
"../../data/diag-segment-2d.mesh", // 1D in 2D
|
||||
"../../data/diag-segment-3d.mesh", // 1D in 3D
|
||||
"../../data/star-surf.mesh" // 2D in 3D
|
||||
);
|
||||
|
||||
// Using order > 1 to ensure curvature is used if supported by mesh
|
||||
const int order = 3;
|
||||
|
||||
Mesh mesh = Mesh::LoadFromFile(mesh_fname);
|
||||
const int dim = mesh.Dimension();
|
||||
const int sdim = mesh.SpaceDimension();
|
||||
|
||||
REQUIRE(dim < sdim);
|
||||
|
||||
// Ensure high-order curvature for non-trivial Jacobians where possible
|
||||
mesh.SetCurvature(order);
|
||||
|
||||
const FiniteElementSpace *fes = mesh.GetNodalFESpace();
|
||||
GridFunction *nodes = mesh.GetNodes();
|
||||
|
||||
// Quadrature space
|
||||
QuadratureSpace qs(&mesh, 2*order);
|
||||
const QuadratureInterpolator *qi = fes->GetQuadratureInterpolator(qs);
|
||||
qi->SetOutputLayout(QVectorLayout::byVDIM);
|
||||
|
||||
// Prepare E-vector from nodes
|
||||
const ElementDofOrdering ordering =
|
||||
(mesh.Dimension() == 1 || mesh.MeshGenerator() == 2) ?
|
||||
ElementDofOrdering::LEXICOGRAPHIC : ElementDofOrdering::NATIVE;
|
||||
|
||||
const Operator *R = fes->GetElementRestriction(ordering);
|
||||
Vector e_vec(R->Height());
|
||||
R->Mult(*nodes, e_vec);
|
||||
|
||||
// Compute determinants (weights) via QI
|
||||
// Output vector size: qs.GetSize() * 1 (since determinant is scalar)
|
||||
Vector q_det(qs.GetSize());
|
||||
qi->Determinants(e_vec, q_det);
|
||||
|
||||
// Verify against ElementTransformation::Weight()
|
||||
Vector q_weights(qs.GetSize());
|
||||
const int ne = qs.GetNE();
|
||||
int idx_counter = 0;
|
||||
for (int i = 0; i < ne; i++)
|
||||
{
|
||||
ElementTransformation *T = mesh.GetElementTransformation(i);
|
||||
const IntegrationRule &ir = qs.GetIntRule(i);
|
||||
for (int j = 0; j < ir.GetNPoints(); j++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(j);
|
||||
T->SetIntPoint(&ip);
|
||||
q_weights(idx_counter++) = T->Weight();
|
||||
}
|
||||
}
|
||||
|
||||
// Compare
|
||||
Vector diff = q_det;
|
||||
diff -= q_weights;
|
||||
const real_t norm_w = q_weights.Normlinf();
|
||||
const real_t norm_d = diff.Normlinf();
|
||||
|
||||
// If weights are effectively zero (e.g. degenerate), direct comparison might differ
|
||||
// but for these valid meshes, weight should be > 0.
|
||||
if (norm_w > 1e-12)
|
||||
{
|
||||
REQUIRE(norm_d / norm_w < 1e-12);
|
||||
}
|
||||
else
|
||||
{
|
||||
REQUIRE(norm_d < 1e-12);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user