Compare commits

..
Author SHA1 Message Date
thatguynoe 2a8898a89e correct essential boundary comment 2026-07-07 15:07:41 -04:00
thatguynoe 07c92a7e00 Merge branch 'ex43' of https://github.com/mfem/mfem into ex43 2026-07-07 14:23:25 -04:00
Noe ReyesandSocratis Petrides 66e5d3782b set ess_bdr correctly in parallel
Co-authored-by: Socratis Petrides <petrides1@llnl.gov>
2026-07-07 14:23:13 -04:00
thatguynoe 4d3928d9b9 decrease solver tolerance in serial 2026-07-07 14:21:12 -04:00
thatguynoe 784148ca84 use doxygen compatible comments 2026-07-07 14:19:00 -04:00
thatguynoe aed9fb51c2 remove useless variable 2026-07-07 13:40:08 -04:00
thatguynoe 1bb3b662c6 add boundary-only check 2026-07-07 13:38:08 -04:00
thatguynoe e92710ae06 correct variable name 2026-07-07 13:35:38 -04:00
thatguynoe c95bbd8c56 add sample run for star.mesh 2026-01-22 00:18:20 -05:00
thatguynoe f9d9479d4d update examples/CMakeLists.txt 2026-01-21 20:24:39 -05:00
thatguynoe c6b5e80ebc update doc/CodeDocumentation.dox 2026-01-21 20:24:22 -05:00
Noe Reyes 8944ee325b Merge branch 'master' into ex43 2026-01-21 20:12:35 -05:00
Brendan Keith 31551a3b03 Merge branch 'master' into ex43 2025-12-22 17:34:07 -05:00
thatguynoe 411a35656e apply style 2025-12-02 13:22:06 -05:00
Noe Reyes 125e883264 Merge branch 'master' into ex43 2025-12-02 12:53:36 -05:00
thatguynoe cff888bbaa shorten name of linear form integrator 2025-12-02 12:51:14 -05:00
thatguynoe ca8aa8aef2 include connection with ex28 2025-12-02 12:49:44 -05:00
thatguynoe 240dcdc693 correct comment about nt 2025-12-01 20:06:04 -05:00
thatguynoe b35a103a9d fixed -> displaced in description 2025-12-01 20:05:39 -05:00
thatguynoe 40edc5b23c add ex43 2025-12-01 17:26:27 -05:00
87 changed files with 2469 additions and 4457 deletions
-1
View File
@@ -369,7 +369,6 @@ miniapps/shifted/lsf_integral
miniapps/tools/display-basis
miniapps/tools/load-dc
miniapps/tools/convert-dc
miniapps/tools/compare-dc
miniapps/tools/gridfunction-bounds
miniapps/tools/lor-transfer
miniapps/tools/plor-transfer
-22
View File
@@ -22,11 +22,6 @@ Meshing improvements
- Improved support for 1D NURBS meshes with variable order, including using
the patches construct for 1D NURBS meshes.
New and updated examples and miniapps
-------------------------------------
- Electromagnetics/lorentz miniapp has been updated to leverage the ParticleSet
capability.
Version 4.9, released on Dec 11, 2025
=====================================
@@ -111,23 +106,6 @@ Linear and nonlinear solvers
Filtering (AMGF), providing robust preconditioning for linear systems arising
in constrained optimization problems such as frictionless contact.
Added 'GetResiduals' and 'GetFinalAbsResidualNorm' to 'HyprePCG',
'HypreGMRES', and 'HypreFGMRES' to get 'r' and '|r|_p'. Note that the latter
computes '|r|_p' from 'r' instead of returning a cached value like the
relative 'GetFinalResidualNorm'. These require Hypre >= 2.15.0.
Changed the default solver parameters for 'HyprePCG' to 'tol=1e-6' and
'max_iter=1000'. This matches the default parameters in Hypre 3.0.
Added various helper functions for querying/modifying Hypre solvers:
'HypreSmoother::GetType', 'HypreSmoother::GetSOROptions',
'HypreSmoother::GetPolyOptions', 'HypreSmoother::GetWindowParameters',
'HypreSmoother::IsOperatorSymmetric', 'HyprePCG::GetTol',
'HyprePCG::GetAbsTol', 'HyprePCG::GetMaxIter', 'HyprePCG::SetUseTwoNorm',
'HypreGMRES::GetTol', 'HypreGMRES::GetAbsTol', 'HypreGMRES::GetMaxIter',
'HypreGMRES::GetKDim', 'HypreFGMRES::GetTol', 'HypreFGMRES::GetMaxIter',
'HypreFGMRES::GetKDim', and 'HypreBoomerAMG::GetMaxIter'.
GPU computing
-------------
- Added the 'gpu', 'raja-gpu', and 'ceed-gpu' backend aliases/shortcuts which
-1
View File
@@ -18,7 +18,6 @@
# Some choices below are based on the OS type:
NOTMAC := $(subst Darwin,,$(shell uname -s))
ASTYLE_BIN = astyle
ETAGS_BIN = $(shell command -v etags 2> /dev/null)
EGREP_BIN = $(shell command -v egrep 2> /dev/null)
+2
View File
@@ -119,6 +119,8 @@ namespace mfem {
* - <a class="el" href="ex40p_8cpp_source.html">Example 40p</a>: parallel eikonal equation
* - <a class="el" href="ex41_8cpp_source.html">Example 41</a>: DG/CG IMEX time-dependent advection-diffusion
* - <a class="el" href="ex41p_8cpp_source.html">Example 41p</a>: parallel DG/CG IMEX time-dependent advection-diffusion
* - <a class="el" href="ex43_8cpp_source.html">Example 43</a>: sliding boundary conditions in linear elasticity
* - <a class="el" href="ex43p_8cpp_source.html">Example 43p</a>: parallel sliding boundary conditions in linear elasticity
*
* <H4>AmgX Examples</H4>
* - Variants of Examples
+2
View File
@@ -47,6 +47,7 @@ list(APPEND ALL_EXE_SRCS
ex39.cpp
ex40.cpp
ex41.cpp
ex43.cpp
)
if (MFEM_USE_MPI)
@@ -91,6 +92,7 @@ if (MFEM_USE_MPI)
ex39p.cpp
ex40p.cpp
ex41p.cpp
ex43p.cpp
)
endif()
+1 -5
View File
@@ -9,7 +9,6 @@
// ex4 -m ../data/beam-hex.mesh -o 2 -pa
// ex4 -m ../data/escher.mesh
// ex4 -m ../data/fichera.mesh -o 2 -hb
// ex4 -m ../data/fichera.mesh -o 2 -hb -ea
// ex4 -m ../data/fichera-q2.vtk
// ex4 -m ../data/fichera-q3.mesh -o 2 -sc
// ex4 -m ../data/square-disc-nurbs.mesh
@@ -19,7 +18,6 @@
// ex4 -m ../data/amr-quad.mesh
// ex4 -m ../data/amr-hex.mesh
// ex4 -m ../data/amr-hex.mesh -o 2 -hb
// ex4 -m ../data/amr-hex.mesh -o 2 -hb -ea
// ex4 -m ../data/fichera-amr.mesh -o 2 -sc
// ex4 -m ../data/ref-prism.mesh -o 1
// ex4 -m ../data/octahedron.mesh -o 1
@@ -27,8 +25,6 @@
//
// Device sample runs:
// ex4 -m ../data/star.mesh -pa -d cuda
// ex4 -m ../data/star.mesh -hb -ea -d cuda
// ex4 -m ../data/amr-quad.mesh -hb -ea -d cuda
// ex4 -m ../data/star.mesh -pa -d raja-cuda
// ex4 -m ../data/star.mesh -pa -d raja-omp
// ex4 -m ../data/beam-hex.mesh -pa -d cuda
@@ -197,7 +193,7 @@ int main(int argc, char *argv[])
cout << "Size of linear system: " << A->Height() << endl;
// 11. Solve the linear system A X = B.
if (!pa && (!ea || hybridization))
if (!pa)
{
#ifndef MFEM_USE_SUITESPARSE
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
+278
View File
@@ -0,0 +1,278 @@
// MFEM Example 43
//
// Compile with: make ex43
//
// Sample runs: ex43 -m ../data/ball-nurbs.mesh -r 2
// ex43 -m ../data/ref-cube.mesh -r 2
// ex43 -m ../data/fichera.mesh
// ex43 -m ../data/star.mesh
//
// Description: This example code solves a linear elasticity problem using
// Nitsche's method to enforce sliding boundary conditions. In
// particular, we consider a linear elastic body that is displaced
// in the normal direction on the entire boundary, but is free to
// slide in the tangential direction. This is achieved by imposing
// homogeneous Dirichlet boundary conditions on the normal
// component of the displacement, while applying homogeneous
// Neumann boundary conditions on the tangential components of the
// displacement. By enforcing a uniform, constant normal
// displacement on the boundary, we can simulate the effect of
// compressing or expanding the elastic body uniformly. These
// boundary conditions are applied weakly using Nitsche's method,
// allowing for more flexibility in handling complex geometries in
// either 2D or 3D.
//
// The strong form is given by:
//
// Div(σ(u)) = 0 in Ω
// u ⋅ n = g on Γ
// σ(u) ⊥ n on Γ
//
// where σ(u) = λ tr(ε(u)) I + 2μ ε(u) is the stress tensor, ε(u)
// is the strain tensor, λ and μ are the Lamé parameters, and g is
// the prescribed displacement on the boundary. Here, n is the
// outward normal on the boundary Γ = ∂Ω.
//
// The weak form using Nitsche's method is:
//
// Find u ∈ V such that a(u,v) = b(v) for all v ∈ V
//
// where
//
// a(u,v) := ∫_Ω σ(u) : ε(v) dx
// - ∫_Γ (σ(u) n ⋅ n) (v ⋅ n) dS
// - ∫_Γ (σ(v) n ⋅ n) (u ⋅ n) dS
// + κ ∫_Γ h⁻¹ (λ + 2μ) (u ⋅ n) (v ⋅ n) dS,
//
// b(v) := - ∫_Γ σ(v) n ⋅ n g dS
// + κ ∫_Γ h⁻¹ (λ + 2μ) (v ⋅ n) g dS,
//
// with κ > 0 being a penalty parameter. Here, h is a
// characteristic element size on the boundary. The function
// space V is a vector H1-conforming finite element space.
//
// This example can be viewed as an alternative to Example 28.
// Whereas Example 28 imposes sliding boundary conditions using
// the general-purpose constrained system solvers found in
// mfem/linalg/constraints.hpp, this example employs Nitsche's
// method to weakly enforce the same condition by modifying the
// underlying variational formulation. Unlike Example 28, the
// approach here is specialized to isotropic linear elasticity,
// but it has the advantage of producing a well-conditioned SPD
// stiffness matrix that can be readily preconditioned with
// standard AMG. We recommend reviewing Example 2 before working
// through this example.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
int main(int argc, char *argv[])
{
// 1. Parse command-line options.
const char *mesh_file = "../data/star.mesh";
real_t displ_mag = 0.1;
int order = 1;
int ref_levels = 0;
real_t lambda = 1.0;
real_t mu = 1.0;
real_t kappa = -1.0;
bool static_cond = false;
bool visualization = 1;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&displ_mag, "-g", "--displ",
"Magnitude of the normal displacement.");
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree).");
args.AddOption(&ref_levels, "-r", "--ref_levels",
"Number of uniform mesh refinements.");
args.AddOption(&lambda, "-l", "--lambda", "First Lamé parameter.");
args.AddOption(&mu, "-mu", "--mu", "Second Lamé parameter.");
args.AddOption(&kappa, "-k", "--kappa",
"The penalty parameter, should be positive."
" Negative values are replaced with (order+1)^2.");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.Parse();
if (!args.Good())
{
args.PrintUsage(cout);
return 1;
}
if (kappa < 0)
{
kappa = (order+1)*(order+1);
}
args.PrintOptions(cout);
// 2. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral or hexahedral elements with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 3. Select the order of the finite element discretization space. For NURBS
// meshes, we increase the order by degree elevation.
if (mesh->NURBSext)
{
mesh->DegreeElevate(order, order);
}
// 4. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement.
for (int i = 0; i < ref_levels; i++)
{
mesh->UniformRefinement();
}
// 5. Interpolate the geometry after refinement to control geometry error.
int curvature_order = max(order, 2);
mesh->SetCurvature(curvature_order);
// 6. Define a finite element space on the mesh. Here we use vector finite
// elements, i.e. dim copies of a scalar finite element space. The vector
// dimension is specified by the last argument of the FiniteElementSpace
// constructor. For NURBS meshes, we use the (degree elevated) NURBS space
// associated with the mesh nodes.
FiniteElementCollection *fec;
FiniteElementSpace *fespace;
if (mesh->NURBSext)
{
fec = NULL;
fespace = mesh->GetNodes()->FESpace();
}
else
{
fec = new H1_FECollection(order, dim);
fespace = new FiniteElementSpace(mesh, fec, dim);
}
cout << "Number of finite element unknowns: " << fespace->GetTrueVSize()
<< endl << "Assembling: " << flush;
// 7. Mark the boundary attributes where the sliding (Nitsche) boundary
// conditions are to be applied. These b.c. are imposed weakly, by adding
// the appropriate boundary integrators over the marked 'ess_bdr' to the
// bilinear and linear forms. Thus, no dofs are eliminated; there are no
// essential boundary conditions.
Array<int> ess_tdof_list, ess_bdr(mesh->bdr_attributes.Max());
ess_bdr = 1;
// 8. Define the solution vector x as a finite element grid function
// corresponding to fespace. Initialize x with initial guess of zero,
// which satisfies the boundary conditions.
GridFunction x(fespace);
x = 0.0;
// 9. Set up the bilinear form a(.,.) on the finite element space
// corresponding to the linear elasticity integrator with constant
// coefficients lambda and mu.
ConstantCoefficient lambda_c(lambda);
ConstantCoefficient mu_c(mu);
BilinearForm *a = new BilinearForm(fespace);
a->AddDomainIntegrator(new ElasticityIntegrator(lambda_c,mu_c));
a->AddBdrFaceIntegrator(
new SlidingElasticityIntegrator(lambda_c, mu_c, kappa),
ess_bdr);
// 10. Set up the linear form b(.) corresponding to the Nitsche method
// to impose the Dirichlet boundary conditions. Here, we set the
// prescribed displacement on the Dirichlet boundary to be a constant
// normal displacement of magnitude 'displ_mag'.
ConstantCoefficient g(displ_mag);
LinearForm *b = new LinearForm(fespace);
b->AddBdrFaceIntegrator(
new SlidingElasticityLFIntegrator(
g, lambda_c, mu_c, kappa), ess_bdr);
b->Assemble();
// 11. Assemble the bilinear form and the corresponding linear system,
// applying any necessary transformations such as: eliminating boundary
// conditions, applying conforming constraints for non-conforming AMR,
// static condensation, etc.
cout << "matrix ... " << flush;
if (static_cond) { a->EnableStaticCondensation(); }
a->Assemble();
SparseMatrix A;
Vector B, X;
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
cout << "done." << endl;
cout << "Size of linear system: " << A.Height() << endl;
#ifndef MFEM_USE_SUITESPARSE
// 12. Define a simple symmetric Gauss-Seidel preconditioner and use it to
// solve the system Ax=b with PCG.
GSSmoother M(A);
PCG(A, M, B, X, 1, 500, 1e-12, 0.0);
#else
// 12. If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
UMFPackSolver umf_solver;
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
umf_solver.SetOperator(A);
umf_solver.Mult(B, X);
#endif
// 13. Recover the solution as a finite element grid function.
a->RecoverFEMSolution(X, *b, x);
// 14. For non-NURBS meshes, make the mesh curved based on the finite element
// space. This means that we define the mesh elements through a fespace
// based transformation of the reference element. This allows us to save
// the displaced mesh as a curved mesh when using high-order finite
// element displacement field. We assume that the initial mesh (read from
// the file) is not higher order curved mesh compared to the chosen FE
// space.
if (!mesh->NURBSext)
{
mesh->SetNodalFESpace(fespace);
}
// 15. Save the displaced mesh and the inverted solution (which gives the
// backward displacements to the original grid). This output can be
// viewed later using GLVis: "glvis -m displaced.mesh -g sol.gf".
{
GridFunction *nodes = mesh->GetNodes();
*nodes += x;
x *= -1;
ofstream mesh_ofs("displaced.mesh");
mesh_ofs.precision(8);
mesh->Print(mesh_ofs);
ofstream sol_ofs("sol.gf");
sol_ofs.precision(8);
x.Save(sol_ofs);
}
// 16. Send the above data by socket to a GLVis server. Use the "n" and "b"
// keys in GLVis to visualize the displacements.
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "solution\n" << *mesh << x << flush;
}
// 17. Free the used memory.
delete a;
delete b;
if (fec)
{
delete fespace;
delete fec;
}
delete mesh;
return 0;
}
+332
View File
@@ -0,0 +1,332 @@
// MFEM Example 43 - Parallel Version
//
// Compile with: make ex43p
//
// Sample runs: mpirun -np 4 ex43p -m ../data/ball-nurbs.mesh -r 2
// mpirun -np 4 ex43p -m ../data/ref-cube.mesh -r 2
// mpirun -np 4 ex43p -m ../data/fichera.mesh
// mpirun -np 4 ex43p -m ../data/star.mesh
//
// Description: This example code solves a linear elasticity problem using
// Nitsche's method to enforce sliding boundary conditions. In
// particular, we consider a linear elastic body that is displaced
// in the normal direction on the entire boundary, but is free to
// slide in the tangential direction. This is achieved by imposing
// homogeneous Dirichlet boundary conditions on the normal
// component of the displacement, while applying homogeneous
// Neumann boundary conditions on the tangential components of the
// displacement. By enforcing a uniform, constant normal
// displacement on the boundary, we can simulate the effect of
// compressing or expanding the elastic body uniformly. These
// boundary conditions are applied weakly using Nitsche's method,
// allowing for more flexibility in handling complex geometries in
// either 2D or 3D.
//
// The strong form is given by:
//
// Div(σ(u)) = 0 in Ω
// u ⋅ n = g on Γ
// σ(u) ⊥ n on Γ
//
// where σ(u) = λ tr(ε(u)) I + 2μ ε(u) is the stress tensor, ε(u)
// is the strain tensor, λ and μ are the Lamé parameters, and g is
// the prescribed displacement on the boundary. Here, n is the
// outward normal on the boundary Γ = ∂Ω.
//
// The weak form using Nitsche's method is:
//
// Find u ∈ V such that a(u,v) = b(v) for all v ∈ V
//
// where
//
// a(u,v) := ∫_Ω σ(u) : ε(v) dx
// - ∫_Γ (σ(u) n ⋅ n) (v ⋅ n) dS
// - ∫_Γ (σ(v) n ⋅ n) (u ⋅ n) dS
// + κ ∫_Γ h⁻¹ (λ + 2μ) (u ⋅ n) (v ⋅ n) dS,
//
// b(v) := - ∫_Γ σ(v) n ⋅ n g dS
// + κ ∫_Γ h⁻¹ (λ + 2μ) (v ⋅ n) g dS,
//
// with κ > 0 being a penalty parameter. Here, h is a
// characteristic element size on the boundary. The function
// space V is a vector H1-conforming finite element space.
//
// This example can be viewed as an alternative to Example 28.
// Whereas Example 28 imposes sliding boundary conditions using
// the general-purpose constrained system solvers found in
// mfem/linalg/constraints.hpp, this example employs Nitsche's
// method to weakly enforce the same condition by modifying the
// underlying variational formulation. Unlike Example 28, the
// approach here is specialized to isotropic linear elasticity,
// but it has the advantage of producing a well-conditioned SPD
// stiffness matrix that can be readily preconditioned with
// standard AMG. We recommend reviewing Example 2 before working
// through this example.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
int main(int argc, char *argv[])
{
// 1. Initialize MPI and HYPRE.
Mpi::Init(argc, argv);
int num_procs = Mpi::WorldSize();
int myid = Mpi::WorldRank();
Hypre::Init();
// 2. Parse command-line options.
const char *mesh_file = "../data/star.mesh";
real_t displ_mag = 0.1;
int order = 1;
int ref_levels = 0;
real_t lambda = 1.0;
real_t mu = 1.0;
real_t kappa = -1.0;
bool static_cond = false;
bool reorder_space = false;
bool visualization = 1;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&displ_mag, "-g", "--displ",
"Magnitude of the normal displacement.");
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree).");
args.AddOption(&ref_levels, "-r", "--ref_levels",
"Number of uniform mesh refinements.");
args.AddOption(&lambda, "-l", "--lambda", "First Lamé parameter.");
args.AddOption(&mu, "-mu", "--mu", "Second Lamé parameter.");
args.AddOption(&kappa, "-k", "--kappa",
"The penalty parameter, should be positive."
" Negative values are replaced with (order+1)^2.");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&reorder_space, "-nodes", "--by-nodes", "-vdim", "--by-vdim",
"Use byNODES ordering of vector space instead of byVDIM");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.Parse();
if (!args.Good())
{
if (myid == 0)
{
args.PrintUsage(cout);
}
return 1;
}
if (kappa < 0)
{
kappa = (order+1)*(order+1);
}
if (myid == 0)
{
args.PrintOptions(cout);
}
// 3. Read the (serial) mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral or hexahedral elements with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Select the order of the finite element discretization space. For NURBS
// meshes, we increase the order by degree elevation.
if (mesh->NURBSext)
{
mesh->DegreeElevate(order, order);
}
// 5. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement.
for (int i = 0; i < ref_levels; i++)
{
mesh->UniformRefinement();
}
// 6. Interpolate the geometry after refinement to control geometry error.
int curvature_order = max(order, 2);
mesh->SetCurvature(curvature_order);
// 7. Define a parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
// 8. Define a finite element space on the mesh. Here we use vector finite
// elements, i.e. dim copies of a scalar finite element space. The vector
// dimension is specified by the last argument of the FiniteElementSpace
// constructor. For NURBS meshes, we use the (degree elevated) NURBS space
// associated with the mesh nodes.
FiniteElementCollection *fec;
ParFiniteElementSpace *fespace;
const bool use_nodal_fespace = pmesh->NURBSext;
if (use_nodal_fespace)
{
fec = NULL;
fespace = (ParFiniteElementSpace *)pmesh->GetNodes()->FESpace();
}
else
{
fec = new H1_FECollection(order, dim);
if (reorder_space)
{
fespace = new ParFiniteElementSpace(pmesh, fec, dim, Ordering::byNODES);
}
else
{
fespace = new ParFiniteElementSpace(pmesh, fec, dim, Ordering::byVDIM);
}
}
HYPRE_BigInt size = fespace->GlobalTrueVSize();
if (myid == 0)
{
cout << "Number of finite element unknowns: " << size << endl
<< "Assembling: " << flush;
}
// 9. Mark the boundary attributes where the sliding (Nitsche) boundary
// conditions are to be applied. These b.c. are imposed weakly, by adding
// the appropriate boundary integrators over the marked 'ess_bdr' to the
// bilinear and linear forms. Thus, no dofs are eliminated; there are no
// essential boundary conditions.
Array<int> ess_tdof_list, ess_bdr;
if (pmesh->bdr_attributes.Size())
{
ess_bdr.SetSize(pmesh->bdr_attributes.Max());
ess_bdr = 1;
}
// 10. Define the solution vector x as a finite element grid function
// corresponding to fespace. Initialize x with initial guess of zero,
// which satisfies the boundary conditions.
ParGridFunction x(fespace);
x = 0.0;
// 11. Set up the bilinear form a(.,.) on the finite element space
// corresponding to the linear elasticity integrator with constant
// coefficients lambda and mu.
ConstantCoefficient lambda_c(lambda);
ConstantCoefficient mu_c(mu);
ParBilinearForm *a = new ParBilinearForm(fespace);
a->AddDomainIntegrator(new ElasticityIntegrator(lambda_c,mu_c));
a->AddBdrFaceIntegrator(
new SlidingElasticityIntegrator(lambda_c, mu_c, kappa),
ess_bdr);
// 12. Set up the linear form b(.) corresponding to the Nitsche method
// to impose the Dirichlet boundary conditions. Here, we set the
// prescribed displacement on the Dirichlet boundary to be a constant
// normal displacement of magnitude 'displ_mag'.
ConstantCoefficient g(displ_mag);
ParLinearForm *b = new ParLinearForm(fespace);
b->AddBdrFaceIntegrator(
new SlidingElasticityLFIntegrator(
g, lambda_c, mu_c, kappa), ess_bdr);
b->Assemble();
// 13. Assemble the parallel bilinear form and the corresponding linear
// system, applying any necessary transformations such as: parallel
// assembly, eliminating boundary conditions, applying conforming
// constraints for non-conforming AMR, static condensation, etc.
if (myid == 0) { cout << "matrix ... " << flush; }
if (static_cond) { a->EnableStaticCondensation(); }
a->Assemble();
HypreParMatrix A;
Vector B, X;
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
if (myid == 0)
{
cout << "done." << endl;
cout << "Size of linear system: " << A.GetGlobalNumRows() << endl;
}
// 14. Define and apply a parallel PCG solver for A X = B with the BoomerAMG
// preconditioner from hypre.
HypreBoomerAMG *amg = new HypreBoomerAMG(A);
if (!a->StaticCondensationIsEnabled())
{
amg->SetElasticityOptions(fespace);
}
else
{
amg->SetSystemsOptions(dim, reorder_space);
}
HyprePCG *pcg = new HyprePCG(A);
pcg->SetTol(1e-8);
pcg->SetMaxIter(500);
pcg->SetPrintLevel(2);
pcg->SetPreconditioner(*amg);
pcg->Mult(B, X);
// 15. Recover the parallel grid function corresponding to X. This is the
// local finite element solution on each processor.
a->RecoverFEMSolution(X, *b, x);
// 16. For non-NURBS meshes, make the mesh curved based on the finite element
// space. This means that we define the mesh elements through a fespace
// based transformation of the reference element. This allows us to save
// the displaced mesh as a curved mesh when using high-order finite
// element displacement field. We assume that the initial mesh (read from
// the file) is not higher order curved mesh compared to the chosen FE
// space.
if (!use_nodal_fespace)
{
pmesh->SetNodalFESpace(fespace);
}
// 17. Save in parallel the displaced mesh and the inverted solution (which
// gives the backward displacements to the original grid). This output
// can be viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
{
GridFunction *nodes = pmesh->GetNodes();
*nodes += x;
x *= -1;
ostringstream mesh_name, sol_name;
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
sol_name << "sol." << setfill('0') << setw(6) << myid;
ofstream mesh_ofs(mesh_name.str().c_str());
mesh_ofs.precision(8);
pmesh->Print(mesh_ofs);
ofstream sol_ofs(sol_name.str().c_str());
sol_ofs.precision(8);
x.Save(sol_ofs);
}
// 18. Send the above data by socket to a GLVis server. Use the "n" and "b"
// keys in GLVis to visualize the displacements.
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock.precision(8);
sol_sock << "solution\n" << *pmesh << x << flush;
}
// 19. Free the used memory.
delete pcg;
delete amg;
delete a;
delete b;
if (fec)
{
delete fespace;
delete fec;
}
delete pmesh;
return 0;
}
+1 -6
View File
@@ -9,7 +9,6 @@
// mpirun -np 4 ex4p -m ../data/beam-hex.mesh -o 2 -pa
// mpirun -np 4 ex4p -m ../data/escher.mesh -o 2 -sc
// mpirun -np 4 ex4p -m ../data/fichera.mesh -o 2 -hb
// mpirun -np 4 ex4p -m ../data/fichera.mesh -o 2 -hb -ea
// mpirun -np 4 ex4p -m ../data/fichera-q2.vtk
// mpirun -np 4 ex4p -m ../data/fichera-q3.mesh -o 2 -sc
// mpirun -np 4 ex4p -m ../data/square-disc-nurbs.mesh -o 3
@@ -18,18 +17,14 @@
// mpirun -np 4 ex4p -m ../data/periodic-cube.mesh -no-bc
// mpirun -np 4 ex4p -m ../data/amr-quad.mesh
// mpirun -np 3 ex4p -m ../data/amr-quad.mesh -o 2 -hb
// mpirun -np 3 ex4p -m ../data/amr-quad.mesh -o 2 -hb -ea
// mpirun -np 4 ex4p -m ../data/amr-hex.mesh -o 2 -sc
// mpirun -np 4 ex4p -m ../data/amr-hex.mesh -o 2 -hb
// mpirun -np 4 ex4p -m ../data/amr-hex.mesh -o 2 -hb -ea
// mpirun -np 4 ex4p -m ../data/ref-prism.mesh -o 1
// mpirun -np 4 ex4p -m ../data/octahedron.mesh -o 1
// mpirun -np 4 ex4p -m ../data/star-surf.mesh -o 3 -hb
//
// Device sample runs:
// mpirun -np 4 ex4p -m ../data/star.mesh -pa -d cuda
// mpirun -np 4 ex4p -m ../data/star.mesh -ea -hb -d cuda
// mpirun -np 4 ex4p -m ../data/amr-hex.mesh -ea -hb -d cuda
// mpirun -np 4 ex4p -m ../data/star.mesh -pa -d raja-cuda
// mpirun -np 4 ex4p -m ../data/star.mesh -pa -d raja-omp
// mpirun -np 4 ex4p -m ../data/beam-hex.mesh -pa -d cuda
@@ -235,7 +230,7 @@ int main(int argc, char *argv[])
pcg->SetMaxIter(2000);
pcg->SetPrintLevel(1);
if (hybridization) { prec = new HypreBoomerAMG(*A.As<HypreParMatrix>()); }
else if (pa || ea) { prec = new OperatorJacobiSmoother(*a, ess_tdof_list); }
else if (pa) { prec = new OperatorJacobiSmoother(*a, ess_tdof_list); }
else
{
ParFiniteElementSpace *prec_fespace =
+2 -2
View File
@@ -22,11 +22,11 @@ MFEM_LIB_FILE = mfem_is_not_built
SEQ_EXAMPLES = ex0 ex1 ex2 ex3 ex4 ex5 ex6 ex7 ex8 ex9 ex10 ex14 ex15 ex16 \
ex17 ex18 ex19 ex20 ex21 ex22 ex23 ex24 ex25 ex26 ex27 ex28 ex29 ex30 \
ex31 ex33 ex34 ex36 ex37 ex38 ex39 ex40 ex41
ex31 ex33 ex34 ex36 ex37 ex38 ex39 ex40 ex41 ex43
PAR_EXAMPLES = ex0p ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex8p ex9p ex10p ex11p \
ex12p ex13p ex14p ex15p ex16p ex17p ex18p ex19p ex20p ex21p ex22p ex24p \
ex25p ex26p ex27p ex28p ex29p ex30p ex31p ex32p ex33p ex34p ex35p ex36p \
ex37p ex39p ex40p ex41p
ex37p ex39p ex40p ex41p ex43p
SEQ_DEVICE_EXAMPLES = ex1 ex3 ex4 ex5 ex6 ex9 ex14 ex22 ex24 ex25 ex26 ex34
PAR_DEVICE_EXAMPLES = ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex9p ex13p ex14p \
ex22p ex24p ex25p ex26p ex34p ex35p
+6 -35
View File
@@ -825,46 +825,14 @@ void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
Vector &b, OperatorHandle &A, Vector &X,
Vector &B, int copy_interior)
{
const SparseMatrix *P = fes->GetConformingProlongation();
const SparseMatrix *R = fes->GetConformingRestriction();
if (ext)
{
if (hybridization)
{
FormSystemMatrix(ess_tdof_list, A);
std::unique_ptr<ConstrainedOperator> A_constrained([&]()
{
Operator *op;
Operator::FormSystemOperator(ess_tdof_list, op);
return dynamic_cast<ConstrainedOperator*>(op);
}());
MFEM_ASSERT(A_constrained != nullptr, "");
Vector conf_b, conf_x;
if (P)
{
// Nonconforming
conf_b.SetSize(P->Width());
conf_x.SetSize(P->Width());
P->MultTranspose(b, conf_b);
R->Mult(x, conf_x);
}
else
{
// Conforming
conf_b.MakeRef(b, 0, b.Size());
conf_x.MakeRef(x, 0, x.Size());
}
A_constrained->EliminateRHS(conf_x, conf_b);
if (P)
{
R->MultTranspose(conf_b, b); // store eliminated rhs in b
}
hybridization->ReduceRHS(conf_b, B);
ConstrainedOperator A_constrained(this, ess_tdof_list);
A_constrained.EliminateRHS(x, b);
hybridization->ReduceRHS(b, B);
X.SetSize(B.Size());
X = 0.0;
}
@@ -874,6 +842,7 @@ void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
}
return;
}
const SparseMatrix *P = fes->GetConformingProlongation();
FormSystemMatrix(ess_tdof_list, A);
// Transform the system and perform the elimination in B, based on the
@@ -909,6 +878,7 @@ void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
if (hybridization)
{
// Reduction to the Lagrange multipliers system
const SparseMatrix *R = fes->GetConformingRestriction();
Vector conf_b(P->Width()), conf_x(P->Width());
P->MultTranspose(b, conf_b);
R->Mult(x, conf_x);
@@ -921,6 +891,7 @@ void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
else
{
// Variational restriction with P
const SparseMatrix *R = fes->GetConformingRestriction();
B.SetSize(P->Width());
P->MultTranspose(b, B);
X.SetSize(R->Height());
+175
View File
@@ -4213,6 +4213,181 @@ void DGElasticityIntegrator::AssembleFaceMatrix(
}
}
void SlidingElasticityIntegrator::AssembleFaceMatrix(
const FiniteElement &el1, const FiniteElement &el2,
FaceElementTransformations &Trans, DenseMatrix &elmat)
{
MFEM_ASSERT(Trans.Elem2No < 0,
"support for interior faces is not implemented");
#ifdef MFEM_THREAD_SAFE
// For descriptions of these variables, see the class declaration.
Vector shape1;
DenseMatrix dshape1;
DenseMatrix adjJ;
DenseMatrix dshape1_ps;
Vector nor;
Vector nL1;
Vector nM1;
Vector nt1;
Vector dshape1_dnM;
Vector dshape1_dnt;
DenseMatrix jmat;
#endif
const int dim = el1.GetDim();
const int ndofs1 = el1.GetDof();
const int nvdofs = dim * ndofs1;
// Initially 'elmat' corresponds to the term:
// < { sigma(u) n . ñ }, v . ñ > =
// < { (lambda div(u) I + mu (grad(u) + grad(u)^T)) n . ñ }, v . ñ >
// But eventually, it's going to be replaced by:
// elmat := -elmat + alpha*elmat^T + jmat
elmat.SetSize(nvdofs);
elmat = 0.;
const bool kappa_is_nonzero = (kappa != 0.0);
if (kappa_is_nonzero)
{
jmat.SetSize(nvdofs);
jmat = 0.;
}
adjJ.SetSize(dim);
shape1.SetSize(ndofs1);
dshape1.SetSize(ndofs1, dim);
dshape1_ps.SetSize(ndofs1, dim);
nor.SetSize(dim);
nL1.SetSize(dim);
nM1.SetSize(dim);
nt1.SetSize(dim);
dshape1_dnM.SetSize(ndofs1);
dshape1_dnt.SetSize(ndofs1);
const IntegrationRule *ir = IntRule;
if (ir == NULL)
{
// a simple choice for the integration order; is this OK?
const int order = 2 * el1.GetOrder();
ir = &IntRules.Get(Trans.GetGeometryType(), order);
}
for (int pind = 0; pind < ir->GetNPoints(); ++pind)
{
const IntegrationPoint &ip = ir->IntPoint(pind);
// Set the integration point in the face and the neighboring elements
Trans.SetAllIntPoints(&ip);
// Access the neighboring element's integration point
const IntegrationPoint &eip1 = Trans.GetElement1IntPoint();
el1.CalcShape(eip1, shape1);
el1.CalcDShape(eip1, dshape1);
CalcAdjugate(Trans.Elem1->Jacobian(), adjJ);
Mult(dshape1, adjJ, dshape1_ps);
if (dim == 1)
{
nor(0) = 2*eip1.x - 1.0;
}
else
{
CalcOrtho(Trans.Jacobian(), nor);
}
if (!nt)
{
// Set ñ to the unit normal vector if not provided
nt1 = nor;
nt1 /= nt1.Norml2();
}
else
{
// Evaluate vector function ñ at integration point
nt->Eval(nt1, *Trans.Elem1, eip1);
}
const real_t W = ip.weight;
const real_t W1 = W / Trans.Elem1->Weight();
const real_t WL1 = W1 * lambda->Eval(*Trans.Elem1, eip1);
const real_t WM1 = W1 * mu->Eval(*Trans.Elem1, eip1);
nL1.Set(WL1, nor);
nM1.Set(WM1, nor);
const real_t WLM = WL1 + 2.0*WM1;
dshape1_ps.Mult(nM1, dshape1_dnM);
dshape1_ps.Mult(nt1, dshape1_dnt);
const real_t jmatcoef = kappa * (nor*nor) * WLM;
const real_t nL_dot_nt1 = nL1 * nt1;
for (int jm = 0, j = 0; jm < dim; ++jm)
{
for (int jdof = 0; jdof < ndofs1; ++jdof, ++j)
{
const real_t t1 = dshape1_ps(jdof, jm) * nL_dot_nt1;
const real_t t2 = dshape1_dnM(jdof) * nt1(jm);
const real_t t3 = dshape1_dnt(jdof) * nM1(jm);
const real_t tt = t1 + t2 + t3;
for (int im = 0, i = 0; im < dim; ++im)
{
for (int idof = 0; idof < ndofs1; ++idof, ++i)
{
elmat(i, j) += tt * shape1(idof) * nt1(im);
}
}
}
}
if (kappa_is_nonzero)
{
for (int jm = 0, j = 0; jm < dim; ++jm)
{
for (int jdof = 0; jdof < ndofs1; ++jdof, ++j)
{
const real_t sj = jmatcoef * shape1(jdof) * nt1(jm);
for (int im = 0, i = 0; im < dim; ++im)
{
for (int idof = 0; idof < ndofs1; ++idof, ++i)
{
jmat(i, j) += shape1(idof) * sj * nt1(im);
}
}
}
}
}
}
// elmat := -elmat + alpha*elmat^t + jmat
if (kappa_is_nonzero)
{
for (int i = 0; i < nvdofs; ++i)
{
for (int j = 0; j < i; ++j)
{
real_t aij = elmat(i,j), aji = elmat(j,i), mij = jmat(i,j);
elmat(i,j) = alpha*aji - aij + mij;
elmat(j,i) = alpha*aij - aji + mij;
}
elmat(i,i) = (alpha - 1.)*elmat(i,i) + jmat(i,i);
}
}
else
{
for (int i = 0; i < nvdofs; ++i)
{
for (int j = 0; j < i; ++j)
{
real_t aij = elmat(i,j), aji = elmat(j,i);
elmat(i,j) = alpha*aji - aij;
elmat(j,i) = alpha*aij - aji;
}
elmat(i,i) *= (alpha - 1.);
}
}
}
void TraceJumpIntegrator::AssembleFaceMatrix(
const FiniteElement &trial_face_fe, const FiniteElement &test_fe1,
+78
View File
@@ -3738,6 +3738,84 @@ protected:
DenseMatrix &elmat, DenseMatrix &jmat);
};
/** Integrator for the Nitsche elasticity form:
$$
\begin{split}
a(u,v)
&:= -\langle \sigma(u)\, \vec{n} \cdot \tilde{n},\ v \cdot \tilde{n}
\rangle + \alpha \langle \sigma(v)\, \vec{n} \cdot \tilde{n},\ u \cdot
\tilde{n} \rangle + \kappa \langle h^{-1} (\lambda + 2\mu)\, u \cdot
\tilde{n},\ v \cdot \tilde{n} \rangle \\
&= -\int_\Gamma (\sigma(u)\, n \cdot \tilde{n})(v \cdot \tilde{n})\, dS +
\alpha \int_\Gamma (\sigma(v)\, n \cdot \tilde{n})(u \cdot \tilde{n})\,
dS + \kappa \int_\Gamma h^{-1} (\lambda + 2\mu)(u \cdot \tilde{n})(v
\cdot \tilde{n})\, dS.
\end{split}
$$
For isotropic media,
$$
\begin{split}
\sigma(u) &= \lambda \nabla \cdot u I + 2 \mu \varepsilon(u) \\
&= \lambda \nabla \cdot u I + 2 \mu \frac{1}{2} (\nabla u + \nabla
u^{\mathrm{T}}) \\
&= \lambda \nabla \cdot u I + \mu (\nabla u + \nabla u^{\mathrm{T}})
\end{split}
$$
where $I$ is the identity matrix, $\lambda$ and $\mu$ are the Lamé
coefficients (see ElasticityIntegrator), $\tilde{n}$ is a unit vector
field, $\alpha = \pm 1$ and $\kappa > 0$ are the Nitsche parameters, and
$u$, $v$ are the trial and test functions, respectively.
This is a '%Vector' integrator, i.e. defined for FE spaces using multiple
copies of a scalar FE space.
*/
class SlidingElasticityIntegrator : public BilinearFormIntegrator
{
public:
SlidingElasticityIntegrator(Coefficient &lambda_, Coefficient &mu_,
real_t kappa_)
: nt(NULL), lambda(&lambda_), mu(&mu_), alpha(-1.0), kappa(kappa_) { }
SlidingElasticityIntegrator(VectorCoefficient &nt_, Coefficient &lambda_,
Coefficient &mu_, real_t alpha_, real_t kappa_)
: nt(&nt_), lambda(&lambda_), mu(&mu_), alpha(alpha_), kappa(kappa_) { }
using BilinearFormIntegrator::AssembleFaceMatrix;
void AssembleFaceMatrix(const FiniteElement &el1,
const FiniteElement &el2,
FaceElementTransformations &Trans,
DenseMatrix &elmat) override;
protected:
VectorCoefficient *nt;
Coefficient *lambda, *mu;
real_t alpha, kappa;
#ifndef MFEM_THREAD_SAFE
// values of all scalar basis functions for one component of u (which is a
// vector) at the integration point in the reference space
Vector shape1;
// values of derivatives of all scalar basis functions for one component
// of u (which is a vector) at the integration point in the reference space
DenseMatrix dshape1;
// Adjugate of the Jacobian of the transformation: adjJ = det(J) J^{-1}
DenseMatrix adjJ;
// gradient of shape functions in the real (physical, not reference)
// coordinates, scaled by det(J):
// dshape_ps(jdof,jm) = sum_{t} adjJ(t,jm)*dshape(jdof,t)
DenseMatrix dshape1_ps;
Vector nor; // nor = |weight(J_face)| n
Vector nL1; // nL1 = (lambda1 * ip.weight / detJ1) nor
Vector nM1; // nM1 = (mu1 * ip.weight / detJ1) nor
Vector nt1; // nt1 = vector function ñ evaluated at ip1
Vector dshape1_dnM; // dshape1_dnM = dshape1_ps . nM1
Vector dshape1_dnt; // dshape1_dnt = dshape1_ps . nt1
// 'jmat' corresponds to the term: kappa <h⁻¹ u ⋅ ñ, v ⋅ ñ>
DenseMatrix jmat;
#endif
};
/** Integrator for the DPG form:$ \langle v, [w] \rangle $ over all faces (the interface) where
the trial variable $v$ is defined on the interface and the test variable $w$ is
defined inside the elements, generally in a DG space. */
+1 -1
View File
@@ -387,7 +387,7 @@ void DGMassInverse::DGMassCGIteration(const Vector &b_, Vector &u_) const
static constexpr int NB = Q1D ? Q1D : 1; // block size
mfem::forall_2D<NB*NB>(NE, NB, NB, [=] MFEM_HOST_DEVICE (int e)
mfem::forall_2D(NE, NB, NB, [=] MFEM_HOST_DEVICE (int e)
{
// Perform change of basis if needed
if (CHANGE_BASIS)
+3 -3
View File
@@ -69,9 +69,9 @@ inline int ToLexOrdering2D(const int face_id, const int size1d, const int i)
}
/// @brief Given a face DOF index on a shared face, ordered lexicographically
/// relative to the element (where the local face is face_id), return the
/// corresponding face DOF index ordered lexicographically relative to the face
/// itself.
/// relative to element the element (where the local face is face_id), and
/// return the corresponding face DOF index ordered lexicographically relative
/// to the face itself.
MFEM_HOST_DEVICE
inline int PermuteFace2D(const int face_id, const int orientation,
const int size1d, const int index)
+34 -22
View File
@@ -231,7 +231,7 @@ void FiniteElement::CalcPhysLaplacian(ElementTransformation &Trans,
{
for (int nd = 0; nd < dof; nd++)
{
Laplacian[nd] = hess(nd,0) + hess(nd,3) + hess(nd,5);
Laplacian[nd] = hess(nd,0) + hess(nd,4) + hess(nd,5);
}
}
else if (dim == 2)
@@ -268,9 +268,11 @@ void FiniteElement::CalcPhysLinLaplacian(ElementTransformation &Trans,
scale[0] = Gij(0,0);
scale[1] = 2*Gij(0,1);
scale[2] = 2*Gij(0,2);
scale[3] = Gij(1,1);
scale[4] = 2*Gij(1,2);
scale[5] = Gij(2,2);
scale[3] = 2*Gij(1,2);
scale[4] = Gij(2,2);
scale[5] = Gij(1,1);
}
else if (dim == 2)
{
@@ -307,12 +309,12 @@ void FiniteElement::CalcPhysHessian(ElementTransformation &Trans,
map[2] = 2;
map[3] = 1;
map[4] = 3;
map[5] = 4;
map[4] = 5;
map[5] = 3;
map[6] = 2;
map[7] = 4;
map[8] = 5;
map[7] = 3;
map[8] = 4;
}
else if (dim == 2)
{
@@ -380,7 +382,11 @@ const DofToQuad &FiniteElement::GetDofToQuad(const IntegrationRule &ir,
#pragma omp critical (DofToQuad)
#endif
{
d2q = DofToQuad::SearchArray(dof2quad_array, ir, mode);
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
}
if (!d2q)
{
#ifdef MFEM_THREAD_SAFE
@@ -655,22 +661,14 @@ void ScalarFiniteElement::ScalarLocalL2Restriction(
void NodalFiniteElement::CreateLexicographicFullMap(const IntegrationRule &ir)
const
{
// Get the FULL version of the map. This call contains omp critical region,
// so it is done before the critical region below.
auto &d2q = GetDofToQuad(ir, DofToQuad::FULL);
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
{
// If the new Dof2Quad is already present, e.g. added in a previous call
// or added by another omp thread, return.
if (DofToQuad::SearchArray(dof2quad_array, ir,
DofToQuad::LEXICOGRAPHIC_FULL))
{ return; }
// Undo the native ordering which is what FiniteElement::GetDofToQuad
// returns.
// Get the FULL version of the map.
auto &d2q = GetDofToQuad(ir, DofToQuad::FULL);
//Undo the native ordering which is what FiniteElement::GetDofToQuad returns.
auto *d2q_new = new DofToQuad(d2q);
d2q_new->mode = DofToQuad::LEXICOGRAPHIC_FULL;
const int nqpt = ir.GetNPoints();
@@ -726,7 +724,13 @@ const DofToQuad &NodalFiniteElement::GetDofToQuad(const IntegrationRule &ir,
#pragma omp critical (DofToQuad)
#endif
{
d2q = DofToQuad::SearchArray(dof2quad_array, ir, mode);
//Should make this loop a function of FiniteElement
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule == &ir && d2q->mode == mode) { break; }
d2q = nullptr;
}
}
if (d2q) { return *d2q; }
if (mode != DofToQuad::LEXICOGRAPHIC_FULL)
@@ -2627,7 +2631,15 @@ const DofToQuad &TensorBasisElement::GetTensorDofToQuad(
#pragma omp critical (DofToQuad)
#endif
{
d2q = DofToQuad::SearchArray(dof2quad_array, ir, mode);
for (int i = 0; i < dof2quad_array.Size(); i++)
{
auto* d2q_ = dof2quad_array[i];
if (d2q_->IntRule == &ir && d2q_->mode == mode)
{
d2q = d2q_;
break;
}
}
if (!d2q)
{
d2q = new DofToQuad;
-22
View File
@@ -222,12 +222,6 @@ public:
/// Returns absolute value of the maps
DofToQuad Abs() const;
/// Auxiliary function for searching DofToQuad arrays.
static inline DofToQuad *SearchArray(
const Array<DofToQuad*> &dof2quad_array,
const IntegrationRule &ir,
DofToQuad::Mode mode);
};
/// Describes the function space on each element
@@ -413,7 +407,6 @@ public:
/** Each row of the result DenseMatrix @a Hessian contains upper triangular
part of the Hessian of one shape function.
The order in 2D is {u_xx, u_xy, u_yy}.
The order in 3D is {u_xx, u_xy, u_xz, u_yy, u_yz, u_zz}.
The size (#dof x (#dim (#dim+1)/2) of @a Hessian must be set in advance.*/
virtual void CalcHessian(const IntegrationPoint &ip,
DenseMatrix &Hessian) const;
@@ -1383,21 +1376,6 @@ public:
void InvertLinearTrans(ElementTransformation &trans,
const IntegrationPoint &pt, Vector &x);
// static inline method
inline DofToQuad *DofToQuad::SearchArray(
const Array<DofToQuad*> &dof2quad_array,
const IntegrationRule &ir,
DofToQuad::Mode mode)
{
for (int i = 0; i < dof2quad_array.Size(); i++)
{
DofToQuad *d2q = dof2quad_array[i];
if (d2q->IntRule == &ir && d2q->mode == mode) { return d2q; }
}
return nullptr;
}
} // namespace mfem
#endif
-48
View File
@@ -60,12 +60,6 @@ void Linear1DFiniteElement::CalcDShape(const IntegrationPoint &ip,
dshape(1,0) = 1.;
}
void Linear1DFiniteElement::CalcHessian(const IntegrationPoint &ip,
DenseMatrix &h) const
{
h = 0.0;
}
Linear2DFiniteElement::Linear2DFiniteElement()
: NodalFiniteElement(2, Geometry::TRIANGLE, 3, 1)
{
@@ -93,11 +87,6 @@ void Linear2DFiniteElement::CalcDShape(const IntegrationPoint &ip,
dshape(2,0) = 0.; dshape(2,1) = 1.;
}
void Linear2DFiniteElement::CalcHessian(const IntegrationPoint &ip,
DenseMatrix &h) const
{
h = 0.0;
}
BiLinear2DFiniteElement::BiLinear2DFiniteElement()
: NodalFiniteElement(2, Geometry::SQUARE, 4, 1, FunctionSpace::Qk)
@@ -1267,12 +1256,6 @@ void Linear3DFiniteElement::CalcDShape(const IntegrationPoint &ip,
}
}
void Linear3DFiniteElement::CalcHessian(const IntegrationPoint &ip,
DenseMatrix &h) const
{
h = 0.0;
}
void Linear3DFiniteElement::GetFaceDofs (int face, int **dofs, int *ndofs)
const
{
@@ -1649,37 +1632,6 @@ void TriLinear3DFiniteElement::CalcDShape(const IntegrationPoint &ip,
dshape(7,2) = ox * y;
}
void TriLinear3DFiniteElement::CalcHessian(const IntegrationPoint &ip,
DenseMatrix &h) const
{
real_t x = ip.x, y = ip.y, z = ip.z;
real_t ox = 1.-x, oy = 1.-y, oz = 1.-z;
h(0,0) = 0.; h(0,1) = oz; h(0,2) = oy;
h(0,3) = 0.; h(0,4) = ox; h(0,5) = 0.;
h(1,0) = 0.; h(1,1) = -oz; h(1,2) = -oy;
h(1,3) = 0.; h(1,4) = x; h(1,5) = 0.;
h(2,0) = 0.; h(2,1) = oz; h(2,2) = -y;
h(2,3) = 0.; h(2,4) = -x; h(2,5) = 0.;
h(3,0) = 0.; h(3,1) = -oz; h(3,2) = y;
h(3,3) = 0.; h(3,4) = -ox; h(3,5) = 0.;
h(4,0) = 0.; h(4,1) = z; h(4,2) = -oy;
h(4,3) = 0.; h(4,4) = -ox; h(4,5) = 0.;
h(5,0) = 0.; h(5,1) = -z; h(5,2) = oy;
h(5,3) = 0.; h(5,4) = -x; h(5,5) = 0.;
h(6,0) = 0.; h(6,1) = z; h(6,2) = y;
h(6,3) = 0.; h(6,4) = x; h(6,5) = 0.;
h(7,0) = 0.; h(7,1) = -z; h(7,2) = -y;
h(7,3) = 0.; h(7,4) = ox; h(7,5) = 0.;
}
P0SegmentFiniteElement::P0SegmentFiniteElement(int Ord)
: NodalFiniteElement(1, Geometry::SEGMENT, 1, Ord) // default Ord = 0
+1 -9
View File
@@ -50,8 +50,6 @@ public:
contains the derivative of one shape function */
void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const override;
void CalcHessian(const IntegrationPoint &ip,
DenseMatrix &h) const override;
};
/// A 2D linear element on triangle with nodes at the vertices of the triangle
@@ -72,8 +70,6 @@ public:
so that each row contains the derivatives of one shape function */
void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const override;
void CalcHessian(const IntegrationPoint &ip,
DenseMatrix &h) const override;
void ProjectDelta(int vertex, Vector &dofs) const override
{ dofs = 0.0; dofs(vertex) = 1.0; }
};
@@ -408,9 +404,6 @@ public:
void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const override;
void CalcHessian(const IntegrationPoint &ip,
DenseMatrix &h) const override;
void ProjectDelta(int vertex, Vector &dofs) const override
{ dofs = 0.0; dofs(vertex) = 1.0; }
@@ -452,8 +445,7 @@ public:
so that each row contains the derivatives of one shape function */
void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const override;
void CalcHessian(const IntegrationPoint &ip,
DenseMatrix &h) const override;
void ProjectDelta(int vertex, Vector &dofs) const override
{ dofs = 0.0; dofs(vertex) = 1.0; }
};
+4 -3
View File
@@ -445,10 +445,11 @@ void NURBS3DFiniteElement::CalcHessian (const IntegrationPoint &ip,
d2sum[0] += ( hessian(o,0) = d2sx*sy*sz*weights(o) );
d2sum[1] += ( hessian(o,1) = dsx*dsy*sz*weights(o) );
d2sum[2] += ( hessian(o,2) = dsx*sy*dsz*weights(o) );
d2sum[3] += ( hessian(o,3) = sx*d2sy*sz*weights(o) );
d2sum[4] += ( hessian(o,4) = sx*dsy*dsz*weights(o) );
d2sum[5] += ( hessian(o,5) = sx*sy*d2sz*weights(o) );
d2sum[3] += ( hessian(o,3) = sx*dsy*dsz*weights(o) );
d2sum[4] += ( hessian(o,4) = sx*sy*d2sz*weights(o) );
d2sum[5] += ( hessian(o,5) = sx*d2sy*sz*weights(o) );
}
}
}
+13 -50
View File
@@ -1516,76 +1516,36 @@ const FaceRestriction *FiniteElementSpace::GetFaceRestriction(
const bool is_dg_space = IsDGSpace();
const L2FaceValues m = (is_dg_space && mul==L2FaceValues::DoubleValued) ?
L2FaceValues::DoubleValued : L2FaceValues::SingleValued;
auto key = std::make_tuple(is_dg_space, f_ordering, type, m);
key_face key = std::make_tuple(is_dg_space, f_ordering, type, m);
auto itr = L2F.find(key);
if (itr != L2F.end())
{
return itr->second.get();
return itr->second;
}
else
{
std::unique_ptr<FaceRestriction> res;
FaceRestriction *res;
if (is_dg_space)
{
if (Conforming())
{
res.reset(new L2FaceRestriction(*this, f_ordering, type, m));
res = new L2FaceRestriction(*this, f_ordering, type, m);
}
else
{
res.reset(new NCL2FaceRestriction(*this, f_ordering, type, m));
res = new NCL2FaceRestriction(*this, f_ordering, type, m);
}
}
else if (dynamic_cast<const DG_Interface_FECollection*>(fec))
{
res.reset(new L2InterfaceFaceRestriction(*this, f_ordering, type));
res = new L2InterfaceFaceRestriction(*this, f_ordering, type);
}
else
{
res.reset(new ConformingFaceRestriction(*this, f_ordering, type));
res = new ConformingFaceRestriction(*this, f_ordering, type);
}
return L2F.emplace(key, std::move(res)).first->second.get();
}
}
const InterpolationManager &FiniteElementSpace::GetInterpolationManager(
ElementDofOrdering f_ordering, FaceType type) const
{
const auto key = make_tuple(f_ordering, type);
auto it = interpolations.find(key);
if (it != interpolations.end())
{
return *it->second;
}
else
{
auto interp = make_unique<InterpolationManager>(*this, f_ordering, type);
int face_idx = 0;
for (int f = 0; f < mesh->GetNumFacesWithGhost(); ++f)
{
Mesh::FaceInformation face = mesh->GetFaceInformation(f);
if (!face.IsOfFaceType(type) || face.IsNonconformingCoarse())
{
continue;
}
if (face.IsConforming() || face.IsBoundary())
{
interp->RegisterFaceConformingInterpolation(face, face_idx);
}
else
{
interp->RegisterFaceCoarseToFineInterpolation(face, face_idx);
}
++face_idx;
}
// Transform the interpolation matrix map into contiguous memory.
interp->LinearizeInterpolatorMapIntoVector();
interp->InitializeNCInterpConfig();
return *interpolations.emplace(key, std::move(interp)).first->second;
L2F[key] = res;
return res;
}
}
@@ -4009,8 +3969,11 @@ void FiniteElementSpace::Destroy()
delete E2Q_array[i];
}
E2Q_array.SetSize(0);
for (auto &x : L2F)
{
delete x.second;
}
L2F.clear();
interpolations.clear();
for (int i = 0; i < E2IFQ_array.Size(); i++)
{
delete E2IFQ_array[i];
+12 -9
View File
@@ -13,7 +13,6 @@
#define MFEM_FESPACE
#include "../config/config.hpp"
#include "../general/hash_util.hpp"
#include "../linalg/ordering.hpp"
#include "../linalg/sparsemat.hpp"
#include "../mesh/mesh.hpp"
@@ -321,11 +320,18 @@ protected:
mutable OperatorHandle L2E_nat, L2E_lex;
/// The face restriction operators, see GetFaceRestriction().
using key_face = std::tuple<bool, ElementDofOrdering, FaceType, L2FaceValues>;
mutable std::unordered_map<key_face,std::unique_ptr<FaceRestriction>,
TupleHasher> L2F;
mutable std::unordered_map<std::tuple<ElementDofOrdering,FaceType>,
std::unique_ptr<InterpolationManager>, TupleHasher> interpolations;
struct key_hash
{
std::size_t operator()(const key_face& k) const
{
return std::get<0>(k)
+ 2 * (int)std::get<1>(k)
+ 4 * (int)std::get<2>(k)
+ 8 * (int)std::get<3>(k);
}
};
using map_L2F = std::unordered_map<const key_face,FaceRestriction*,key_hash>;
mutable map_L2F L2F;
mutable Array<QuadratureInterpolator*> E2Q_array;
mutable Array<FaceQuadratureInterpolator*> E2IFQ_array;
@@ -745,9 +751,6 @@ public:
ElementDofOrdering f_ordering, FaceType,
L2FaceValues mul = L2FaceValues::DoubleValued) const;
const InterpolationManager &GetInterpolationManager(
ElementDofOrdering f_ordering, FaceType type) const;
/** @brief Return a QuadratureInterpolator that interpolates E-vectors to
quadrature point values and/or derivatives (Q-vectors). */
/** An E-vector represents the element-wise discontinuous version of the FE
+32 -59
View File
@@ -234,7 +234,7 @@ void FindPointsGSLIB::Setup(Mesh &m, const double bb_t, const double newt_tol,
}
void FindPointsGSLIB::FindPoints(const Vector &point_pos,
const int point_pos_ordering)
int point_pos_ordering)
{
MFEM_VERIFY(setupflag, "Use FindPointsGSLIB::Setup before finding points.");
bool dev_mode = (point_pos.UseDevice() && Device::IsEnabled());
@@ -482,7 +482,7 @@ void FindPointsGSLIB::SetupDevice()
}
void FindPointsGSLIB::FindPointsOnDevice(const Vector &point_pos,
const int point_pos_ordering)
int point_pos_ordering)
{
if (!DEV.setup_device)
{
@@ -505,13 +505,13 @@ void FindPointsGSLIB::FindPointsOnDevice(const Vector &point_pos,
if (dim == 2)
{
FindPointsLocal2(point_pos, point_pos_ordering, gsl_code, gsl_elem,
gsl_ref, gsl_dist, points_cnt);
FindPointsLocal2(point_pos, point_pos_ordering, gsl_code, gsl_elem, gsl_ref,
gsl_dist, points_cnt);
}
else
{
FindPointsLocal3(point_pos, point_pos_ordering, gsl_code, gsl_elem,
gsl_ref, gsl_dist, points_cnt);
FindPointsLocal3(point_pos, point_pos_ordering, gsl_code, gsl_elem, gsl_ref,
gsl_dist, points_cnt);
}
// Sync from device to host
@@ -1085,7 +1085,7 @@ void FindPointsGSLIB::InterpolateOnDevice(const Vector &field_in_evec,
#else
void FindPointsGSLIB::SetupDevice() {};
void FindPointsGSLIB::FindPointsOnDevice(const Vector &point_pos,
const int point_pos_ordering) {};
int point_pos_ordering) {};
void FindPointsGSLIB::InterpolateOnDevice(const Vector &field_in_evec,
Vector &field_out,
const int nel, const int ncomp,
@@ -1094,8 +1094,7 @@ void FindPointsGSLIB::InterpolateOnDevice(const Vector &field_in_evec,
#endif
void FindPointsGSLIB::FindPoints(Mesh &m, const Vector &point_pos,
const int point_pos_ordering,
const double bb_t,
int point_pos_ordering, const double bb_t,
const double newt_tol, const int npt_max)
{
if (!setupflag || (mesh != &m) )
@@ -1106,28 +1105,16 @@ void FindPointsGSLIB::FindPoints(Mesh &m, const Vector &point_pos,
}
void FindPointsGSLIB::Interpolate(const Vector &point_pos,
const GridFunction &field_in,
Vector &field_out,
const int point_pos_ordering)
const GridFunction &field_in, Vector &field_out,
int point_pos_ordering)
{
FindPoints(point_pos, point_pos_ordering);
Interpolate(field_in, field_out);
}
void FindPointsGSLIB::Interpolate(const Vector &point_pos,
const GridFunction &field_in,
Vector &field_out,
const int point_pos_ordering,
const int field_out_ordering)
{
FindPoints(point_pos, point_pos_ordering);
Interpolate(field_in, field_out, field_out_ordering);
}
void FindPointsGSLIB::Interpolate(Mesh &m, const Vector &point_pos,
const GridFunction &field_in,
Vector &field_out,
const int point_pos_ordering)
const GridFunction &field_in, Vector &field_out,
int point_pos_ordering)
{
FindPoints(m, point_pos, point_pos_ordering);
Interpolate(field_in, field_out);
@@ -1483,7 +1470,7 @@ void FindPointsGSLIB::SetupSplitMeshesAndIntegrationRules(const int order)
}
void FindPointsGSLIB::GetNodalValues(const GridFunction *gf_in,
Vector &node_vals) const
Vector &node_vals)
{
const GridFunction *nodes = gf_in;
const FiniteElementSpace *fes = nodes->FESpace();
@@ -1771,13 +1758,6 @@ void FindPointsGSLIB::MapRefPosAndElemIndices()
void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
Vector &field_out)
{
Interpolate(field_in, field_out, field_in.FESpace()->GetOrdering());
}
void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
Vector &field_out,
const int field_out_ordering)
{
const int gf_order = field_in.FESpace()->GetMaxElementOrder(),
mesh_order = mesh->GetNodalFESpace()->GetMaxElementOrder();
@@ -1820,7 +1800,7 @@ void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
const int maxOrder = field_in.FESpace()->GetMaxElementOrder();
InterpolateOnDevice(node_vals, field_out, NE_split_total, ncomp,
maxOrder+1, field_out_ordering);
maxOrder+1, field_in.FESpace()->GetOrdering());
return;
#endif
}
@@ -1832,13 +1812,12 @@ void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
field_in.FESpace()->IsVariableOrder() ==
mesh->GetNodalFESpace()->IsVariableOrder())
{
InterpolateH1(field_in, field_out, field_out_ordering);
InterpolateH1(field_in, field_out);
return;
}
else
{
InterpolateGeneral(field_in, field_out,
field_out_ordering);
InterpolateGeneral(field_in, field_out);
if (!fec_l2 || avgtype == AvgType::NONE) { return; }
}
@@ -1882,11 +1861,11 @@ void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
if (gf_order_h1 == mesh_order) // basis is GaussLobatto by default
{
InterpolateH1(field_in_h1, field_out_l2, field_out_ordering);
InterpolateH1(field_in_h1, field_out_l2);
}
else
{
InterpolateGeneral(field_in_h1, field_out_l2, field_out_ordering);
InterpolateGeneral(field_in_h1, field_out_l2);
}
// Copy interpolated values for the points on element border
@@ -1894,7 +1873,7 @@ void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
{
for (int i = 0; i < indl2.Size(); i++)
{
int idx = field_out_ordering == Ordering::byNODES?
int idx = field_in_h1.FESpace()->GetOrdering() == Ordering::byNODES?
indl2[i] + j*points_cnt:
indl2[i]*ncomp + j;
field_out(idx) = field_out_l2(idx);
@@ -1904,8 +1883,7 @@ void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
}
void FindPointsGSLIB::InterpolateH1(const GridFunction &field_in,
Vector &field_out,
const int field_out_ordering)
Vector &field_out)
{
FiniteElementSpace ind_fes(mesh, field_in.FESpace()->FEColl());
if (field_in.FESpace()->IsVariableOrder())
@@ -1935,8 +1913,7 @@ void FindPointsGSLIB::InterpolateH1(const GridFunction &field_in,
dataptrout = i*points_cnt;
if (field_in.FESpace()->GetOrdering() == Ordering::byNODES)
{
field_in_scalar.NewDataAndSize(field_in.GetData()+dataptrin,
points_fld);
field_in_scalar.NewDataAndSize(field_in.GetData()+dataptrin, points_fld);
}
else
{
@@ -1968,7 +1945,7 @@ void FindPointsGSLIB::InterpolateH1(const GridFunction &field_in,
(gslib::findpts_data_3 *)this->fdataD);
}
}
if (field_out_ordering == Ordering::byVDIM)
if (field_in.FESpace()->GetOrdering() == Ordering::byVDIM)
{
Vector field_out_temp = field_out;
for (int i = 0; i < ncomp; i++)
@@ -1982,8 +1959,7 @@ void FindPointsGSLIB::InterpolateH1(const GridFunction &field_in,
}
void FindPointsGSLIB::InterpolateGeneral(const GridFunction &field_in,
Vector &field_out,
const int field_out_ordering)
Vector &field_out)
{
int ncomp = field_in.VectorDim(),
nptorig = points_cnt,
@@ -2003,7 +1979,7 @@ void FindPointsGSLIB::InterpolateGeneral(const GridFunction &field_in,
if (dim == 3) { ip.z = gsl_mfem_ref(index*dim + 2); }
Vector localval(ncomp);
field_in.GetVectorValue(gsl_mfem_elem[index], ip, localval);
if (field_out_ordering == Ordering::byNODES)
if (field_in.FESpace()->GetOrdering() == Ordering::byNODES)
{
for (int i = 0; i < ncomp; i++)
{
@@ -2038,10 +2014,7 @@ void FindPointsGSLIB::InterpolateGeneral(const GridFunction &field_in,
for (int index = 0; index < npt; index++)
{
if (gsl_code[index] == 2) { continue; }
for (int d = 0; d < dim; ++d)
{
pt->r[d]= gsl_mfem_ref(index*dim + d);
}
for (int d = 0; d < dim; ++d) { pt->r[d]= gsl_mfem_ref(index*dim + d); }
pt->index = index;
pt->proc = gsl_proc[index];
pt->el = gsl_mfem_elem[index];
@@ -2131,7 +2104,7 @@ void FindPointsGSLIB::InterpolateGeneral(const GridFunction &field_in,
sdpt = (struct send_pt *)sendpt->ptr;
for (int index = 0; index < static_cast<int>(sendpt->n); index++)
{
int idx = field_out_ordering == Ordering::byNODES ?
int idx = field_in.FESpace()->GetOrdering() == Ordering::byNODES ?
sdpt->index + j*nptorig :
sdpt->index*ncomp + j;
field_out(idx) = sdpt->ival;
@@ -2273,7 +2246,7 @@ void FindPointsGSLIB::DistributeInterpolatedValues(const Vector &int_vals,
}
}
void FindPointsGSLIB::GetAxisAlignedBoundingBoxes(Vector &aabb) const
void FindPointsGSLIB::GetAxisAlignedBoundingBoxes(Vector &aabb)
{
MFEM_VERIFY(setupflag, "Call FindPointsGSLIB::Setup method first");
auto *findptsData3 = (gslib::findpts_data_3 *)this->fdataD;
@@ -2344,7 +2317,7 @@ void FindPointsGSLIB::GetAxisAlignedBoundingBoxes(Vector &aabb) const
}
void FindPointsGSLIB::GetOrientedBoundingBoxes(DenseTensor &obbA, Vector &obbC,
Vector &obbV) const
Vector &obbV)
{
MFEM_VERIFY(setupflag, "Call FindPointsGSLIB::Setup method first");
auto *findptsData3 = (gslib::findpts_data_3 *)this->fdataD;
@@ -2529,8 +2502,8 @@ void OversetFindPointsGSLIB::Setup(Mesh &m, const int meshid,
}
void OversetFindPointsGSLIB::FindPoints(const Vector &point_pos,
const Array<unsigned int> &point_id,
const int point_pos_ordering)
Array<unsigned int> &point_id,
int point_pos_ordering)
{
MFEM_VERIFY(setupflag, "Use OversetFindPointsGSLIB::Setup before "
"finding points.");
@@ -2609,10 +2582,10 @@ void OversetFindPointsGSLIB::FindPoints(const Vector &point_pos,
}
void OversetFindPointsGSLIB::Interpolate(const Vector &point_pos,
const Array<unsigned int> &point_id,
Array<unsigned int> &point_id,
const GridFunction &field_in,
Vector &field_out,
const int point_pos_ordering)
int point_pos_ordering)
{
FindPoints(point_pos, point_id, point_pos_ordering);
Interpolate(field_in, field_out);
+15 -32
View File
@@ -119,13 +119,11 @@ protected:
} DEV;
/// Use GSLIB for communication and interpolation
virtual void InterpolateH1(const GridFunction &field_in, Vector &field_out,
const int field_out_ordering);
virtual void InterpolateH1(const GridFunction &field_in, Vector &field_out);
/// Uses GSLIB Crystal Router for communication followed by MFEM's
/// interpolation functions
virtual void InterpolateGeneral(const GridFunction &field_in,
Vector &field_out,
const int field_out_ordering);
Vector &field_out);
/// Since GSLIB is designed to work with quads/hexes, we split every
/// triangle/tet/prism/pyramid element into quads/hexes.
@@ -142,7 +140,7 @@ protected:
virtual void SetupSplitMeshesAndIntegrationRules(const int order);
/// Get GridFunction value at the points expected by GSLIB.
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals) const;
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals);
/// Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For simplices,
/// find the original element number (that was split into micro quads/hexes)
@@ -184,7 +182,7 @@ protected:
These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
byVDim: (XYZ,XYZ,....XYZ) specified by @a point_pos_ordering. */
void FindPointsOnDevice(const Vector &point_pos,
const int point_pos_ordering = Ordering::byNODES);
int point_pos_ordering = Ordering::byNODES);
/** Interpolation of field values at prescribed reference space positions.
@param[in] field_in_evec E-vector of grid function to be interpolated.
@@ -255,15 +253,10 @@ public:
#gsl_dist Distance between the sought and the found point
in physical space. */
void FindPoints(const Vector &point_pos,
const int point_pos_ordering = Ordering::byNODES);
/// Convenience function when point positions are in a ParticleVector
void FindPoints(const ParticleVector &point_pos)
{
FindPoints(point_pos, point_pos.GetOrdering());
}
int point_pos_ordering = Ordering::byNODES);
/// Setup FindPoints and search positions
void FindPoints(Mesh &m, const Vector &point_pos,
const int point_pos_ordering = Ordering::byNODES,
int point_pos_ordering = Ordering::byNODES,
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
const int npt_max = 256);
@@ -273,28 +266,20 @@ public:
\p field_in is in H1 and in the same space as the
mesh that was given to Setup().
@param[out] field_out Interpolated values. For points that are not found
the value is set to #default_interp_value.
The output ordering is determined from field_in.*/
the value is set to #default_interp_value. */
virtual void Interpolate(const GridFunction &field_in, Vector &field_out);
/// Interpolation of field values, with output ordering specification.
virtual void Interpolate(const GridFunction &field_in, Vector &field_out,
const int field_out_ordering);
/** Search positions and interpolate. The ordering (byNODES or byVDIM) of
the output values in \p field_out corresponds to the ordering used
in the input GridFunction \p field_in. */
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
Vector &field_out,
const int point_pos_ordering = Ordering::byNODES);
/// Search positions and interpolate with given point and output ordering.
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
Vector &field_out, const int point_pos_ordering,
const int field_out_ordering);
int point_pos_ordering = Ordering::byNODES);
/** Setup FindPoints, search positions and interpolate. The ordering (byNODES
or byVDIM) of the output values in \p field_out corresponds to the
ordering used in the input GridFunction \p field_in. */
void Interpolate(Mesh &m, const Vector &point_pos,
const GridFunction &field_in, Vector &field_out,
const int point_pos_ordering = Ordering::byNODES);
int point_pos_ordering = Ordering::byNODES);
/// Average type to be used for L2 functions in-case a point is located at
/// an element boundary where the function might be multi-valued.
@@ -391,7 +376,7 @@ public:
/// The size of the returned vector is (nel x nverts x dim), where nel is the
/// number of elements (after splitting for simplcies), nverts is number of
/// vertices (4 in 2D, 8 in 3D), and dim is the spatial dimension.
void GetAxisAlignedBoundingBoxes(Vector &aabb) const;
void GetAxisAlignedBoundingBoxes(Vector &aabb);
/// Return the oriented bounding boxes (OBB) computed during \ref Setup.
/// Each OBB is represented using the inverse transformation (A^{-1}) and
@@ -401,8 +386,7 @@ public:
/// size (dim x dim x nel), and the OBB centers are returned in \p obbC,
/// a vector of size (nel x dim). The vertices of the OBBs are returned in
/// \p obbV, a vector of size (nel x nverts x dim) .
void GetOrientedBoundingBoxes(DenseTensor &obbA, Vector &obbC,
Vector &obbV) const;
void GetOrientedBoundingBoxes(DenseTensor &obbA, Vector &obbC, Vector &obbV);
};
/** \brief OversetFindPointsGSLIB enables use of findpts for arbitrary number of
@@ -462,14 +446,13 @@ public:
byNodes: (XXX...,YYY...,ZZZ) or
byVDim: (XYZ,XYZ,....XYZ) */
void FindPoints(const Vector &point_pos,
const Array<unsigned int> &point_id,
const int point_pos_ordering = Ordering::byNODES);
Array<unsigned int> &point_id,
int point_pos_ordering = Ordering::byNODES);
/** Search positions and interpolate */
void Interpolate(const Vector &point_pos,
const Array<unsigned int> &point_id,
void Interpolate(const Vector &point_pos, Array<unsigned int> &point_id,
const GridFunction &field_in, Vector &field_out,
const int point_pos_ordering = Ordering::byNODES);
int point_pos_ordering = Ordering::byNODES);
using FindPointsGSLIB::Interpolate;
};
+1 -7
View File
@@ -789,6 +789,7 @@ void Hybridization::ComputeH()
}
else
{
// TODO: add ones on the diagonal of zero rows
V->Finalize();
Array<HYPRE_BigInt> V_J(V->NumNonZeroElems());
MFEM_ASSERT(c_pfes, "");
@@ -822,13 +823,6 @@ void Hybridization::ComputeH()
MFEM_VERIFY(pH.Type() != Operator::PETSC_MATIS, "To be implemented");
pH.MakePtAP(plpH, pP);
delete lpH;
HypreParMatrix *hH = pH.As<HypreParMatrix>();
MFEM_ASSERT(hH, "");
SparseMatrix H_diag;
hH->GetDiag(H_diag);
H_diag.SetDiagIdentity();
}
#endif
}
+273 -453
View File
File diff suppressed because it is too large Load Diff
+2 -20
View File
@@ -14,11 +14,8 @@
#include "../config/config.hpp"
#include "../general/array.hpp"
#include "../linalg/operator.hpp"
#include "../linalg/vector.hpp"
#include <memory>
namespace mfem
{
@@ -48,30 +45,15 @@ protected:
Array<int> hat_dof_gather_map;
Array<DofType> hat_dof_marker;
Array<int> el_to_face; ///< Element to face connectivity.
Array<int> el_face_offsets; ///< Per-element offsets into @a el_to_face.
Array<int> face_to_el; ///< Face-to-element connectivity.
Array<int> face_face_offsets; ///< Face-to-face offsets.
int n_el_face; ///< Total number of element-to-face connections.
int n_face_face; ///< Total number of face-to-face connections.
Array<int> el_to_face;
Array<int> face_to_el;
Vector Ct_mat; ///< Constraint matrix (transposed) stored element-wise.
/// @name For parallel non-conforming meshes
///@{
std::unique_ptr<Operator> P_pc; ///< Partially conforming prolongation.
std::unique_ptr<Operator> P_nbr; ///< Face-neighbor prolongation.
///@}
Array<int> idofs, bdofs;
Vector Ahat, Ahat_ii, Ahat_ib, Ahat_bi, Ahat_bb;
Array<int> Ahat_ii_piv, Ahat_bb_piv;
/// Return the (partially) conforming prolongation on the constraint space.
const Operator &GetProlongation() const;
public:
/// Construct the constraint matrix.
void ConstructC();
+5 -8
View File
@@ -1004,16 +1004,13 @@ inline void SmemPADiffusionApply3D(const int NE,
const int max_d1d = T_D1D ? T_D1D : DeviceDofQuadLimits::Get().MAX_D1D;
MFEM_VERIFY(D1D <= max_d1d, "");
MFEM_VERIFY(Q1D <= max_q1d, "");
const auto b = Reshape(b_.Read(), Q1D, D1D);
const auto g = Reshape(g_.Read(), Q1D, D1D);
const auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, symmetric ? 6 : 9, NE);
const auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, symmetric ? 6 : 9, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_VERIFY(D1D <= Q1D, "THREAD_DIRECT requires D1D <= Q1D");
mfem::forall_3D<T_Q1D*T_Q1D*T_Q1D>(NE,
Q1D, Q1D, Q1D,
[=] MFEM_HOST_DEVICE (int e)
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
+6 -6
View File
@@ -1133,11 +1133,11 @@ inline void SmemPAMassApply3D(const int NE,
const int max_d1d = T_D1D ? T_D1D : DeviceDofQuadLimits::Get().MAX_D1D;
MFEM_VERIFY(D1D <= max_d1d, "");
MFEM_VERIFY(Q1D <= max_q1d, "");
const auto b = b_.Read();
const auto d = d_.Read();
const auto x = x_.Read();
auto b = b_.Read();
auto d = d_.Read();
auto x = x_.Read();
auto y = y_.ReadWrite();
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
internal::SmemPAMassApply3D_Element<T_D1D,T_Q1D>(e, NE, b, d, x, y, d1d, q1d);
});
@@ -1156,8 +1156,8 @@ inline void EAMassAssemble1D(const int NE,
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(basis.Read(), Q1D, D1D);
const auto D = Reshape(padata.Read(), Q1D, NE);
auto B = Reshape(basis.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, NE);
auto M = Reshape(add ? eadata.ReadWrite() : eadata.Write(), D1D, D1D, NE);
mfem::forall_2D(NE, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
{
+20 -98
View File
@@ -28,7 +28,7 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
const FaceType ftype = FaceType::Interior;
const int nf = mesh.GetNFbyType(ftype);
const Geometry::Type geom = mesh.GetTypicalFaceGeometry();
const Geometry::Type geom = mesh.GetFaceGeometry(0);
const int trial_order = trial_fes.GetMaxElementOrder();
const int test_order = test_fes.GetMaxElementOrder();
const int qorder = test_order + trial_order - 1;
@@ -47,7 +47,7 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
});
}
const FiniteElement &trial_face_el = *trial_fes.GetTypicalTraceElement();
const FiniteElement &trial_face_el = *trial_fes.GetFaceElement(0);
const auto maps = &trial_face_el.GetDofToQuad(ir, DofToQuad::TENSOR);
const int ndof_face = trial_face_el.GetDof();
@@ -72,7 +72,7 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
MFEM_ABORT("Unknown kernel.");
}
const FiniteElement &test_el = *test_fes.GetTypicalFE();
const FiniteElement &test_el = *test_fes.GetFE(0);
const int n_faces_per_el = 2*dim; // assuming tensor product
// Get all the local face maps (mapping from lexicographic face index to
// lexicographic volume index, depending on the local face index).
@@ -90,10 +90,10 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
Array<int> face_info(nf * 4);
{
int fidx = 0;
for (int f = 0; f < mesh.GetNumFacesWithGhost(); ++f)
for (int f = 0; f < mesh.GetNumFaces(); ++f)
{
Mesh::FaceInformation finfo = mesh.GetFaceInformation(f);
if (!finfo.IsInterior() || finfo.IsNonconformingCoarse()) { continue; }
if (!finfo.IsInterior()) { continue; }
face_info[0 + fidx*4] = finfo.element[0].local_face_id;
face_info[1 + fidx*4] = finfo.element[0].orientation;
face_info[2 + fidx*4] = finfo.element[1].local_face_id;
@@ -114,7 +114,7 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
else
{
d_emat = emat.Write();
emat = 0.0; // Will execute on device, since Write() sets the device flag
mfem::forall(emat.Size(), [=] MFEM_HOST_DEVICE (int i) { d_emat[i] = 0.0; });
}
const auto face_mats = Reshape(mass_emat.Read(), ndof_face, ndof_face, nf);
@@ -133,104 +133,26 @@ void NormalTraceJumpIntegrator::AssembleEAInteriorFaces(
}
};
auto permute_face_2 = [=] MFEM_HOST_DEVICE(int local_face_1, int local_face_2,
int orient, int size1d, int index)
mfem::forall_3D(nf, ndof_face, ndof_face, 2, [=] MFEM_HOST_DEVICE (int f)
{
if (dim == 2)
MFEM_FOREACH_THREAD(el_i, z, 2)
{
return internal::PermuteFace2D(local_face_1, local_face_2, orient,
size1d, index);
}
else // dim == 3
{
return internal::PermuteFace3D(local_face_1, local_face_2, orient,
size1d, index);
}
};
if (mesh.Conforming())
{
mfem::forall_3D(nf, ndof_face, ndof_face, 2, [=] MFEM_HOST_DEVICE (int f)
{
MFEM_FOREACH_THREAD(el_i, z, 2)
const int lf_i = d_face_info(0, el_i, f);
const int orient = d_face_info(1, el_i, f);
// Loop over face indices in "native ordering"
MFEM_FOREACH_THREAD(i_lex, x, ndof_face)
{
const int lf_i = d_face_info(0, el_i, f);
const int orient = d_face_info(1, el_i, f);
// Loop over face indices in "native ordering"
MFEM_FOREACH_THREAD(i_lex, x, ndof_face)
// Convert to lexicographic relative to the face itself
const int i_face = permute_face(lf_i, orient, d1d, i_lex);
// Convert from lexicographic face DOF to volume DOF
const int i = d_face_maps(i_lex, lf_i);
MFEM_FOREACH_THREAD(j, y, ndof_face)
{
// Convert to lexicographic relative to the face itself
const int i_face = permute_face(lf_i, orient, d1d, i_lex);
// Convert from lexicographic face DOF to volume DOF
const int i = d_face_maps(i_lex, lf_i);
MFEM_FOREACH_THREAD(j, y, ndof_face)
{
el_mats(i, j, el_i, f) += face_mats(i_face, j, f);
}
el_mats(i, j, el_i, f) += face_mats(i_face, j, f);
}
}
});
}
else
{
const InterpolationManager &interp =
test_fes.GetInterpolationManager(ElementDofOrdering::LEXICOGRAPHIC, ftype);
auto interp_configs = interp.GetFaceInterpConfig().Read();
const int nc_size = interp.GetNumInterpolators();
auto d_interp = Reshape(interp.GetInterpolators().Read(),
ndof_face, ndof_face, nc_size);
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int f)
{
const InterpConfig conf = interp_configs[f];
const int master_side = conf.master_side;
const int interp_index = conf.index;
const int lf_0 = d_face_info(0, 0, f);
for (int el_i = 0; el_i < 2; ++el_i)
{
const int lf_i = d_face_info(0, el_i, f);
const int orient = d_face_info(1, el_i, f);
for (int j = 0; j < ndof_face; j++)
{
for (int i_lex = 0; i_lex < ndof_face; i_lex++)
{
real_t val = 0.0;
if (conf.is_non_conforming && el_i == master_side)
{
// Interpolate from el_i (coarse element) to the fine face.
// The mapping is given by d_interp, which uses indices
// relative to element 0.
// i0 is lexicographic relative to element 0
const int i0 = permute_face_2(lf_i, lf_0, orient, d1d, i_lex);
// k0 is lexicographic relative to element 0
for (int k0 = 0; k0 < ndof_face; k0++)
{
// k is relative to the face itself
const int k = permute_face(lf_0, orient, d1d, k0);
val += d_interp(k0, i0, interp_index)
* face_mats(k, j, f);
}
}
else
{
// Convert to lexicographic relative to the face itself
const int i_face = permute_face(lf_i, orient, d1d, i_lex);
val = face_mats(i_face, j, f);
}
// Convert from lexicographic face DOF to volume DOF
const int i = d_face_maps(i_lex, lf_i);
el_mats(i, j, el_i, f) += val;
}
}
}
});
}
}
});
}
}
+2 -2
View File
@@ -54,7 +54,7 @@ void SmemPAVectorDiffusionApply2D(const int NE,
const auto XE = Reshape(x.Read(), D1D, D1D, SDIM, NE);
auto YE = Reshape(y.ReadWrite(), D1D, D1D, SDIM, NE);
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
@@ -120,7 +120,7 @@ void SmemPAVectorDiffusionApply3D(const int NE,
const auto XE = Reshape(x.Read(), D1D, D1D, D1D, SDIM, NE);
auto YE = Reshape(y.ReadWrite(), D1D, D1D, D1D, SDIM, NE);
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
+2 -2
View File
@@ -51,7 +51,7 @@ void SmemPAVectorMassApply2D(const int NE,
const auto X = Reshape(x.Read(), D1D, D1D, VDIM, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
@@ -119,7 +119,7 @@ void SmemPAVectorMassApply3D(const int NE,
const auto X = Reshape(x.Read(), D1D, D1D, D1D, VDIM, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
+31 -3
View File
@@ -14,7 +14,6 @@
#include "../config/config.hpp"
#include "kernel_reporter.hpp"
#include "../general/hash_util.hpp"
#include <unordered_map>
#include <tuple>
#include <type_traits>
@@ -87,6 +86,35 @@ namespace mfem
} \
}
/// @brief Hashes variadic packs for which each type contained in the variadic
/// pack has a specialization of `std::hash` available.
///
/// For example, packs containing int, bool, enum values, etc.
template<typename ...KernelParameters>
struct KernelDispatchKeyHash
{
private:
template<int N>
size_t operator()(std::tuple<KernelParameters...> value) const { return 0; }
// The hashing formula here is taken directly from the Boost library, with
// the magic number 0x9e3779b9 chosen to minimize hashing collisions.
template<std::size_t N, typename THead, typename... TTail>
size_t operator()(std::tuple<KernelParameters...> value) const
{
constexpr int Index = N - sizeof...(TTail) - 1;
auto lhs_hash = std::hash<THead>()(std::get<Index>(value));
auto rhs_hash = operator()<N, TTail...>(value);
return lhs_hash^(rhs_hash + 0x9e3779b9 + (lhs_hash<<6) + (lhs_hash>>2));
}
public:
/// Returns the hash of the given @a value.
size_t operator()(std::tuple<KernelParameters...> value) const
{
return operator()<sizeof...(KernelParameters),KernelParameters...>(value);
}
};
namespace internal { template<typename... Types> struct KernelTypeList { }; }
template<typename... T> class KernelDispatchTable { };
@@ -100,8 +128,8 @@ class KernelDispatchTable<Kernels,
internal::KernelTypeList<Params...>,
internal::KernelTypeList<OptParams...>>
{
using TableType =
std::unordered_map<std::tuple<Params...>, Signature, TupleHasher>;
using TableType = std::unordered_map<std::tuple<Params...>,
Signature, KernelDispatchKeyHash<Params...>>;
TableType table;
/// @brief Call function @a f with arguments @a args (perfect forwaring).
+147
View File
@@ -1054,7 +1054,154 @@ void DGElasticityDirichletLFIntegrator::AssembleRHSElementVect(
}
}
void SlidingElasticityLFIntegrator::AssembleRHSElementVect(
const FiniteElement &el, ElementTransformation &Tr, Vector &elvect)
{
mfem_error("SlidingElasticityLFIntegrator::AssembleRHSElementVect");
}
void SlidingElasticityLFIntegrator::AssembleRHSElementVect(
const FiniteElement &el, FaceElementTransformations &Tr, Vector &elvect)
{
MFEM_ASSERT(Tr.Elem2No < 0, "interior boundary is not supported");
#ifdef MFEM_THREAD_SAFE
Vector shape;
DenseMatrix dshape;
DenseMatrix adjJ;
DenseMatrix dshape_ps;
Vector nor;
Vector dshape_dn;
Vector dshape_du;
real_t g_val;
Vector nt_val;
#endif
const int dim = el.GetDim();
const int ndofs = el.GetDof();
const int nvdofs = dim*ndofs;
elvect.SetSize(nvdofs);
elvect = 0.0;
adjJ.SetSize(dim);
shape.SetSize(ndofs);
dshape.SetSize(ndofs, dim);
dshape_ps.SetSize(ndofs, dim);
nor.SetSize(dim);
dshape_dn.SetSize(ndofs);
dshape_du.SetSize(ndofs);
nt_val.SetSize(dim);
const IntegrationRule *ir = IntRule;
if (ir == NULL)
{
const int order = 2*el.GetOrder(); // <-----
ir = &IntRules.Get(Tr.GetGeometryType(), order);
}
for (int pi = 0; pi < ir->GetNPoints(); ++pi)
{
const IntegrationPoint &ip = ir->IntPoint(pi);
// Set the integration point in the face and the neighboring element
Tr.SetAllIntPoints(&ip);
// Access the neighboring element's integration point
const IntegrationPoint &eip = Tr.GetElement1IntPoint();
el.CalcShape(eip, shape);
el.CalcDShape(eip, dshape);
CalcAdjugate(Tr.Elem1->Jacobian(), adjJ);
Mult(dshape, adjJ, dshape_ps);
if (dim == 1)
{
nor(0) = 2*eip.x - 1.0;
}
else
{
CalcOrtho(Tr.Jacobian(), nor);
}
if (!nt)
{
// Set nt to the unit normal vector if not provided
nt_val = nor;
nt_val /= nt_val.Norml2();
}
else
{
// Evaluate the vector field using the face transformation.
nt->Eval(nt_val, Tr, ip);
}
// Evaluate the Dirichlet b.c. using the face transformation.
g_val = g->Eval(Tr, ip);
real_t WL, WM, jcoef;
{
const real_t W = ip.weight / Tr.Elem1->Weight();
WL = W * lambda->Eval(*Tr.Elem1, eip);
WM = W * mu->Eval(*Tr.Elem1, eip);
jcoef = kappa * (WL + 2.0*WM) * (nor*nor);
dshape_ps.Mult(nor, dshape_dn);
dshape_ps.Mult(nt_val, dshape_du);
}
// alpha < g, (lambda div(v) I + mu (grad(v) + grad(v)^T)) n . ñ > +
// + kappa < h^{-1} (lambda + 2 mu) g, v . ñ >
// i = idof + ndofs * im
// v_phi(i,d) = delta(im,d) phi(idof)
// div(v_phi(i)) = dphi(idof,im)
// (grad(v_phi(i)))(k,l) = delta(im,k) dphi(idof,l)
//
// term 1:
// alpha < g, lambda div(v_phi(i)) n . ñ > =
// alpha lambda g div(v_phi(i)) (n.ñ) =
// alpha lambda g dphi(idof,im) (n.ñ) --> quadrature -->
// ip.weight/det(J1) alpha lambda g (nor.ñ) dshape_ps(idof,im) =
// alpha * WL * g_val * (nor*nt_val) * dshape_ps(idof,im)
// term 2:
// alpha < g, mu grad(v_phi(i)) n . ñ > =
// alpha mu g ñ^T grad(v_phi(i)) n =
// alpha mu g ñ(k) delta(im,k) dphi(idof,l) n(l) =
// alpha mu g ñ(im) dphi(idof,l) n(l) --> quadrature -->
// ip.weight/det(J1) alpha mu ñ(im) g dshape_ps(idof,l) nor(l) =
// alpha * WM * g_val * nt_val(im) * dshape_dn(idof)
// term 3:
// alpha < g, mu (grad(v_phi(i)))^T n . ñ > =
// alpha mu g n^T grad(v_phi(i)) ñ =
// alpha mu g n(k) delta(im,k) dphi(idof,l) ñ(l) =
// alpha mu g n(im) dphi(idof,l) ñ(l) --> quadrature -->
// ip.weight/det(J1) alpha mu g nor(im) dshape_ps(idof,l) ñ(l) =
// alpha * WM * g_val * nor(im) * dshape_du(idof)
// term j:
// < kappa h^{-1} (lambda + 2 mu) g, ñ . v_phi(i) > =
// kappa/h (lambda + 2 mu) g ñ(k) v_phi(i,k) =
// kappa/h (lambda + 2 mu) g ñ(k) delta(im,k) phi(idof) =
// kappa/h (lambda + 2 mu) g ñ(im) phi(idof) --> quadrature -->
// [ 1/h = |nor|/det(J1) ]
// ip.weight/det(J1) |nor|^2 (lambda + 2 mu) kappa g ñ(im) phi(idof) =
// jcoef * g_val * nt_val(im) * shape(idof)
WM *= alpha;
const real_t t1 = alpha * WL * g_val * (nor*nt_val);
for (int im = 0, i = 0; im < dim; ++im)
{
const real_t t2 = WM * g_val * nt_val(im);
const real_t t3 = WM * g_val * nor(im);
const real_t tj = jcoef * g_val * nt_val(im);
for (int idof = 0; idof < ndofs; ++idof, ++i)
{
elvect(i) += (t1*dshape_ps(idof,im) + t2*dshape_dn(idof) +
t3*dshape_du(idof) + tj*shape(idof));
}
}
}
}
void WhiteGaussianNoiseDomainLFIntegrator::AssembleRHSElementVect
(const FiniteElement &el,
+56
View File
@@ -646,6 +646,62 @@ public:
using LinearFormIntegrator::AssembleRHSElementVect;
};
/** Boundary linear form integrator for imposing non-zero Dirichlet boundary
conditions, in a Nitsche elasticity formulation. Specifically, the linear
form is given by
$$
\begin{split}
b(v) &:= \alpha \int_\Gamma (\lambda\, \mathrm{div}(v)\, I + \mu (\nabla v
+ \nabla v^{\mathrm{T}}))\, n \cdot \tilde{n}\, g\, dS + \kappa \int_\Gamma
h^{-1} (\lambda + 2\mu) (v \cdot \tilde{n})\, g\, dS
\end{split}
$$
where $g$ is the given Dirichlet data, $n$ is the unit normal, $\tilde{n}$ is
a unit vector field, and $\alpha = \pm 1$, $\kappa > 0$ are the Nitsche
parameters. The parameters $\lambda$ and $\mu$ should match the parameters
with the same names used in the bilinear form integrator,
SlidingElasticityIntegrator.
*/
class SlidingElasticityLFIntegrator : public LinearFormIntegrator
{
protected:
Coefficient *g;
VectorCoefficient *nt;
Coefficient *lambda, *mu;
real_t alpha, kappa;
#ifndef MFEM_THREAD_SAFE
Vector shape;
DenseMatrix dshape;
DenseMatrix adjJ;
DenseMatrix dshape_ps;
Vector nor;
Vector dshape_dn;
Vector dshape_du;
real_t g_val;
Vector nt_val;
#endif
public:
SlidingElasticityLFIntegrator(Coefficient &g_,
Coefficient &lambda_, Coefficient &mu_,
real_t kappa_)
: g(&g_), nt(NULL), lambda(&lambda_), mu(&mu_), alpha(-1.0), kappa(kappa_) {}
SlidingElasticityLFIntegrator(Coefficient &g_, VectorCoefficient &nt_,
Coefficient &lambda_, Coefficient &mu_,
real_t alpha_, real_t kappa_)
: g(&g_), nt(&nt_), lambda(&lambda_), mu(&mu_), alpha(alpha_), kappa(kappa_) {}
void AssembleRHSElementVect(const FiniteElement &el,
ElementTransformation &Tr,
Vector &elvect) override;
void AssembleRHSElementVect(const FiniteElement &el,
FaceElementTransformations &Tr,
Vector &elvect) override;
using LinearFormIntegrator::AssembleRHSElementVect;
};
/** Class for spatial white Gaussian noise integration.
+3 -9
View File
@@ -488,16 +488,10 @@ void ParBilinearForm::FormLinearSystem(
R.Mult(x, true_X);
FormSystemMatrix(ess_tdof_list, A);
std::unique_ptr<ConstrainedOperator> A_constrained([&]()
{
Operator *op;
Operator::FormSystemOperator(ess_tdof_list, op);
return dynamic_cast<ConstrainedOperator*>(op);
}());
MFEM_ASSERT(A_constrained != nullptr, "");
ConstrainedOperator *A_constrained;
Operator::FormConstrainedSystemOperator(ess_tdof_list, A_constrained);
A_constrained->EliminateRHS(true_X, true_B);
delete A_constrained;
R.MultTranspose(true_B, b);
hybridization->ReduceRHS(true_B, B);
X.SetSize(B.Size());
+9 -8
View File
@@ -646,38 +646,39 @@ const FaceRestriction *ParFiniteElementSpace::GetFaceRestriction(
auto itr = L2F.find(key);
if (itr != L2F.end())
{
return itr->second.get();
return itr->second;
}
else
{
std::unique_ptr<FaceRestriction> res;
FaceRestriction *res;
if (is_dg_space)
{
if (Conforming())
{
res.reset(new ParL2FaceRestriction(*this, f_ordering, type, m));
res = new ParL2FaceRestriction(*this, f_ordering, type, m);
}
else
{
res.reset(new ParNCL2FaceRestriction(*this, f_ordering, type, m));
res = new ParNCL2FaceRestriction(*this, f_ordering, type, m);
}
}
else if (dynamic_cast<const DG_Interface_FECollection*>(fec))
{
res.reset(new L2InterfaceFaceRestriction(*this, f_ordering, type));
res = new L2InterfaceFaceRestriction(*this, f_ordering, type);
}
else
{
if (Conforming())
{
res.reset(new ConformingFaceRestriction(*this, f_ordering, type));
res = new ConformingFaceRestriction(*this, f_ordering, type);
}
else
{
res.reset(new ParNCH1FaceRestriction(*this, f_ordering, type));
res = new ParNCH1FaceRestriction(*this, f_ordering, type);
}
}
return L2F.emplace(key, std::move(res)).first->second.get();
L2F[key] = res;
return res;
}
}
-2
View File
@@ -483,8 +483,6 @@ public:
const FiniteElement *GetFaceNbrFaceFE(int i) const;
const Array<HYPRE_BigInt> &GetFaceNbrGlobalDofMapArray() { return face_nbr_glob_dof_map; }
const HYPRE_BigInt *GetFaceNbrGlobalDofMap() { return face_nbr_glob_dof_map; }
const Array<HYPRE_BigInt> &GetFaceNbrGlobalDofMapArray() const
{ return face_nbr_glob_dof_map; }
ElementTransformation *GetFaceNbrElementTransformation(int i) const
{ return pmesh->GetFaceNbrElementTransformation(i); }
+7
View File
@@ -994,6 +994,7 @@ void ParNCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
{
if ( face.IsConforming() )
{
interpolations.RegisterFaceConformingInterpolation(face,f_ind);
SetFaceDofsScatterIndices1(face,f_ind);
if ( m==L2FaceValues::DoubleValued )
{
@@ -1009,6 +1010,7 @@ void ParNCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
}
else // Non-conforming face
{
interpolations.RegisterFaceCoarseToFineInterpolation(face,f_ind);
SetFaceDofsScatterIndices1(face,f_ind);
if ( m==L2FaceValues::DoubleValued )
{
@@ -1026,6 +1028,7 @@ void ParNCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
}
else if (type==FaceType::Boundary && face.IsBoundary())
{
interpolations.RegisterFaceConformingInterpolation(face,f_ind);
SetFaceDofsScatterIndices1(face,f_ind);
if ( m==L2FaceValues::DoubleValued )
{
@@ -1043,6 +1046,10 @@ void ParNCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
{
gather_offsets[i] += gather_offsets[i - 1];
}
// Transform the interpolation matrix map into a contiguous memory structure.
interpolations.LinearizeInterpolatorMapIntoVector();
interpolations.InitializeNCInterpConfig();
}
void ParNCL2FaceRestriction::ComputeGatherIndices()
+6 -2
View File
@@ -326,7 +326,9 @@ public:
@param[in] keep_nbr_block When set to true the SparseMatrix will
include the rows (in addition to the columns)
corresponding to face-neighbor dofs. The
default behavior is to disregard those rows. */
default behavior is to disregard those rows.
@warning This method is not implemented yet. */
void FillI(SparseMatrix &mat,
const bool keep_nbr_block = false) const override;
@@ -362,7 +364,9 @@ public:
@param[in] keep_nbr_block When set to true the SparseMatrix will
include the rows (in addition to the columns)
corresponding to face-neighbor dofs. The
default behavior is to disregard those rows. */
default behavior is to disregard those rows.
@warning This method is not implemented yet. */
void FillJAndData(const Vector &fea_data,
SparseMatrix &mat,
const bool keep_nbr_block = false) const override;
+1 -7
View File
@@ -50,13 +50,7 @@ QuadratureInterpolator::DetKernelType
QuadratureInterpolator::DetKernels::Fallback(
int DIM, int SDIM, int D1D, int Q1D)
{
if (DIM == 1)
{
if (SDIM == 1) { return internal::quadrature_interpolator::Det1D; }
else if (SDIM == 2) { return internal::quadrature_interpolator::Det1DSurface<0,0,2>; }
else if (SDIM == 3) { return internal::quadrature_interpolator::Det1DSurface<0,0,3>; }
else { MFEM_ABORT(""); }
}
if (DIM == 1) { return internal::quadrature_interpolator::Det1D; }
else if (DIM == 2 && SDIM == 2) { return internal::quadrature_interpolator::Det2D; }
else if (DIM == 2 && SDIM == 3) { return internal::quadrature_interpolator::Det2DSurface; }
else if (DIM == 3)
+1 -51
View File
@@ -56,50 +56,6 @@ inline void Det1D(const int NE,
});
}
template<int T_D1D = 0, int T_Q1D = 0, int T_SDIM = 3>
inline void Det1DSurface(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(b);
MFEM_CONTRACT_VAR(d_buff);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, T_SDIM, NE);
auto Y = Reshape(y, Q1D, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int q = 0; q < Q1D; q++)
{
real_t grad[T_SDIM];
for (int s = 0; s < T_SDIM; s++) { grad[s] = 0.0; }
for (int d = 0; d < D1D; d++)
{
const real_t gval = G(q, d);
for (int s = 0; s < T_SDIM; s++)
{
grad[s] += gval * X(d, s, e);
}
}
real_t norm2 = 0.0;
for (int s = 0; s < T_SDIM; s++)
{
norm2 += grad[s] * grad[s];
}
Y(q, e) = std::sqrt(norm2);
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
inline void Det2D(const int NE,
const real_t *b,
@@ -334,13 +290,7 @@ template<int DIM, int SDIM, int D1D, int Q1D>
QuadratureInterpolator::DetKernelType
QuadratureInterpolator::DetKernels::Kernel()
{
if (DIM == 1)
{
if (SDIM == 1) { return internal::quadrature_interpolator::Det1D; }
else if (SDIM == 2) { return internal::quadrature_interpolator::Det1DSurface<D1D, Q1D, 2>; }
else if (SDIM == 3) { return internal::quadrature_interpolator::Det1DSurface<D1D, Q1D, 3>; }
else { MFEM_ABORT(""); }
}
if (DIM == 1) { return internal::quadrature_interpolator::Det1D; }
else if (DIM == 2 && SDIM == 2) { return internal::quadrature_interpolator::Det2D<D1D, Q1D>; }
else if (DIM == 2 && SDIM == 3) { return internal::quadrature_interpolator::Det2DSurface<D1D, Q1D>; }
else if (DIM == 3) { return internal::quadrature_interpolator::Det3D<D1D, Q1D>; }
+1 -2
View File
@@ -542,8 +542,7 @@ void QuadratureInterpolator::Mult(const Vector &e_vec,
}
MFEM_ASSERT(!(eval_flags & DETERMINANTS) || dim == vdim ||
(dim == 2 && vdim == 3) || (dim == 1 && vdim == 2) ||
(dim == 1 && vdim == 3), "Invalid dimensions for determinants.");
(dim == 2 && vdim == 3), "Invalid dimensions for determinants.");
MFEM_ASSERT(fespace->GetMesh()->GetNumGeometries(
fespace->GetMesh()->Dimension()) == 1,
"mixed meshes are not supported");
+42 -118
View File
@@ -1506,12 +1506,12 @@ void L2FaceRestriction::EnsureNormalDerivativeRestriction() const
}
}
InterpolationManager::InterpolationManager(const FiniteElementSpace &fes_,
ElementDofOrdering ordering_,
InterpolationManager::InterpolationManager(const FiniteElementSpace &fes,
ElementDofOrdering ordering,
FaceType type)
: fes(fes_),
ordering(ordering_),
interp_config(fes.GetNFbyType(type)),
: fes(fes),
ordering(ordering),
interp_config( fes.GetNFbyType(type) ),
nc_cpt(0)
{ }
@@ -1536,8 +1536,7 @@ void InterpolationManager::RegisterFaceCoarseToFineInterpolation(
face.element[0].local_face_id +
6*face.element[1].local_face_id +
36*face.element[1].orientation ;
// Unfortunately we can't trust uniqueness of the ptMat to identify the
// transformation.
// Unfortunately we can't trust unicity of the ptMat to identify the transformation.
Key key(ptMat, face_key);
auto itr = interp_map.find(key);
if ( itr == interp_map.end() )
@@ -1584,27 +1583,17 @@ const DenseMatrix* InterpolationManager::GetCoarseToFineInterpolation(
IsoparametricTransformation isotr;
isotr.SetIdentityTransformation(trace_fe->GetGeomType());
isotr.SetPointMat(*ptMat);
DenseMatrix& trans_pt_mat = isotr.GetPointMat();
// PointMatrix needs to be flipped in 2D
if ( trace_fe->GetGeomType()==Geometry::SEGMENT && !is_ghost_slave )
{
std::swap(trans_pt_mat(0,0),trans_pt_mat(0,1));
}
DenseMatrix native_interpolator(face_dofs,face_dofs);
trace_fe->GetLocalInterpolation(isotr, native_interpolator);
if (trace_fe->GetMapType() == FiniteElement::INTEGRAL)
{
// Handle potentially inverted Jacobian matrix
isotr.SetIntPoint(&Geometries.GetCenter(trace_fe->GetGeomType()));
native_interpolator *= (isotr.Weight() >= 0) ? 1.0 : -1.0;
}
const int dim = trace_fe->GetDim()+1;
const int dof1d = trace_fe->GetOrder()+1;
int orientation_i = face.element[1].orientation;
const int orientation_j = face.element[1].orientation;
// In 2D, need to flip orientation of the segments`
if (trace_fe->GetGeomType() == Geometry::SEGMENT && !is_ghost_slave)
{
orientation_i = 1;
}
const int orientation = face.element[1].orientation;
for (int i = 0; i < face_dofs; i++)
{
const int ni = (dof_map.Size()==0) ? i : dof_map[i];
@@ -1613,7 +1602,7 @@ const DenseMatrix* InterpolationManager::GetCoarseToFineInterpolation(
{
// master side is elem 2, so we permute to order dofs as elem 1.
li = PermuteFaceL2(dim, face_id2, face_id1,
orientation_i, dof1d, li);
orientation, dof1d, li);
}
for (int j = 0; j < face_dofs; j++)
{
@@ -1622,7 +1611,7 @@ const DenseMatrix* InterpolationManager::GetCoarseToFineInterpolation(
{
// master side is elem 2, so we permute to order dofs as elem 1.
lj = PermuteFaceL2(dim, face_id2, face_id1,
orientation_j, dof1d, lj);
orientation, dof1d, lj);
}
const int nj = (dof_map.Size()==0) ? j : dof_map[j];
(*interpolator)(li,lj) = native_interpolator(ni,nj);
@@ -1687,7 +1676,7 @@ NCL2FaceRestriction::NCL2FaceRestriction(const FiniteElementSpace &fes,
const L2FaceValues m,
bool build)
: L2FaceRestriction(fes, f_ordering, type, m, false),
interpolations(fes.GetInterpolationManager(ordering, type))
interpolations(fes, f_ordering, type)
{
if (!build) { return; }
x_interp.UseDevice(true);
@@ -2213,6 +2202,14 @@ void NCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
{
PermuteAndSetFaceDofsScatterIndices2(face,f_ind);
}
if ( face.IsConforming() )
{
interpolations.RegisterFaceConformingInterpolation(face,f_ind);
}
else // Non-conforming face
{
interpolations.RegisterFaceCoarseToFineInterpolation(face,f_ind);
}
f_ind++;
}
else if ( type==FaceType::Boundary && face.IsBoundary() )
@@ -2222,6 +2219,7 @@ void NCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
{
SetBoundaryDofsScatterIndices2(face,f_ind);
}
interpolations.RegisterFaceConformingInterpolation(face,f_ind);
f_ind++;
}
}
@@ -2234,6 +2232,10 @@ void NCL2FaceRestriction::ComputeScatterIndicesAndOffsets()
{
gather_offsets[i] += gather_offsets[i - 1];
}
// Transform the interpolation matrix map into a contiguous memory structure.
interpolations.LinearizeInterpolatorMapIntoVector();
interpolations.InitializeNCInterpConfig();
}
void NCL2FaceRestriction::ComputeGatherIndices()
@@ -2276,18 +2278,6 @@ void NCL2FaceRestriction::ComputeGatherIndices()
gather_offsets[0] = 0;
}
static int GetSharedVSize(const FiniteElementSpace &fes)
{
#ifdef MFEM_USE_MPI
if (auto pfes = dynamic_cast<const ParFiniteElementSpace*>(&fes))
{
const_cast<ParFiniteElementSpace*>(pfes)->ExchangeFaceNbrData();
return pfes->GetFaceNbrVSize();
}
#endif
return 0;
}
L2InterfaceFaceRestriction::L2InterfaceFaceRestriction(
const FiniteElementSpace& fes_,
const ElementDofOrdering ordering_,
@@ -2298,54 +2288,25 @@ L2InterfaceFaceRestriction::L2InterfaceFaceRestriction(
nfaces(fes.GetNFbyType(type)),
vdim(fes.GetVDim()),
byvdim(fes.GetOrdering() == Ordering::byVDIM),
face_dofs(fes.GetTypicalTraceElement()->GetDof()),
face_dofs(nfaces > 0 ? fes.GetFaceElement(0)->GetDof() : 0),
nfdofs(face_dofs*nfaces),
ndofs(fes.GetNDofs()),
nsdofs(GetSharedVSize(fes))
ndofs(fes.GetNDofs())
{
height = nfdofs;
width = ndofs;
#ifdef MFEM_USE_MPI
auto pfes = dynamic_cast<const ParFiniteElementSpace*>(&fes);
#endif
const Table &face2dof = fes.GetFaceToDofTable();
const Mesh &mesh = *fes.GetMesh();
int face_idx = 0;
scatter_map.SetSize(nfdofs);
gather_map.SetSize(ndofs + nsdofs);
gather_map = -1;
Array<int> dofs;
for (int f = 0; f < mesh.GetNumFacesWithGhost(); ++f)
gather_map.SetSize(nfdofs);
for (int f = 0; f < mesh.GetNumFaces(); ++f)
{
Mesh::FaceInformation face = mesh.GetFaceInformation(f);
if (!face.IsOfFaceType(type) || face.IsNonconformingCoarse()) { continue; }
if (f < mesh.GetNumFaces())
if (!face.IsOfFaceType(type)) { continue; }
for (int i = 0; i < face_dofs; ++i)
{
// Local face
face2dof.GetRow(f, dofs);
for (int i = 0; i < face_dofs; ++i)
{
scatter_map[i + face_idx*face_dofs] = dofs[i];
gather_map[dofs[i]] = i + face_idx*face_dofs;
}
}
else
{
// Shared (non-conforming) ghost face
#ifdef MFEM_USE_MPI
MFEM_ASSERT(pfes != nullptr, "");
pfes->GetFaceNbrFaceVDofs(f, dofs);
for (int i = 0; i < face_dofs; ++i)
{
scatter_map[i + face_idx*face_dofs] = ndofs + dofs[i];
gather_map[ndofs + dofs[i]] = i + face_idx*face_dofs;
}
#endif
gather_map[i + face_idx*face_dofs] = face2dof.GetJ()[i + f*face_dofs];
}
++face_idx;
}
@@ -2353,19 +2314,13 @@ L2InterfaceFaceRestriction::L2InterfaceFaceRestriction(
void L2InterfaceFaceRestriction::Mult(const Vector &x, Vector &y) const
{
const int NDOFS = ndofs;
const int nd = face_dofs;
const int nf = nfaces;
const int vd = vdim;
const bool t = byvdim;
const int *map = scatter_map.Read();
Vector face_nbr_data = GetLVectorFaceNbrData(fes, x, type);
MFEM_ASSERT(face_nbr_data.Size() / vd == nsdofs, "");
const int *map = gather_map.Read();
const auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
const auto d_x_shared = Reshape(face_nbr_data.Read(),
t?vd:nsdofs, t?nsdofs:vd);
auto d_y = Reshape(y.Write(), nd, vd, nf);
mfem::forall(nd*nf, [=] MFEM_HOST_DEVICE (int i)
@@ -2373,8 +2328,7 @@ void L2InterfaceFaceRestriction::Mult(const Vector &x, Vector &y) const
const int j = map[i];
for (int c = 0; c < vd; ++c)
{
if (j < NDOFS) { d_y(i % nd, c, i / nd) = d_x(t?c:j, t?j:c); }
else { d_y(i % nd, c, i / nd) = d_x_shared(t?c:(j-NDOFS), t?(j-NDOFS):c); }
d_y(i % nd, c, i / nd) = d_x(t?c:j, t?j:c);
}
});
}
@@ -2389,39 +2343,15 @@ void L2InterfaceFaceRestriction::AddMultTranspose(
const int *map = gather_map.Read();
const auto d_x = Reshape(x.Read(), nd, vd, nf);
auto d_y = Reshape(y.ReadWrite(), t?vd:ndofs, t?ndofs:vd);
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
mfem::forall(ndofs, [=] MFEM_HOST_DEVICE (int i)
mfem::forall(ndofs, [=] MFEM_HOST_DEVICE (int i) { d_y[i] = 0.0; });
mfem::forall(nd*nf, [=] MFEM_HOST_DEVICE (int i)
{
const int j = map[i];
if (j < 0) { return; }
for (int c = 0; c < vd; ++c)
{
d_y(t?c:i, t?i:c) += a*d_x(j % nd, c, j / nd);
}
});
}
void L2InterfaceFaceRestriction::MultTransposeShared(
const Vector &x, Vector &y) const
{
const int nd = face_dofs;
const int nf = nfaces;
const int vd = vdim;
const bool t = byvdim;
const int *map = gather_map.Read();
const auto d_x = Reshape(x.Read(), nd, vd, nf);
auto d_y = Reshape(y.Write(), t?vd:(ndofs+nsdofs), t?(ndofs+nsdofs):vd);
y = 0.0;
mfem::forall(ndofs + nsdofs, [=] MFEM_HOST_DEVICE (int i)
{
const int j = map[i];
if (j < 0) { return; }
for (int c = 0; c < vd; ++c)
{
d_y(t?c:i, t?i:c) = d_x(j % nd, c, j / nd);
d_y(t?c:j, t?j:c) = d_x(i % nd, c, i / nd);
}
});
}
@@ -2431,11 +2361,6 @@ const Array<int> &L2InterfaceFaceRestriction::GatherMap() const
return gather_map;
}
const Array<int> &L2InterfaceFaceRestriction::ScatterMap() const
{
return scatter_map;
}
Vector GetLVectorFaceNbrData(
const FiniteElementSpace &fes, const Vector &x, FaceType ftype)
{
@@ -2457,7 +2382,6 @@ Vector GetLVectorFaceNbrData(
{
ParGridFunction gf(pfes, const_cast<Vector&>(x));
gf.ExchangeFaceNbrData();
x.SyncMemory(gf);
return std::move(gf.FaceNbrData());
}
}
+14 -26
View File
@@ -812,12 +812,13 @@ protected:
PointMatrix and a local face identifier. */
using Key = std::pair<const DenseMatrix*,int>;
/// The temporary map used to store the different interpolators.
using Map =
std::unordered_map<Key, std::pair<int,const DenseMatrix*>, PairHasher>;
using Map = std::map<Key, std::pair<int,const DenseMatrix*>>;
Map interp_map; // The temporary map that stores the interpolators.
public:
/** @brief Constructor.
InterpolationManager() = delete;
/** @brief main constructor.
@param[in] fes The FiniteElementSpace on which this operates
@param[in] ordering Request a specific element ordering.
@@ -908,7 +909,7 @@ private:
class NCL2FaceRestriction : virtual public L2FaceRestriction
{
protected:
const InterpolationManager &interpolations;
InterpolationManager interpolations;
mutable Vector x_interp;
/** @brief Constructs an NCL2FaceRestriction, this is a specialization of a
@@ -995,7 +996,9 @@ public:
@param[in] keep_nbr_block When set to true the SparseMatrix will
include the rows (in addition to the columns)
corresponding to face-neighbor dofs. The
default behavior is to disregard those rows. */
default behavior is to disregard those rows.
@warning This method is not implemented yet. */
void FillI(SparseMatrix &mat,
const bool keep_nbr_block = false) const override;
@@ -1013,7 +1016,9 @@ public:
@param[in] keep_nbr_block When set to true the SparseMatrix will
include the rows (in addition to the columns)
corresponding to face-neighbor dofs. The
default behavior is to disregard those rows. */
default behavior is to disregard those rows.
@warning This method is not implemented yet. */
void FillJAndData(const Vector &fea_data,
SparseMatrix &mat,
const bool keep_nbr_block = false) const override;
@@ -1031,7 +1036,9 @@ public:
added the face contributions.
The format is: dofs x dofs x ne, where dofs is the
number of dofs per element and ne the number of
elements. */
elements.
@warning This method is not implemented yet. */
void AddFaceMatricesToElementMatrices(const Vector &fea_data,
Vector &ea_data) const override;
@@ -1123,9 +1130,7 @@ protected:
const int face_dofs; ///< Number of dofs on each face
const int nfdofs; ///< Total number of dofs on the faces (E-vector size)
const int ndofs; ///< Number of dofs in the space (L-vector size)
const int nsdofs; ///< Number of shared face neighbor (ghost) dofs
Array<int> gather_map; ///< Gather map
Array<int> scatter_map; ///< Scatter map
public:
/** @brief Constructs an L2InterfaceFaceRestriction.
@@ -1163,24 +1168,7 @@ public:
void AddMultTranspose(const Vector &x, Vector &y,
const real_t a = 1.0) const override;
/// @brief Gather degrees of freedom, from face E-vector to L-vector and
/// shared (ghost) DOFs.
///
/// @param[in] x The face E-Vector degrees of freedom with size
/// (face_dofs, vdim, nf), where nf is the number of
/// interior or boundary faces requested by @a type in the
/// constructor. The face_dofs should be ordered according
/// to the given ElementDofOrdering
/// @param[out] y Vector of length vsize + face neighbor vsize
void MultTransposeShared(const Vector &x, Vector &y) const;
const Array<int> &GatherMap() const override;
/// @brief Return the low-level mapping from L-dofs to E-dofs.
///
/// L-dofs that do not correspond to an E-dof (e.g. that lie on a face of a
/// different type) are given index -1.
const Array<int> &ScatterMap() const;
};
/** @brief Convert a dof face index from Native ordering to lexicographic
+12 -31
View File
@@ -333,12 +333,6 @@ void L2ProjectionGridTransfer::L2Projection::MixedMassEA(
int nel_ho = mesh_ho->GetNE();
int nel_lor = mesh_lor->GetNE();
if (nel_ho == 0)
{
M_LH.SetSize(0);
return;
}
const CoarseFineTransformations& cf_tr = mesh_lor->GetRefinementTransforms();
int nref_max = 0;
@@ -837,17 +831,11 @@ void L2ProjectionGridTransfer::L2ProjectionL2Space::Mult(
void L2ProjectionGridTransfer::L2ProjectionL2Space::EAMult(
const Vector &x, Vector &y) const
{
const int nel_ho = fes_ho.GetMesh()->GetNE();
if (nel_ho == 0)
{
return;
}
const int iho = 0;
const int nref = ho2lor.RowSize(iho);
const int ndof_ho = fes_ho.GetFE(iho)->GetDof();
const int ndof_lor = fes_lor.GetFE(ho2lor.GetRow(iho)[0])->GetDof();
const int nel_ho = fes_ho.GetMesh()->GetNE();
DenseTensor R_dt;
R_dt.NewMemoryAndSize(R.GetMemory(), ndof_lor*nref, ndof_ho, nel_ho, false);
@@ -899,17 +887,11 @@ void L2ProjectionGridTransfer::L2ProjectionL2Space::MultTranspose(
void L2ProjectionGridTransfer::L2ProjectionL2Space::EAMultTranspose(
const Vector &x, Vector &y) const
{
const int nel_ho = fes_ho.GetMesh()->GetNE();
if (nel_ho == 0)
{
return;
}
const int iho = 0;
const int nref = ho2lor.RowSize(iho);
const int ndof_ho = fes_ho.GetFE(iho)->GetDof();
const int ndof_lor = fes_lor.GetFE(ho2lor.GetRow(iho)[0])->GetDof();
const int nel_ho = fes_ho.GetMesh()->GetNE();
DenseTensor R_dt;
R_dt.NewMemoryAndSize(R.GetMemory(), ndof_lor*nref, ndof_ho, nel_ho, false);
@@ -919,6 +901,7 @@ void L2ProjectionGridTransfer::L2ProjectionL2Space::EAMultTranspose(
void L2ProjectionGridTransfer::L2ProjectionL2Space::Prolongate(
const Vector &x, Vector &y) const
{
if (fes_ho.GetNE() == 0) { return; }
if (use_ea)
@@ -977,13 +960,14 @@ void L2ProjectionGridTransfer::L2ProjectionL2Space::EAProlongate(
void L2ProjectionGridTransfer::L2ProjectionL2Space::ProlongateTranspose(
const Vector &x, Vector &y) const
{
if (fes_ho.GetNE() == 0) { return; }
if (use_ea)
{
return EAProlongateTranspose(x,y);
}
if (fes_ho.GetNE() == 0) { return; }
MFEM_VERIFY(P.Size() > 0, "Prolongation not supported for these spaces.")
int vdim = fes_ho.GetVDim();
Array<int> vdofs;
@@ -1260,6 +1244,13 @@ void L2ProjectionGridTransfer::L2ProjectionH1Space::EAL2ProjectionH1Space
int ndof_ho = pfes_ho.GetNDofs();
int ndof_lor = pfes_lor.GetNDofs();
// If the local mesh is empty, skip all computations
if (nel_ho == 0)
{
return;
}
const CoarseFineTransformations& cf_tr = mesh_lor->GetRefinementTransforms();
int nref_max = 0;
@@ -1869,11 +1860,6 @@ L2ProjectionGridTransfer::H1SpaceMixedMassOperator::H1SpaceMixedMassOperator(
void L2ProjectionGridTransfer::H1SpaceMixedMassOperator::Mult(const Vector &x,
Vector &y) const
{
if (fes_ho->GetNE() == 0)
{
return;
}
const Operator* elem_restrict_ho = fes_ho->GetElementRestriction(
ElementDofOrdering::NATIVE);
const Operator* elem_restrict_lor = fes_lor->GetElementRestriction(
@@ -1920,11 +1906,6 @@ void L2ProjectionGridTransfer::H1SpaceMixedMassOperator::Mult(const Vector &x,
void L2ProjectionGridTransfer::H1SpaceMixedMassOperator::MultTranspose(
const Vector &x, Vector &y) const
{
if (fes_ho->GetNE() == 0)
{
return;
}
const Operator* elem_restrict_ho = fes_ho->GetElementRestriction(
ElementDofOrdering::NATIVE);
const Operator* elem_restrict_lor = fes_lor->GetElementRestriction(
-2
View File
@@ -18,7 +18,6 @@ list(APPEND SRCS
gecko.cpp
globals.cpp
hash.cpp
hash_util.cpp
isockstream.cpp
mem_manager.cpp
occa.cpp
@@ -47,7 +46,6 @@ list(APPEND HDRS
globals.hpp
zstr.hpp
hash.hpp
hash_util.hpp
isockstream.hpp
kdtree.hpp
mem_alloc.hpp
-2
View File
@@ -44,7 +44,6 @@
#endif
#if !defined(MFEM_USE_CUDA_OR_HIP)
constexpr bool mfem_use_gpu = false;
#define MFEM_DEVICE
#define MFEM_HOST
#define MFEM_LAMBDA
@@ -53,7 +52,6 @@ constexpr bool mfem_use_gpu = false;
#define MFEM_DEVICE_SYNC
// MFEM_STREAM_SYNC is used for UVM and MPI GPU-Aware kernels
#define MFEM_STREAM_SYNC
#define MFEM_LAUNCH_BOUNDS(...)
#endif
#if !((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
-2
View File
@@ -20,11 +20,9 @@
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#define MFEM_USE_CUDA_OR_HIP
constexpr bool mfem_use_gpu = true;
#define MFEM_DEVICE __device__
#define MFEM_HOST __host__
#define MFEM_LAMBDA __host__
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(cudaDeviceSynchronize())
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(cudaStreamSynchronize(0))
+40 -203
View File
@@ -295,12 +295,11 @@ using hip_threads_z =
#endif
#if defined(MFEM_USE_RAJA) && defined(RAJA_ENABLE_CUDA) && defined(__CUDACC__)
template <typename DBODY>
template <const int BLOCKS = MFEM_CUDA_BLOCKS, typename DBODY>
void RajaCuWrap1D(const int N, DBODY &&d_body)
{
//true denotes asynchronous kernel
RAJA::forall<RAJA::cuda_exec<MFEM_CUDA_BLOCKS,true>>(RAJA::RangeSegment(0,N),
d_body);
RAJA::forall<RAJA::cuda_exec<BLOCKS,true>>(RAJA::RangeSegment(0,N),d_body);
}
template <typename DBODY>
@@ -363,18 +362,18 @@ struct RajaCuWrap;
template <>
struct RajaCuWrap<1>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
RajaCuWrap1D(N, d_body);
RajaCuWrap1D<BLCK>(N, d_body);
}
};
template <>
struct RajaCuWrap<2>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
@@ -385,7 +384,7 @@ struct RajaCuWrap<2>
template <>
struct RajaCuWrap<3>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
@@ -396,12 +395,11 @@ struct RajaCuWrap<3>
#endif
#if defined(MFEM_USE_RAJA) && defined(RAJA_ENABLE_HIP) && defined(__HIP__)
template <typename DBODY>
template <const int BLOCKS = MFEM_HIP_BLOCKS, typename DBODY>
void RajaHipWrap1D(const int N, DBODY &&d_body)
{
//true denotes asynchronous kernel
RAJA::forall<RAJA::hip_exec<MFEM_HIP_BLOCKS,true>>(RAJA::RangeSegment(0,N),
d_body);
RAJA::forall<RAJA::hip_exec<BLOCKS,true>>(RAJA::RangeSegment(0,N),d_body);
}
template <typename DBODY>
@@ -464,18 +462,18 @@ struct RajaHipWrap;
template <>
struct RajaHipWrap<1>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
RajaHipWrap1D(N, d_body);
RajaHipWrap1D<BLCK>(N, d_body);
}
};
template <>
struct RajaHipWrap<2>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
@@ -486,7 +484,7 @@ struct RajaHipWrap<2>
template <>
struct RajaHipWrap<3>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
@@ -586,31 +584,12 @@ void CuKernel2D(const int N, BODY body)
body(k);
}
// __launch_bounds__ second argument is omitted to get the default behavior
template <int MAX_THREADS_PER_BLOCK, typename BODY>
__global__
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
static void CuKernel2DLaunchBounds(const int N, BODY body)
{
const int k = blockIdx.x*blockDim.z + threadIdx.z;
if (k >= N) { return; }
body(k);
}
template <typename BODY> __global__ static
void CuKernel3D(const int N, BODY body)
{
for (int k = blockIdx.x; k < N; k += gridDim.x) { body(k); }
}
template <int MAX_THREADS_PER_BLOCK, typename BODY>
__global__
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
static void CuKernel3DLaunchBounds(const int N, BODY body)
{
for (int k = blockIdx.x; k < N; k += gridDim.x) { body(k); }
}
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
void CuWrap1D(const int N, DBODY &&d_body)
{
@@ -625,8 +604,6 @@ void CuWrap2D(const int N, DBODY &&d_body,
const int X, const int Y, const int BZ)
{
if (N==0) { return; }
// required for optimized GCC/NVCC builds to prevent runtime
// ODR/linkage violations of inlined templated kernel helpers
MFEM_VERIFY(BZ>0, "");
const int GRID = (N+BZ-1)/BZ;
const dim3 BLCK(X,Y,BZ);
@@ -634,19 +611,6 @@ void CuWrap2D(const int N, DBODY &&d_body,
MFEM_GPU_CHECK(cudaGetLastError());
}
template <int MAX_THREADS_PER_BLOCK, typename DBODY>
void CuWrap2DLaunchBounds(const int N, DBODY &&d_body,
const int X, const int Y, const int BZ)
{
if (N==0) { return; }
MFEM_VERIFY(BZ>0, "");
const int GRID = (N+BZ-1)/BZ;
const dim3 BLCK(X,Y,BZ);
static_assert(MAX_THREADS_PER_BLOCK > 0);
CuKernel2DLaunchBounds<MAX_THREADS_PER_BLOCK><<<GRID,BLCK>>>(N, d_body);
MFEM_GPU_CHECK(cudaGetLastError());
}
template <typename DBODY>
void CuWrap3D(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
@@ -658,35 +622,24 @@ void CuWrap3D(const int N, DBODY &&d_body,
MFEM_GPU_CHECK(cudaGetLastError());
}
template <int MAX_THREADS_PER_BLOCK, typename DBODY>
void CuWrap3DLaunchBounds(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
if (N==0) { return; }
const int GRID = G == 0 ? N : G;
const dim3 BLCK(X,Y,Z);
static_assert(MAX_THREADS_PER_BLOCK > 0);
CuKernel3DLaunchBounds<MAX_THREADS_PER_BLOCK><<<GRID, BLCK>>>(N, d_body);
MFEM_GPU_CHECK(cudaGetLastError());
}
template <int Dim>
struct CuWrap;
template <int Dim, int MAX_THREADS_PER_BLOCK> struct CuWrap;
template <int MAX_THREADS_PER_BLOCK>
struct CuWrap<1, MAX_THREADS_PER_BLOCK>
template <>
struct CuWrap<1>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
CuWrap1D<MFEM_CUDA_BLOCKS>(N, d_body);
CuWrap1D<BLCK>(N, d_body);
}
};
template <>
struct CuWrap<2, 0>
struct CuWrap<2>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
@@ -694,22 +647,10 @@ struct CuWrap<2, 0>
}
};
template <int MAX_THREADS_PER_BLOCK>
struct CuWrap<2, MAX_THREADS_PER_BLOCK>
{
template <typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
static_assert(MAX_THREADS_PER_BLOCK > 0);
CuWrap2DLaunchBounds<MAX_THREADS_PER_BLOCK>(N, d_body, X, Y, Z);
}
};
template <>
struct CuWrap<3, 0>
struct CuWrap<3>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
@@ -717,17 +658,6 @@ struct CuWrap<3, 0>
}
};
template <int MAX_THREADS_PER_BLOCK>
struct CuWrap<3, MAX_THREADS_PER_BLOCK>
{
template <typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
CuWrap3DLaunchBounds<MAX_THREADS_PER_BLOCK>(N, d_body, X, Y, Z, G);
}
};
#endif // defined(MFEM_USE_CUDA) && defined(__CUDACC__)
@@ -750,31 +680,13 @@ void HipKernel2D(const int N, BODY body)
body(k);
}
template <int MAX_THREADS_PER_BLOCK, typename BODY>
__global__
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
static void HipKernel2DLaunchBounds(const int N, BODY body)
{
const int k = hipBlockIdx_x*hipBlockDim_z + hipThreadIdx_z;
if (k >= N) { return; }
body(k);
}
template <typename BODY> __global__ static
void HipKernel3D(const int N, BODY body)
{
for (int k = hipBlockIdx_x; k < N; k += hipGridDim_x) { body(k); }
}
template <int MAX_THREADS_PER_BLOCK, typename BODY>
__global__
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
static void HipKernel3DLaunchBounds(const int N, BODY body)
{
for (int k = hipBlockIdx_x; k < N; k += hipGridDim_x) { body(k); }
}
template <int BLCK = MFEM_HIP_BLOCKS, typename DBODY>
template <const int BLCK = MFEM_HIP_BLOCKS, typename DBODY>
void HipWrap1D(const int N, DBODY &&d_body)
{
if (N==0) { return; }
@@ -788,27 +700,12 @@ void HipWrap2D(const int N, DBODY &&d_body,
const int X, const int Y, const int BZ)
{
if (N==0) { return; }
MFEM_VERIFY(BZ>0, "");
const int GRID = (N+BZ-1)/BZ;
const dim3 BLCK(X,Y,BZ);
hipLaunchKernelGGL(HipKernel2D,GRID,BLCK,0,nullptr,N,d_body);
MFEM_GPU_CHECK(hipGetLastError());
}
template <int MAX_THREADS_PER_BLOCK, typename DBODY>
void HipWrap2DLaunchBounds(const int N, DBODY &&d_body,
const int X, const int Y, const int BZ)
{
if (N==0) { return; }
MFEM_VERIFY(BZ>0, "");
const int GRID = (N+BZ-1)/BZ;
const dim3 BLCK(X,Y,BZ);
static_assert(MAX_THREADS_PER_BLOCK > 0);
HipKernel2DLaunchBounds<MAX_THREADS_PER_BLOCK><<<dim3(GRID), dim3(BLCK), 0, 0>>>
(N, d_body);
MFEM_GPU_CHECK(hipGetLastError());
}
template <typename DBODY>
void HipWrap3D(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
@@ -820,36 +717,24 @@ void HipWrap3D(const int N, DBODY &&d_body,
MFEM_GPU_CHECK(hipGetLastError());
}
template <int MAX_THREADS_PER_BLOCK, typename DBODY>
void HipWrap3DLaunchBounds(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
if (N==0) { return; }
const int GRID = G == 0 ? N : G;
const dim3 BLCK(X,Y,Z);
static_assert(MAX_THREADS_PER_BLOCK > 0);
HipKernel3DLaunchBounds<MAX_THREADS_PER_BLOCK><<<dim3(GRID), dim3(BLCK), 0, 0>>>
(N, d_body);
MFEM_GPU_CHECK(hipGetLastError());
}
template <int Dim>
struct HipWrap;
template <int Dim, int MAX_THREADS_PER_BLOCK> struct HipWrap;
template <int MAX_THREADS_PER_BLOCK>
struct HipWrap<1, MAX_THREADS_PER_BLOCK>
template <>
struct HipWrap<1>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
HipWrap1D<MFEM_HIP_BLOCKS>(N, d_body);
HipWrap1D<BLCK>(N, d_body);
}
};
template <>
struct HipWrap<2, 0>
struct HipWrap<2>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
@@ -857,21 +742,10 @@ struct HipWrap<2, 0>
}
};
template <int MAX_THREADS_PER_BLOCK>
struct HipWrap<2, MAX_THREADS_PER_BLOCK>
{
template <typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
HipWrap2DLaunchBounds<MAX_THREADS_PER_BLOCK>(N, d_body, X, Y, Z);
}
};
template <>
struct HipWrap<3, 0>
struct HipWrap<3>
{
template <typename DBODY>
template <const int BLCK = MFEM_CUDA_BLOCKS, typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
@@ -879,24 +753,11 @@ struct HipWrap<3, 0>
}
};
template <int MAX_THREADS_PER_BLOCK>
struct HipWrap<3, MAX_THREADS_PER_BLOCK>
{
template <typename DBODY>
static void run(const int N, DBODY &&d_body,
const int X, const int Y, const int Z, const int G)
{
HipWrap3DLaunchBounds<MAX_THREADS_PER_BLOCK>(N, d_body, X, Y, Z, G);
}
};
#endif // defined(MFEM_USE_HIP) && defined(__HIP__)
///////////////////////////////////////////////////////////////////////////////
/// Forall host & device kernel dispatch
template <int DIM, int MAX_THREADS_PER_BLOCK = 0,
typename d_lambda, typename h_lambda>
/// The forall kernel body wrapper
template <const int DIM, typename d_lambda, typename h_lambda>
inline void ForallWrap(const bool use_dev, const int N,
d_lambda &&d_body, h_lambda &&h_body,
const int X=0, const int Y=0, const int Z=0,
@@ -929,7 +790,7 @@ inline void ForallWrap(const bool use_dev, const int N,
// If Backend::CUDA is allowed, use it
if (Device::Allows(Backend::CUDA))
{
return CuWrap<DIM, MAX_THREADS_PER_BLOCK>::run(N, d_body, X, Y, Z, G);
return CuWrap<DIM>::run(N, d_body, X, Y, Z, G);
}
#endif
@@ -937,7 +798,7 @@ inline void ForallWrap(const bool use_dev, const int N,
// If Backend::HIP is allowed, use it
if (Device::Allows(Backend::HIP))
{
return HipWrap<DIM, MAX_THREADS_PER_BLOCK>::run(N, d_body, X, Y, Z, G);
return HipWrap<DIM>::run(N, d_body, X, Y, Z, G);
}
#endif
@@ -966,9 +827,7 @@ backend_cpu:
for (int k = 0; k < N; k++) { h_body(k); }
}
///////////////////////////////////////////////////////////////////////////////
/// Forall host & device kernel wrappers
template <int DIM, typename lambda>
template <const int DIM, typename lambda>
inline void ForallWrap(const bool use_dev, const int N, lambda &&body,
const int X=0, const int Y=0, const int Z=0,
const int G=0)
@@ -976,16 +835,6 @@ inline void ForallWrap(const bool use_dev, const int N, lambda &&body,
ForallWrap<DIM>(use_dev, N, body, body, X, Y, Z, G);
}
template <int DIM, int MAX_THREADS_PER_BLOCK, typename lambda>
inline void ForallWrap(const bool use_dev, const int N, lambda &&body,
const int X=0, const int Y=0, const int Z=0,
const int G=0)
{
ForallWrap<DIM, MAX_THREADS_PER_BLOCK>(use_dev, N, body, body, X, Y, Z, G);
}
///////////////////////////////////////////////////////////////////////////////
// forall interfaces
template<typename lambda>
inline void forall(int N, lambda &&body) { ForallWrap<1>(true, N, body); }
@@ -994,7 +843,7 @@ inline void forall(int Nx, int Ny, lambda &&body)
{
if (Device::Allows(Backend::DEVICE_MASK))
{
mfem::forall(Nx * Ny, [=] MFEM_HOST_DEVICE(int idx)
forall(Nx * Ny, [=] MFEM_HOST_DEVICE(int idx)
{
int j = idx / Nx;
int i = idx % Nx;
@@ -1030,7 +879,7 @@ inline void forall(int Nx, int Ny, int Nz, lambda &&body)
{
if (Device::Allows(Backend::DEVICE_MASK))
{
mfem::forall(Nx * Ny * Nz, [=] MFEM_HOST_DEVICE(int idx)
forall(Nx * Ny * Nz, [=] MFEM_HOST_DEVICE(int idx)
{
int i = idx % Nx;
int j = idx / Nx;
@@ -1078,12 +927,6 @@ inline void forall_2D(int N, int X, int Y, lambda &&body)
ForallWrap<2>(true, N, body, X, Y, 1);
}
template<int MAX_THREADS_PER_BLOCK, typename lambda>
inline void forall_2D(int N, int X, int Y, lambda &&body)
{
ForallWrap<2, MAX_THREADS_PER_BLOCK>(true, N, body, X, Y, 1);
}
template<typename lambda>
inline void forall_2D_batch(int N, int X, int Y, int BZ, lambda &&body)
{
@@ -1096,12 +939,6 @@ inline void forall_3D(int N, int X, int Y, int Z, lambda &&body)
ForallWrap<3>(true, N, body, X, Y, Z, 0);
}
template<int MAX_THREADS_PER_BLOCK, typename lambda>
inline void forall_3D(int N, int X, int Y, int Z, lambda &&body)
{
ForallWrap<3, MAX_THREADS_PER_BLOCK>(true, N, body, X, Y, Z, 0);
}
template<typename lambda>
inline void forall_3D_grid(int N, int X, int Y, int Z, int G, lambda &&body)
{
+155
View File
@@ -80,4 +80,159 @@ std::string HashFunction::GetHash() const
return hash;
}
constexpr static uint64_t rotl64(uint64_t x, int r)
{
return (x << r) | (x >> (64 - r));
}
void Hasher::init(uint64_t seed)
{
data[0] = seed;
data[1] = seed;
nbytes = 0;
}
void Hasher::add_block(uint64_t k1, uint64_t k2)
{
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
constexpr uint64_t c2 = 0x4cf5ad432745937full;
k1 *= c1;
k1 = rotl64(k1, 31);
k1 *= c2;
data[0] ^= k1;
data[0] = rotl64(data[0], 27);
data[0] += data[1];
data[0] = data[0] * 5 + 0x52dce729ull;
k2 *= c2;
k2 = rotl64(k2, 33);
k2 *= c1;
data[1] ^= k2;
data[1] = rotl64(data[1], 31);
data[1] += data[0];
data[1] = data[1] * 5 + 0x38495ab5ull;
}
static uint64_t fmix64(uint64_t k)
{
// http://zimbry.blogspot.com/2011/09/better-bit-mixing-improving-on.html
// mix13
k ^= k >> 30;
k *= 0xbf58476d1ce4e5b9ull;
k ^= k >> 27;
k *= 0x94d049bb133111ebull;
k ^= k >> 31;
return k;
}
void Hasher::append(const uint8_t *vs, uint64_t bytes)
{
if (bytes == 0)
{
return;
}
auto rem = nbytes % 16;
nbytes += bytes;
uint8_t *tmp = reinterpret_cast<uint8_t *>(buf_);
while (true)
{
if (bytes + rem >= 16)
{
std::copy(vs, vs + 16 - rem, tmp + rem);
add_block(buf_[0], buf_[1]);
vs += (16 - rem);
bytes -= (16 - rem);
rem = 0;
}
else
{
std::copy(vs, vs + bytes, tmp + rem);
return;
}
}
}
void Hasher::finalize()
{
auto rem = nbytes % 16;
if (rem > 0)
{
nbytes -= rem;
if (rem <= 8)
{
finalize(buf_[0], rem);
}
else
{
finalize(buf_[0], buf_[1], rem);
}
return;
}
data[0] ^= nbytes;
data[1] ^= nbytes;
data[0] += data[1];
data[1] += data[0];
data[0] = fmix64(data[0]);
data[1] = fmix64(data[1]);
data[0] += data[1];
data[1] += data[0];
}
void Hasher::finalize(uint64_t k1, int num)
{
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
constexpr uint64_t c2 = 0x4cf5ad432745937full;
nbytes += num;
k1 *= c1;
k1 = rotl64(k1, 31);
k1 *= c2;
data[0] ^= k1;
data[0] ^= nbytes;
data[1] ^= nbytes;
data[0] += data[1];
data[1] += data[0];
data[0] = fmix64(data[0]);
data[1] = fmix64(data[1]);
data[0] += data[1];
data[1] += data[0];
}
void Hasher::finalize(uint64_t k1, uint64_t k2, int num)
{
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
constexpr uint64_t c2 = 0x4cf5ad432745937full;
nbytes += num;
k2 *= c2;
k2 = rotl64(k2, 33);
k2 *= c1;
data[1] ^= k2;
k1 *= c1;
k1 = rotl64(k1, 31);
k1 *= c2;
data[0] ^= k1;
data[0] ^= nbytes;
data[1] ^= nbytes;
data[0] += data[1];
data[1] += data[0];
data[0] = fmix64(data[0]);
data[1] = fmix64(data[1]);
data[0] += data[1];
data[1] += data[0];
}
} // namespace mfem
+70 -1
View File
@@ -15,8 +15,8 @@
#include "../config/config.hpp"
#include "array.hpp"
#include "globals.hpp"
#include "hash_util.hpp"
#include <array>
#include <cstdint>
#include <type_traits>
#include <utility>
@@ -457,6 +457,75 @@ protected:
int BinSize(int idx) const;
};
///
/// @brief streaming implementation for murmurhash3 128 (x64).
/// Constructs the hash in 3 stages: init, append, finalize.
///
struct Hasher
{
/// where the final hash result is stored after finalize. Use data[1] when
/// only 64 bits are required.
uint64_t data[2] = {0, 0};
private:
uint64_t nbytes = 0;
uint64_t buf_[2] = {0, 0};
public:
/// resets this hasher back to an initial seed
void init(uint64_t seed = 0);
void append(const uint8_t *vs, uint64_t bytes);
void finalize();
private:
// add 16 bytes
void add_block(uint64_t k1, uint64_t k2);
// add [1-8] more bytes, then finalize
void finalize(uint64_t k1, int num);
// add [1-15] more bytes, then finalize
// 0 < num < 16
void finalize(uint64_t k1, uint64_t k2, int num);
};
/// Helper class for hashing std::pair. Usable in place of std::hash<std::pair<T,U>>
struct PairHasher
{
template <class T, class V>
size_t operator()(const std::pair<T, V> &v) const noexcept
{
Hasher hash;
// chosen randomly with a 2^64-sided dice
hash.init(0xfebd1fe69813c14full);
hash.append(reinterpret_cast<const uint8_t *>(&v.first), sizeof(T));
hash.append(reinterpret_cast<const uint8_t *>(&v.second), sizeof(V));
hash.finalize();
return hash.data[1];
}
};
/// Helper class for hashing std::array. Usable in place of std::hash<std::array<T,N>>
struct ArrayHasher
{
template <class T, size_t N>
size_t operator()(const std::array<T, N> &v) const noexcept
{
Hasher hash;
// chosen randomly with a 2^64-sided dice
hash.init(0xfebd1fe69813c14full);
for (size_t i = 0; i < N; ++i)
{
hash.append(reinterpret_cast<const uint8_t *>(&v[i]), sizeof(T));
}
hash.finalize();
return hash.data[1];
}
};
/// Hash function for data sequences.
/** Depends on GnuTLS for SHA-256 hashing. */
class HashFunction
-172
View File
@@ -1,172 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "hash_util.hpp"
namespace mfem
{
constexpr static uint64_t rotl64(uint64_t x, int r)
{
return (x << r) | (x >> (64 - r));
}
void Hasher::init(uint64_t seed)
{
data[0] = seed;
data[1] = seed;
nbytes = 0;
}
void Hasher::add_block(uint64_t k1, uint64_t k2)
{
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
constexpr uint64_t c2 = 0x4cf5ad432745937full;
k1 *= c1;
k1 = rotl64(k1, 31);
k1 *= c2;
data[0] ^= k1;
data[0] = rotl64(data[0], 27);
data[0] += data[1];
data[0] = data[0] * 5 + 0x52dce729ull;
k2 *= c2;
k2 = rotl64(k2, 33);
k2 *= c1;
data[1] ^= k2;
data[1] = rotl64(data[1], 31);
data[1] += data[0];
data[1] = data[1] * 5 + 0x38495ab5ull;
}
static uint64_t fmix64(uint64_t k)
{
// http://zimbry.blogspot.com/2011/09/better-bit-mixing-improving-on.html
// mix13
k ^= k >> 30;
k *= 0xbf58476d1ce4e5b9ull;
k ^= k >> 27;
k *= 0x94d049bb133111ebull;
k ^= k >> 31;
return k;
}
void Hasher::append(const std::byte *vs, uint64_t bytes)
{
if (bytes == 0)
{
return;
}
auto rem = nbytes % 16;
nbytes += bytes;
std::byte *tmp = reinterpret_cast<std::byte *>(buf_);
while (true)
{
if (bytes + rem >= 16)
{
std::copy(vs, vs + 16 - rem, tmp + rem);
add_block(buf_[0], buf_[1]);
vs += (16 - rem);
bytes -= (16 - rem);
rem = 0;
}
else
{
std::copy(vs, vs + bytes, tmp + rem);
return;
}
}
}
void Hasher::finalize()
{
auto rem = nbytes % 16;
if (rem > 0)
{
nbytes -= rem;
if (rem <= 8)
{
finalize(buf_[0], rem);
}
else
{
finalize(buf_[0], buf_[1], rem);
}
return;
}
data[0] ^= nbytes;
data[1] ^= nbytes;
data[0] += data[1];
data[1] += data[0];
data[0] = fmix64(data[0]);
data[1] = fmix64(data[1]);
data[0] += data[1];
data[1] += data[0];
}
void Hasher::finalize(uint64_t k1, int num)
{
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
constexpr uint64_t c2 = 0x4cf5ad432745937full;
nbytes += num;
k1 *= c1;
k1 = rotl64(k1, 31);
k1 *= c2;
data[0] ^= k1;
data[0] ^= nbytes;
data[1] ^= nbytes;
data[0] += data[1];
data[1] += data[0];
data[0] = fmix64(data[0]);
data[1] = fmix64(data[1]);
data[0] += data[1];
data[1] += data[0];
}
void Hasher::finalize(uint64_t k1, uint64_t k2, int num)
{
constexpr uint64_t c1 = 0x87c37b91114253d5ull;
constexpr uint64_t c2 = 0x4cf5ad432745937full;
nbytes += num;
k2 *= c2;
k2 = rotl64(k2, 33);
k2 *= c1;
data[1] ^= k2;
k1 *= c1;
k1 = rotl64(k1, 31);
k1 *= c2;
data[0] ^= k1;
data[0] ^= nbytes;
data[1] ^= nbytes;
data[0] += data[1];
data[1] += data[0];
data[0] = fmix64(data[0]);
data[1] = fmix64(data[1]);
data[0] += data[1];
data[1] += data[0];
}
}
-172
View File
@@ -1,172 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_HASH_UTIL_HPP
#define MFEM_HASH_UTIL_HPP
#include <array>
#include <cstddef>
#include <tuple>
#include <functional>
#include <utility>
#include <cstdint>
namespace mfem
{
/// @brief streaming implementation for murmurhash3 128 (x64).
///
/// Constructs the hash in 3 stages: init, append, finalize.
struct Hasher
{
/// @brief Storage for the final hash result after finalize() is called.
///
/// Use data[1] when only 64 bits are required.
uint64_t data[2] = {0, 0};
private:
uint64_t nbytes = 0;
uint64_t buf_[2] = {0, 0};
public:
/// Resets the Hasher back to an initial seed
void init(uint64_t seed = 0);
/// Append data @a vs of size @a bytes.
void append(const std::byte *vs, uint64_t bytes);
void finalize();
private:
/// Add a block of 16 bytes.
void add_block(uint64_t k1, uint64_t k2);
/// @brief Add [1-8] more bytes, then finalize.
///
/// @a num must satisfy 0 < num < 9.
void finalize(uint64_t k1, int num);
/// @brief Add [1-15] more bytes, then finalize.
///
/// @a num must satisfy 0 < num < 16.
void finalize(uint64_t k1, uint64_t k2, int num);
};
template <class T> struct ChainedHasher
{
static void Append(Hasher &hasher, const T &value)
{
if constexpr (std::is_fundamental_v<T> || std::is_pointer_v<T>)
{
hasher.append(reinterpret_cast<const std::byte *>(&value), sizeof(T));
}
else
{
std::hash<T> h;
auto v = h(value);
hasher.append(reinterpret_cast<std::byte *>(&v), sizeof(v));
}
}
};
template <class T, class V> struct ChainedHasher<std::pair<T, V>>
{
static void Append(Hasher &hasher, const std::pair<T, V> &value)
{
ChainedHasher<T>::Append(hasher, value.first);
ChainedHasher<V>::Append(hasher, value.second);
}
};
template <class T, size_t N> struct ChainedHasher<std::array<T, N>>
{
static void Append(Hasher &hasher, const std::array<T, N> &value)
{
for (size_t i = 0; i < N; ++i)
{
ChainedHasher<T>::Append(hasher, value[i]);
}
}
};
template<class... Ts> struct ChainedHasher<std::tuple<Ts...>>
{
private:
template <size_t N>
static void AppendImpl(Hasher &hasher, const std::tuple<Ts...> &value)
{
ChainedHasher<std::decay_t<decltype(std::get<N>(value))>>::Append(
hasher, std::get<N>(value));
if constexpr (N + 1 < sizeof...(Ts))
{
AppendImpl<N + 1>(hasher, value);
}
}
public:
static void Append(Hasher &hasher, const std::tuple<Ts...> &value)
{
if constexpr (sizeof...(Ts))
{
AppendImpl<0>(hasher, value);
}
}
};
/// Helper class for hashing std::pair of hashable types.
struct PairHasher
{
template <class T, class V>
size_t operator()(const std::pair<T, V> &v) const noexcept
{
Hasher hash;
// chosen randomly with a 2^64-sided dice
hash.init(0xfebd1fe69813c14full);
ChainedHasher<std::pair<T, V>>::Append(hash, v);
hash.finalize();
return hash.data[1];
}
};
/// Helper class for hashing std::array of a hashable type.
struct ArrayHasher
{
template <class T, size_t N>
size_t operator()(const std::array<T, N> &v) const noexcept
{
Hasher hash;
// chosen randomly with a 2^64-sided dice
hash.init(0xfebd1fe69813c14full);
ChainedHasher<std::array<T, N>>::Append(hash, v);
hash.finalize();
return hash.data[1];
}
};
/// Helper class for hashing std::tuple of hashable types.
struct TupleHasher
{
template <class T>
size_t operator()(const T &v) const noexcept
{
Hasher hash;
// chosen randomly with a 2^64-sided dice
hash.init(0xfebd1fe69813c14full);
ChainedHasher<T>::Append(hash, v);
hash.finalize();
return hash.data[1];
}
};
} // namespace mfem
#endif
-2
View File
@@ -20,11 +20,9 @@
#if defined(MFEM_USE_HIP) && defined(__HIP__)
#define MFEM_USE_CUDA_OR_HIP
constexpr bool mfem_use_gpu = true;
#define MFEM_DEVICE __device__
#define MFEM_HOST __host__
#define MFEM_LAMBDA __host__ __device__
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(hipDeviceSynchronize())
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(hipStreamSynchronize(0))
-1
View File
@@ -55,7 +55,6 @@ list(APPEND HDRS
dinvariants.hpp
dtensor.hpp
dual.hpp
eigensolver.hpp
filteredsolver.hpp
handle.hpp
invariants.hpp
-203
View File
@@ -1,203 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
/**
* @file eigensolver.hpp
*
* @brief This file contains a common interface for all eigensolver classes
*/
#ifndef MFEM_EIGENSOLVER
#define MFEM_EIGENSOLVER
#ifdef MFEM_HYPRE
#include "hypre.hpp"
#endif
#ifdef MFEM_SLEPC
#include "slepc.hpp"
#endif
namespace mfem
{
enum class EigenSolverType
{
HYPRE,
SLEPC,
INVALID_TYPE
};
/// Provides base class for MFEM Eigensolvers
class EigenSolverBase
{
public:
EigenSolverBase() {}
/// Destructor
virtual ~EigenSolverBase() = default;
/// Solves the eigenvalue problem
virtual void Solve() = 0;
/// Set the required number of modes
virtual void SetNumModes(int num_Modes)
{
numModes=num_Modes;
}
/// @brief Set the operator to the eigenvalue problem
/// @param A - operator
virtual void SetOperator(Operator& A) = 0;
/// @brief Sets operators for the generalized eigenvalue problem
/// @param A - operator
/// @param M - mass matrix
virtual void SetOperator(Operator& A, Operator& M)
{
MFEM_ABORT("Generalized eigensolver is not supported!");
}
/// Optional method - sets preconditioner for the
/// eigenvalue solver.
virtual void SetPreconditioner(Solver& precond)
{
MFEM_ABORT("Preconditioner is not supported!");
}
/// Returns the converged eigenvalues
virtual void GetEigenvalues(Array<real_t>& eigen_vals) = 0;
/// Returns the vec_index eigenvector.
virtual void GetEigenvector(int vec_index, Vector& vector) = 0;
/// Returns the eigensolver type.
EigenSolverType GetSolverType() { return eigSolverType; }
protected:
int numModes = 0;
EigenSolverType eigSolverType = EigenSolverType::INVALID_TYPE;
};
#ifdef MFEM_HYPRE
class EigenSolverHypreLOBPCG : public EigenSolverBase
{
public:
EigenSolverHypreLOBPCG(MPI_Comm comm)
{
eigenSolver = std::make_unique<HypreLOBPCG>(comm);
eigSolverType = EigenSolverType::HYPRE;
}
~EigenSolverHypreLOBPCG() {}
void Solve() override { eigenSolver->Solve(); }
void SetNumModes(int num_Modes) override
{
eigenSolver->SetNumModes(num_Modes);
numModes = num_Modes;
}
void SetOperator(Operator& A) override { eigenSolver->SetOperator(A); }
void SetOperator(Operator& A, Operator& M) override
{
eigenSolver->SetOperator(A);
eigenSolver->SetMassMatrix(M);
}
void SetPreconditioner(Solver& precond) override { eigenSolver->SetPreconditioner(precond); }
void GetEigenvalues(Array<real_t>& eigen_vals) override { eigenSolver->GetEigenvalues(eigen_vals); }
void GetEigenvector(int vec_index, Vector& vector) override
{
const HypreParVector& eigenvec = eigenSolver->GetEigenvector(vec_index);
vector = eigenvec;
}
void SetTol(real_t tol) { eigenSolver->SetTol(tol); }
void SetRelTol(real_t rel_tol) { eigenSolver->SetRelTol(rel_tol); }
void SetMaxIter(int max_iter) { eigenSolver->SetMaxIter(max_iter); }
void SetPrintLevel(int logging) { eigenSolver->SetPrintLevel(logging); }
void SetRandomSeed(int seed) { eigenSolver->SetRandomSeed(seed); }
void SetPrecondUsageMode(int usage_mode) { eigenSolver->SetPrecondUsageMode(usage_mode); }
private:
std::unique_ptr<HypreLOBPCG> eigenSolver = nullptr;
};
#endif
#ifdef MFEM_SLEPC
class EigenSolverSlepc : public EigenSolverBase
{
public:
EigenSolverSlepc(MPI_Comm comm)
{
eigSolverType = EigenSolverType::SLEPC;
eigenSolver = std::make_unique<SlepcEigenSolver>(comm);
eigenSolver->SetWhichEigenpairs(SlepcEigenSolver::TARGET_REAL);
eigenSolver->SetTarget(0.0);
eigenSolver->SetSpectralTransformation(SlepcEigenSolver::SHIFT_INVERT);
}
~EigenSolverSlepc() {}
void Solve() override { eigenSolver->Solve(); }
void SetNumModes(int num_Modes) override
{
eigenSolver->SetNumModes(num_Modes);
numModes = num_Modes;
}
/// @brief Set the operator to the slepc eigenvalue problem. This method deep copies data to create a PetscParMatrix
/// @param A - operator, must be of type HypreParMatrix.
void SetOperator(Operator& A) override
{
petscMatA = std::make_unique<PetscParMatrix>
(dynamic_cast<HypreParMatrix*>(&A));
eigenSolver->SetOperator(*petscMatA);
}
/// @brief Set the operators to the slepc eigenvalue problem. This method deep copies data to create a PetscParMatrix
/// @param A - operator, must be of type HypreParMatrix.
/// @param M - operator, must be of type HypreParMatrix.
void SetOperator(Operator& A, Operator& M) override
{
petscMatA = std::make_unique<PetscParMatrix>
(dynamic_cast<const HypreParMatrix*>(&A));
petscMatM = std::make_unique<PetscParMatrix>
(dynamic_cast<const HypreParMatrix*>(&M));
eigenSolver->SetOperators(*petscMatA, *petscMatM);
}
void SetPreconditioner([[maybe_unused]] Solver& precond) override {}
void GetEigenvalues(Array<real_t>& eigen_vals) override
{
eigen_vals.SetSize(numModes);
for (int ik = 0; ik < numModes; ik++)
{
eigenSolver->GetEigenvalue(static_cast<unsigned int>(ik), eigen_vals[ik]);
}
}
void GetEigenvector( int vec_index, Vector& vector) override
{ eigenSolver->GetEigenvector(vec_index, vector); }
void SetTol(real_t tol) { eigenSolver->SetTol(tol); }
void SetMaxIter(int max_iter) { eigenSolver->SetMaxIter(max_iter); }
private:
std::unique_ptr<SlepcEigenSolver> eigenSolver = nullptr;
std::unique_ptr<PetscParMatrix> petscMatA = nullptr;
std::unique_ptr<PetscParMatrix> petscMatM = nullptr;
};
#endif
} // namespace mfem
#endif
-179
View File
@@ -3634,25 +3634,12 @@ void HypreSmoother::SetType(HypreSmoother::Type type_, int relax_times_)
relax_times = relax_times_;
}
void HypreSmoother::GetType(HypreSmoother::Type &type_, int &relax_times_) const
{
type_ = static_cast<HypreSmoother::Type>(type);
relax_times_ = relax_times;
}
void HypreSmoother::SetSOROptions(real_t relax_weight_, real_t omega_)
{
relax_weight = relax_weight_;
omega = omega_;
}
void HypreSmoother::GetSOROptions(real_t &relax_weight_, real_t &omega_) const
{
// TODO: are these used for all smoother types?
relax_weight_ = relax_weight;
omega_ = omega;
}
void HypreSmoother::SetPolyOptions(int poly_order_, real_t poly_fraction_,
int eig_est_cg_iter_)
{
@@ -3661,15 +3648,6 @@ void HypreSmoother::SetPolyOptions(int poly_order_, real_t poly_fraction_,
eig_est_cg_iter = eig_est_cg_iter_;
}
void HypreSmoother::GetPolyOptions(int &poly_order_, real_t &poly_fraction_,
int &eig_est_cg_iter_) const
{
// TODO: are these used for all smoother types?
poly_order_ = poly_order;
poly_fraction_ = poly_fraction;
eig_est_cg_iter_ = eig_est_cg_iter;
}
void HypreSmoother::SetTaubinOptions(real_t lambda_, real_t mu_,
int taubin_iter_)
{
@@ -3678,14 +3656,6 @@ void HypreSmoother::SetTaubinOptions(real_t lambda_, real_t mu_,
taubin_iter = taubin_iter_;
}
void HypreSmoother::GetTaubinOptions(real_t &lambda_, real_t &mu_,
int &taubin_iter_) const
{
lambda_ = lambda;
mu_ = mu;
taubin_iter_ = taubin_iter;
}
void HypreSmoother::SetWindowByName(const char* name)
{
real_t a = -1, b, c;
@@ -3708,13 +3678,6 @@ void HypreSmoother::SetWindowParameters(real_t a, real_t b, real_t c)
window_params[2] = c;
}
void HypreSmoother::GetWindowParameters(real_t &a, real_t &b, real_t &c) const
{
a = window_params[0];
b = window_params[1];
c = window_params[2];
}
void HypreSmoother::SetOperator(const Operator &op)
{
A = const_cast<HypreParMatrix *>(dynamic_cast<const HypreParMatrix *>(&op));
@@ -4210,20 +4173,12 @@ HypreSolver::~HypreSolver()
auxX.Delete();
}
void HyprePCG::SetDefaultOptions()
{
// Explicitly set just in case past/future versions of hypre change the
// defaults
SetTol(1e-6);
SetMaxIter(1000);
}
HyprePCG::HyprePCG(MPI_Comm comm) : precond(NULL)
{
iterative_mode = true;
HYPRE_ParCSRPCGCreate(comm, &pcg_solver);
SetDefaultOptions();
}
HyprePCG::HyprePCG(const HypreParMatrix &A_) : HypreSolver(&A_), precond(NULL)
@@ -4235,7 +4190,6 @@ HyprePCG::HyprePCG(const HypreParMatrix &A_) : HypreSolver(&A_), precond(NULL)
HYPRE_ParCSRMatrixGetComm(*A, &comm);
HYPRE_ParCSRPCGCreate(comm, &pcg_solver);
SetDefaultOptions();
}
void HyprePCG::SetOperator(const Operator &op)
@@ -4260,54 +4214,21 @@ void HyprePCG::SetOperator(const Operator &op)
auxX.Delete(); auxX.Reset();
}
void HyprePCG::SetUseTwoNorm(bool val)
{
HYPRE_PCGSetTwoNorm(pcg_solver, val);
}
bool HyprePCG::GetUseTwoNorm() const
{
HYPRE_Int val;
HYPRE_PCGGetTwoNorm(pcg_solver, &val);
return val != 0;
}
void HyprePCG::SetTol(real_t tol)
{
HYPRE_PCGSetTol(pcg_solver, tol);
}
real_t HyprePCG::GetTol() const
{
HYPRE_Real tol;
HYPRE_PCGGetTol(pcg_solver, &tol);
return tol;
}
void HyprePCG::SetAbsTol(real_t atol)
{
HYPRE_PCGSetAbsoluteTol(pcg_solver, atol);
}
real_t HyprePCG::GetAbsTol() const
{
HYPRE_Real atol;
hypre_PCGGetAbsoluteTol(pcg_solver, &atol);
return atol;
}
void HyprePCG::SetMaxIter(int max_iter)
{
HYPRE_PCGSetMaxIter(pcg_solver, max_iter);
}
int HyprePCG::GetMaxIter() const
{
HYPRE_Int max_iter;
HYPRE_PCGGetMaxIter(pcg_solver, &max_iter);
return max_iter;
}
void HyprePCG::SetLogging(int logging)
{
HYPRE_PCGSetLogging(pcg_solver, logging);
@@ -4423,20 +4344,6 @@ HyprePCG::~HyprePCG()
HYPRE_ParCSRPCGDestroy(pcg_solver);
}
#if MFEM_HYPRE_VERSION >= 21500
HypreParVector HyprePCG::GetResiduals() const
{
HYPRE_ParVector r;
HYPRE_ParCSRPCGGetResidual(pcg_solver, &r);
return HypreParVector(r);
}
void HyprePCG::GetFinalAbsResidualNorm(real_t &final_res_norm, real_t p) const
{
auto r = GetResiduals();
ParNormlp(r, p, r.GetComm());
}
#endif
HypreGMRES::HypreGMRES(MPI_Comm comm) : precond(NULL)
{
@@ -4492,69 +4399,26 @@ void HypreGMRES::SetOperator(const Operator &op)
auxX.Delete(); auxX.Reset();
}
#if MFEM_HYPRE_VERSION >= 21500
HypreParVector HypreGMRES::GetResiduals() const
{
HYPRE_ParVector r;
HYPRE_ParCSRGMRESGetResidual(gmres_solver, &r);
return HypreParVector(r);
}
void HypreGMRES::GetFinalAbsResidualNorm(real_t &final_res_norm, real_t p) const
{
auto r = GetResiduals();
ParNormlp(r, p, r.GetComm());
}
#endif
void HypreGMRES::SetTol(real_t tol)
{
HYPRE_GMRESSetTol(gmres_solver, tol);
}
real_t HypreGMRES::GetTol()const
{
HYPRE_Real tol;
HYPRE_GMRESGetTol(gmres_solver, &tol);
return tol;
}
void HypreGMRES::SetAbsTol(real_t tol)
{
HYPRE_GMRESSetAbsoluteTol(gmres_solver, tol);
}
real_t HypreGMRES::GetAbsTol() const
{
HYPRE_Real atol;
HYPRE_GMRESGetAbsoluteTol(gmres_solver, &atol);
return atol;
}
void HypreGMRES::SetMaxIter(int max_iter)
{
HYPRE_GMRESSetMaxIter(gmres_solver, max_iter);
}
int HypreGMRES::GetMaxIter() const
{
HYPRE_Int max_iter;
HYPRE_GMRESGetMaxIter(gmres_solver, &max_iter);
return max_iter;
}
void HypreGMRES::SetKDim(int k_dim)
{
HYPRE_GMRESSetKDim(gmres_solver, k_dim);
}
int HypreGMRES::GetKDim() const
{
HYPRE_Int k_dim;
HYPRE_GMRESGetKDim(gmres_solver, &k_dim);
return k_dim;
}
void HypreGMRES::SetLogging(int logging)
{
HYPRE_GMRESSetLogging(gmres_solver, logging);
@@ -4712,37 +4576,16 @@ void HypreFGMRES::SetTol(real_t tol)
HYPRE_ParCSRFlexGMRESSetTol(fgmres_solver, tol);
}
real_t HypreFGMRES::GetTol() const
{
HYPRE_Real tol;
HYPRE_FlexGMRESGetTol(fgmres_solver, &tol);
return tol;
}
void HypreFGMRES::SetMaxIter(int max_iter)
{
HYPRE_ParCSRFlexGMRESSetMaxIter(fgmres_solver, max_iter);
}
int HypreFGMRES::GetMaxIter() const
{
HYPRE_Int max_iter;
HYPRE_FlexGMRESGetMaxIter(fgmres_solver, &max_iter);
return max_iter;
}
void HypreFGMRES::SetKDim(int k_dim)
{
HYPRE_ParCSRFlexGMRESSetKDim(fgmres_solver, k_dim);
}
int HypreFGMRES::GetKDim() const
{
HYPRE_Int k_dim;
HYPRE_FlexGMRESGetKDim(fgmres_solver, &k_dim);
return k_dim;
}
void HypreFGMRES::SetLogging(int logging)
{
HYPRE_ParCSRFlexGMRESSetLogging(fgmres_solver, logging);
@@ -4839,21 +4682,6 @@ HypreFGMRES::~HypreFGMRES()
HYPRE_ParCSRFlexGMRESDestroy(fgmres_solver);
}
#if MFEM_HYPRE_VERSION >= 21500
HypreParVector HypreFGMRES::GetResiduals() const
{
HYPRE_ParVector r;
HYPRE_ParCSRFlexGMRESGetResidual(fgmres_solver, &r);
return HypreParVector(r);
}
void HypreFGMRES::GetFinalAbsResidualNorm(real_t &final_res_norm,
real_t p) const
{
auto r = GetResiduals();
ParNormlp(r, p, r.GetComm());
}
#endif
void HypreDiagScale::SetOperator(const Operator &op)
{
@@ -5342,13 +5170,6 @@ void HypreBoomerAMG::ResetAMGPrecond()
}
}
int HypreBoomerAMG::GetMaxIter() const
{
HYPRE_Int max_iter;
HYPRE_BoomerAMGGetMaxIter(amg_precond, &max_iter);
return max_iter;
}
void HypreBoomerAMG::SetOperator(const Operator &op)
{
const HypreParMatrix *new_A = dynamic_cast<const HypreParMatrix *>(&op);
+7 -97
View File
@@ -1160,15 +1160,6 @@ public:
return HypreUsingGPU() ? l1Jacobi : l1GS;
}
/// Default solver settings:
/// type = DefaultType()
/// relax_times = 1
/// omega = 1.0
/// poly_order = 2
/// poly_fraction = 0.3
/// lambda = 0.5
/// mu = -0.5
/// taubin_iter = 40
HypreSmoother();
HypreSmoother(const HypreParMatrix &A_, int type = DefaultType(),
@@ -1178,28 +1169,20 @@ public:
/// Set the relaxation type and number of sweeps
void SetType(HypreSmoother::Type type, int relax_times = 1);
using Operator::GetType;
void GetType(HypreSmoother::Type &type, int &relax_times) const;
/// Set SOR-related parameters
void SetSOROptions(real_t relax_weight, real_t omega);
void GetSOROptions(real_t &relax_weight, real_t &omega) const;
/// Set parameters for polynomial smoothing
/** By default, 10 iterations of CG are used to estimate the eigenvalues.
Setting eig_est_cg_iter = 0 uses hypre's hypre_ParCSRMaxEigEstimate() instead. */
void SetPolyOptions(int poly_order, real_t poly_fraction,
int eig_est_cg_iter = 10);
void GetPolyOptions(int &poly_order, real_t &poly_fraction,
int &eig_est_cg_iter) const;
/// Set parameters for Taubin's lambda-mu method
void SetTaubinOptions(real_t lambda, real_t mu, int iter);
void GetTaubinOptions(real_t &lambda, real_t &mu, int &iter) const;
/// Convenience function for setting canonical windowing parameters
void SetWindowByName(const char* window_name);
/// Set parameters for windowing function for FIR smoother.
void SetWindowParameters(real_t a, real_t b, real_t c);
void GetWindowParameters(real_t &a, real_t &b, real_t &c) const;
/// Compute window and Chebyshev coefficients for given polynomial order.
void SetFIRCoefficients(real_t max_eig);
@@ -1207,15 +1190,12 @@ public:
/** By default, the l1-norms take their sign from the corresponding diagonal
entries in the associated matrix. */
void SetPositiveDiagonal(bool pos = true) { pos_l1_norms = pos; }
bool IsPositiveDiagonal() const { return pos_l1_norms; };
/** Explicitly indicate whether the linear system matrix A is symmetric. If A
is symmetric, the smoother will also be symmetric. In this case, calling
MultTranspose will be redirected to Mult. (This is also done if the
smoother is diagonal.) By default, A is assumed to be nonsymmetric. */
void SetOperatorSymmetry(bool is_sym) { A_is_symmetric = is_sym; }
/// @return true if the smoother assumes A is symmetric, false otherwise
bool IsOperatorSymmetric() const { return A_is_symmetric; }
/** Set/update the associated operator. Must be called after setting the
HypreSmoother type and options. */
@@ -1347,7 +1327,6 @@ public:
#endif
/// PCG solver in hypre
/// Defaults to (relative) tol=1e-6, atol=0, max_iter=1000
class HyprePCG : public HypreSolver
{
private:
@@ -1355,9 +1334,6 @@ private:
HypreSolver * precond;
/// Default PCG options
void SetDefaultOptions();
public:
HyprePCG(MPI_Comm comm);
@@ -1366,11 +1342,8 @@ public:
void SetOperator(const Operator &op) override;
void SetTol(real_t tol);
real_t GetTol() const;
void SetAbsTol(real_t atol);
real_t GetAbsTol() const;
void SetMaxIter(int max_iter);
int GetMaxIter() const;
void SetLogging(int logging);
void SetPrintLevel(int print_lvl);
@@ -1395,32 +1368,12 @@ public:
num_iterations = internal::to_int(num_it);
}
/// Gets the relative residual norm
void GetFinalResidualNorm(real_t &final_res_norm) const
{
HYPRE_ParCSRPCGGetFinalRelativeResidualNorm(pcg_solver,
&final_res_norm);
}
/// @param[in] use
/// Convergence criterion:
/// - when true: (r, r) < max(r_tol^2 (b, b), a_tol^2)
/// - when false: (r, A r) < max(r_tol^2 (b, A b), a_tol^2)
/// @sa HYPRE_PCGSetTwoNorm
void SetUseTwoNorm(bool use);
/// @sa HYPRE_PCGGetTwoNorm
bool GetUseTwoNorm() const;
#if MFEM_HYPRE_VERSION >= 21500
/// Gets the internal Hypre solver residual vector.
/// @sa HYPRE_ParCSRPCGGetResidual
HypreParVector GetResiduals() const;
/// Computes the absolute residual p-norm.
void GetFinalAbsResidualNorm(real_t &final_res_norm, real_t p = 2) const;
#endif
/// The typecast to HYPRE_Solver returns the internal pcg_solver
operator HYPRE_Solver() const override { return pcg_solver; }
@@ -1438,8 +1391,7 @@ public:
virtual ~HyprePCG();
};
/// GMRES solver in hypre.
/// Defaults to k=50, (relative) tol=1e-6, atol=0, max_iter=100.
/// GMRES solver in hypre
class HypreGMRES : public HypreSolver
{
private:
@@ -1458,13 +1410,9 @@ public:
void SetOperator(const Operator &op) override;
void SetTol(real_t tol);
real_t GetTol() const;
void SetAbsTol(real_t tol);
real_t GetAbsTol() const;
void SetMaxIter(int max_iter);
int GetMaxIter() const;
void SetKDim(int dim);
int GetKDim() const;
void SetLogging(int logging);
void SetPrintLevel(int print_lvl);
@@ -1484,22 +1432,12 @@ public:
num_iterations = internal::to_int(num_it);
}
/// Gets the relative residual norm
void GetFinalResidualNorm(real_t &final_res_norm) const
{
HYPRE_ParCSRGMRESGetFinalRelativeResidualNorm(gmres_solver,
&final_res_norm);
}
#if MFEM_HYPRE_VERSION >= 21500
/// Gets the internal Hypre solver residual vector.
/// @sa HYPRE_ParCSRGMRESGetResidual
HypreParVector GetResiduals() const;
/// Computes the absolute residual p-norm.
void GetFinalAbsResidualNorm(real_t &final_res_norm, real_t p = 2) const;
#endif
/// The typecast to HYPRE_Solver returns the internal gmres_solver
operator HYPRE_Solver() const override { return gmres_solver; }
@@ -1517,8 +1455,7 @@ public:
virtual ~HypreGMRES();
};
/// Flexible GMRES solver in hypre.
/// Defaults to k=50, (relative) tol=1e-6, max_iter=100.
/// Flexible GMRES solver in hypre
class HypreFGMRES : public HypreSolver
{
private:
@@ -1537,11 +1474,8 @@ public:
void SetOperator(const Operator &op) override;
void SetTol(real_t tol);
real_t GetTol() const;
void SetMaxIter(int max_iter);
int GetMaxIter() const;
void SetKDim(int dim);
int GetKDim() const;
void SetLogging(int logging);
void SetPrintLevel(int print_lvl);
@@ -1561,22 +1495,12 @@ public:
num_iterations = internal::to_int(num_it);
}
/// Gets the relative residual norm
void GetFinalResidualNorm(real_t &final_res_norm) const
{
HYPRE_ParCSRFlexGMRESGetFinalRelativeResidualNorm(fgmres_solver,
&final_res_norm);
}
#if MFEM_HYPRE_VERSION >= 21500
/// Gets the internal Hypre solver residual vector.
/// @sa HYPRE_ParCSRFlexGMRESGetResidual
HypreParVector GetResiduals() const;
/// Computes the absolute residual p-norm.
void GetFinalAbsResidualNorm(real_t &final_res_norm, real_t p = 2) const;
#endif
/// The typecast to HYPRE_Solver returns the internal fgmres_solver
operator HYPRE_Solver() const override { return fgmres_solver; }
@@ -1632,8 +1556,7 @@ public:
virtual ~HypreDiagScale() { }
};
/// The ParaSails preconditioner in hypre.
/// See SetDefaultOptions() for default solver options.
/// The ParaSails preconditioner in hypre
class HypreParaSails : public HypreSolver
{
private:
@@ -1762,14 +1685,10 @@ public:
/**
@brief Wrapper for Hypre's native parallel ILU preconditioner.
Default parameters: ILU(k) factorization type, tol=0.0 (for use as a
preconditioner), fill level = 1 (for ILU(k)), reverse Cuthill-McKee (RCM)
re-ordering.
If you need to change this, or any other option, you can use the HYPRE_Solver
method to cast the object for use with Hypre's native functions. For example, if
want to use natural ordering rather than RCM reordering, you can use the
following approach:
The default ILU factorization type is ILU(k). If you need to change this, or
any other option, you can use the HYPRE_Solver method to cast the object for use
with Hypre's native functions. For example, if want to use natural ordering
rather than RCM reordering, you can use the following approach:
@code
mfem::HypreILU ilu();
@@ -1910,7 +1829,6 @@ public:
void SetMaxIter(int max_iter)
{ HYPRE_BoomerAMGSetMaxIter(amg_precond, max_iter); }
int GetMaxIter() const;
/// Expert option - consult hypre documentation/team
void SetMaxLevels(int max_levels)
@@ -1935,8 +1853,6 @@ public:
/// Expert option - consult hypre documentation/team
void SetRelaxType(int relax_type)
{ HYPRE_BoomerAMGSetRelaxType(amg_precond, relax_type); }
// not implemented in hypre
// int GetRelaxType() const;
/// Expert option - consult hypre documentation/team
void SetCycleType(int cycle_type)
@@ -2237,14 +2153,8 @@ public:
~HypreLOBPCG();
void SetTol(real_t tol);
// not implemented in HYPRE
// real_t GetTol() const;
void SetRelTol(real_t rel_tol);
// not implemented in HYPRE
// real_t GetRelTol() const;
void SetMaxIter(int max_iter);
// not implemented in HYPRE
// int GetMaxIter() const;
void SetPrintLevel(int logging);
void SetNumModes(int num_eigs) { nev = num_eigs; }
void SetPrecondUsageMode(int pcg_mode);
-13
View File
@@ -3639,20 +3639,12 @@ void PetscBDDCSolver::BDDCSolverConstructor(const PetscBDDCSolverParams &opts)
// make sure ess/nat_dof have been collectively set
PetscBool lpr = PETSC_FALSE,pr;
if (opts.ess_dof) { lpr = PETSC_TRUE; }
#if PETSC_VERSION_LT(3,24,0)
mpiierr = MPI_Allreduce(&lpr,&pr,1,MPIU_BOOL,MPI_LOR,comm);
#else
mpiierr = MPI_Allreduce(&lpr,&pr,1,MPI_C_BOOL,MPI_LOR,comm);
#endif
CCHKERRQ(comm,mpiierr);
MFEM_VERIFY(lpr == pr,"ess_dof should be collectively set");
lpr = PETSC_FALSE;
if (opts.nat_dof) { lpr = PETSC_TRUE; }
#if PETSC_VERSION_LT(3,24,0)
mpiierr = MPI_Allreduce(&lpr,&pr,1,MPIU_BOOL,MPI_LOR,comm);
#else
mpiierr = MPI_Allreduce(&lpr,&pr,1,MPI_C_BOOL,MPI_LOR,comm);
#endif
CCHKERRQ(comm,mpiierr);
MFEM_VERIFY(lpr == pr,"nat_dof should be collectively set");
// make sure fields have been collectively set
@@ -4066,13 +4058,8 @@ void PetscNonlinearSolver::SetOperator(const Operator &op)
ls = (PetscBool)(height == op.Height() && width == op.Width() &&
(void*)&op == fctx &&
(void*)&op == jctx);
#if PETSC_VERSION_LT(3,24,0)
mpiierr = MPI_Allreduce(&ls,&gs,1,MPIU_BOOL,MPI_LAND,
PetscObjectComm((PetscObject)snes));
#else
mpiierr = MPI_Allreduce(&ls,&gs,1,MPI_C_BOOL,MPI_LAND,
PetscObjectComm((PetscObject)snes));
#endif
CCHKERRQ(PetscObjectComm((PetscObject)snes),mpiierr);
if (!gs)
{
-5
View File
@@ -1066,11 +1066,6 @@ void SparseMatrix::BooleanMultTranspose(const Array<int> &x,
y.SetSize(Width());
y = 0;
HostReadI();
HostReadJ();
x.HostRead();
y.HostReadWrite();
for (int i = 0; i < Height(); i++)
{
if (x[i])
+1 -12
View File
@@ -363,19 +363,14 @@ void SuperLUSolver::Init(MPI_Comm comm)
// Set default options:
// options.Fact = DOFACT;
// options.Equil = YES;
// options.ParSymbFact = NO;
// options.ColPerm = METIS_AT_PLUS_A;
// options.RowPerm = LargeDiag_MC64;
// options.ReplaceTinyPivot = NO;
// options.IterRefine = SLU_DOUBLE;
// options.Trans = NOTRANS;
// options.IterRefine = SLU_DOUBLE;
// options.SolveInitialized = NO;
// options.RefineInitialized = NO;
// options.PrintStat = YES;
// options.lookahead_etree = NO;
// options.num_lookaheads = 10;
// options.superlu_acc_offload = 1;
// options.SymPattern = NO;
superlu_dist_options_t *options = (superlu_dist_options_t *)optionsPtr_;
set_default_options_dist(options);
#if SUPERLU_DIST_MAJOR_VERSION > 7 || \
@@ -477,12 +472,6 @@ void SuperLUSolver::SetFact(superlu::Fact fact)
options->Fact = opt;
}
void SuperLUSolver::SetDeviceOffload(bool offload)
{
superlu_dist_options_t *options = (superlu_dist_options_t *)optionsPtr_;
options->superlu_acc_offload = offload;
}
void SuperLUSolver::SetOperator(const Operator &op)
{
// Verify that we have a compatible operator
+1 -6
View File
@@ -250,8 +250,7 @@ public:
work (default false) */
void SetSymmetricPattern(bool sym);
/** @brief Specify whether to perform parallel symbolic factorization
(default false)
/** @brief Specify whether to perform parallel symbolic factorization.
@note If true SuperLU will use superlu::PARMETIS for the Column
Permutation regardless of the setting */
void SetParSymbFact(bool par);
@@ -264,10 +263,6 @@ public:
superlu::FACTORED*/
void SetFact(superlu::Fact fact);
/** @brief Specify whether to offload numerical factorization onto the device
(default true if SuperLU_DIST has been compiled with GPU support) */
void SetDeviceOffload(bool offload);
// Processor grid for SuperLU_DIST.
const int nprow_, npcol_, npdep_;
+1
View File
@@ -794,6 +794,7 @@ status info:
$(info MFEM_MPI_NP = $(MFEM_MPI_NP))
@true
ASTYLE_BIN = astyle
ASTYLE = $(ASTYLE_BIN) --options=$(SRC)config/mfem.astylerc
ASTYLE_VER = "Artistic Style Version 3.1"
FORMAT_FILES = $(foreach dir,$(DIRS) $(EM_DIRS) config,$(dir)/*.?pp)
+10 -15
View File
@@ -2078,13 +2078,12 @@ public:
contrary to the ones obtained through Mesh::GetFacesElements and can
directly be used, e.g., Elem1 and Elem2 indices.
Likewise the orientations for Elem1 and Elem2 already take into account
special cases and can be used as is. */
special cases and can be used as is.
*/
struct FaceInformation
{
/// The face topology (boundary, conforming, or nonconforming).
FaceTopology topology;
/// Information about the adjacent elements.
struct
{
ElementLocation location;
@@ -2094,13 +2093,8 @@ public:
int orientation;
} element[2];
/// Detailed face information (see FaceInfoTag).
FaceInfoTag tag;
/// If the face is nonconforming, the index of the NC face. -1 otherwise.
int ncface;
/// The point matrix for nonconforming faces.
const DenseMatrix* point_matrix;
/** @brief Return true if the face is a local interior face which is NOT
@@ -2119,20 +2113,21 @@ public:
/** @brief return true if the face is an interior face to the computation
domain, either a local or shared interior face (not a boundary face)
which is NOT a master nonconforming face. */
which is NOT a master nonconforming face.
*/
bool IsInterior() const
{
return topology == FaceTopology::Conforming ||
topology == FaceTopology::Nonconforming;
}
/// Return true if the face is a boundary face.
/** @brief Return true if the face is a boundary face. */
bool IsBoundary() const
{
return topology == FaceTopology::Boundary;
}
/// Return true if the face is of the same type as @a type.
/// @brief Return true if the face is of the same type as @a type.
bool IsOfFaceType(FaceType type) const
{
switch (type)
@@ -2146,13 +2141,13 @@ public:
}
}
/// Return true if the face is a conforming face.
/// @brief Return true if the face is a conforming face.
bool IsConforming() const
{
return topology == FaceTopology::Conforming;
}
/// Return true if the face is a nonconforming fine face.
/// @brief Return true if the face is a nonconforming fine face.
bool IsNonconformingFine() const
{
return topology == FaceTopology::Nonconforming &&
@@ -2160,7 +2155,7 @@ public:
element[1].conformity == ElementConformity::Superset);
}
/// Return true if the face is a nonconforming coarse face.
/// @brief Return true if the face is a nonconforming coarse face.
/** Note that ghost nonconforming master faces cannot be clearly
identified as such with the currently available information, so this
method will return false for such faces. */
@@ -2170,7 +2165,7 @@ public:
element[1].conformity == ElementConformity::Subset;
}
/// cast operator from FaceInformation to FaceInfo.
/// @brief cast operator from FaceInformation to FaceInfo.
operator Mesh::FaceInfo() const;
};
+3 -20
View File
@@ -43,30 +43,13 @@ KnotVector::KnotVector(istream &input)
KnotVector::KnotVector(int order, int NCP)
{
if (NCP == -1)
{
NumOfControlPoints = order + 1;
}
else
{
NumOfControlPoints = NCP;
}
Order = order;
NumOfControlPoints = NCP;
knot.SetSize(NumOfControlPoints + Order + 1);
NumOfElements = 0;
coarse = false;
if (NCP == -1)
{
for (int i = 0 ; i < Order + 1; i++)
{
knot[i] = 0.0;
knot[i + Order + 1] = 1.0;
}
}
else
{
knot = -1.;
}
knot = -1.;
}
KnotVector::KnotVector(int order, const Vector &k)
+6 -8
View File
@@ -74,13 +74,9 @@ public:
integers are read, for order and number of control points. */
KnotVector(std::istream &input);
/** @brief Create a KnotVector with order @a order.
When @a NCP is not provided the number of control points is set to
@a order + 1, and the first @a order + 1 knots are set to 0 and last
@a order + 1 knots are set to 1.
When @a NCP is given number of control points is @a NCP and
the knots are initialized to -1) */
KnotVector(int order, int NCP = -1);
/** @brief Create a KnotVector with undefined knots (initialized to -1) of
order @a order and number of control points @a NCP. */
KnotVector(int order, int NCP);
/** @brief Create a KnotVector with order @a order and knots @a knot.
If @a k has the correct number of repeated knots at the begin and end,
@@ -92,10 +88,12 @@ public:
/** @brief Create a KnotVector by passing in a degree, a Vector of interval
lengths of length n, and a list of continuity of length n + 1.
The intervals refer to spans between unique knot values (not counting
zero-size intervals at repeated knots), and the continuity values should
be >= -1 (discontinuous) and <= order-1 (maximally-smooth for the given
polynomial degree). Periodicity is not supported.*/
polynomial degree). Periodicity is not supported.
*/
KnotVector(int order, const Vector& intervals,
const Array<int>& continuity);
+2
View File
@@ -1211,6 +1211,7 @@ void ParNCMesh::GetFaceNeighbors(ParMesh &pmesh)
if (Dim <= 2) { nfaces = NEdges, nghosts = NGhostEdges; }
// enlarge Mesh::faces_info for ghost slaves
MFEM_ASSERT(pmesh.faces_info.Size() == nfaces, "");
MFEM_ASSERT(pmesh.GetNumFaces() == nfaces, "");
pmesh.faces_info.SetSize(nfaces + nghosts);
for (int i = nfaces; i < pmesh.faces_info.Size(); i++)
@@ -1311,6 +1312,7 @@ void ParNCMesh::GetFaceNeighbors(ParMesh &pmesh)
// Mesh::ApplyLocalSlaveTransformation.
}
MFEM_ASSERT(fi.NCFace < 0, "fi.NCFace = " << fi.NCFace);
fi.NCFace = pmesh.nc_faces_info.Size();
pmesh.nc_faces_info.Append(Mesh::NCFaceInfo(true, sf.master, pm));
}
+9 -109
View File
@@ -174,6 +174,7 @@ ParticleTrajectories::ParticleTrajectories(const ParticleSet &particles,
void ParticleTrajectories::AddSegmentStart()
{
if (!pset.GetNParticles()) { return; }
// Create a new mesh for all particle segments for this timestep
segment_meshes.emplace_front(1, pset.GetNParticles()*2,
pset.GetNParticles(),
@@ -199,10 +200,11 @@ void ParticleTrajectories::AddSegmentStart()
void ParticleTrajectories::SetSegmentEnd()
{
if (segment_meshes.empty()) { return; } // no segments to end
const Array<ParticleSet::IDType> &end_ids = pset.GetIDs();
// Add all endpoint vertices + segments for all particles that were in
// SetSegmentStart
// Add all endpoint vertices + segments for all particles
int num_start = segment_ids.front().Size();
for (int i = 0; i < num_start; i++)
{
@@ -228,6 +230,11 @@ void ParticleTrajectories::SetSegmentEnd()
void ParticleTrajectories::Visualize()
{
SetSegmentEnd();
if (segment_meshes.empty() && !mesh)
{
AddSegmentStart();
return;
}
// Create a mesh of all the trajectory segments
std::vector<Mesh*> all_meshes;
@@ -239,23 +246,8 @@ void ParticleTrajectories::Visualize()
{
all_meshes.push_back(mesh);
}
if (mesh_bb)
{
all_meshes.push_back(mesh_bb);
}
Mesh trajectories(all_meshes.data(), all_meshes.size());
bool vis = trajectories.GetNE() > 0;
#ifdef MFEM_USE_MPI
MPI_Allreduce(MPI_IN_PLACE, &vis, 1, MFEM_MPI_CXX_BOOL,
MPI_LOR, pset.GetComm());
#endif // MFEM_USE_MPI
if (!vis) // if all rank have 0 elements, skip visualization
{
AddSegmentStart();
return;
}
#ifdef MFEM_USE_MPI
VisualizeMesh(sock, vishost, visport, trajectories, comm,
@@ -268,97 +260,5 @@ void ParticleTrajectories::Visualize()
AddSegmentStart();
}
void ParticleTrajectories::SetVisualizationBoundingBox(const Vector &xmin,
const Vector &xmax)
{
MFEM_VERIFY(xmin.Size() == pset.GetDim() &&
xmax.Size() == pset.GetDim(),
"Bounding box dimension must match ParticleSet dimension.");
// Create a box mesh for visualization
if (mesh_bb)
{
delete mesh_bb;
mesh_bb = nullptr;
}
if (pset.GetDim() == 2)
{
int dim = 2;
int nvert = 4;
int nelem = 4;
mesh_bb = new Mesh(1, nvert, nelem, 0, dim);
Vector v0(dim), v1(dim), v2(dim), v3(dim);
v0 = xmin;
v1 = xmax;
v2[0] = xmax[0]; v2[1] = xmin[1];
v3[0] = xmin[0]; v3[1] = xmax[1];
mesh_bb->AddVertex(v0);
mesh_bb->AddVertex(v1);
mesh_bb->AddVertex(v2);
mesh_bb->AddVertex(v3);
int vi[2] = {0,1};
mesh_bb->AddSegment(vi);
vi[0] = 1; vi[1] = 2;
mesh_bb->AddSegment(vi);
vi[0] = 2; vi[1] = 3;
mesh_bb->AddSegment(vi);
vi[0] = 3; vi[1] = 0;
mesh_bb->AddSegment(vi);
mesh_bb->FinalizeMesh();
}
else // dim == 3
{
int dim = 3;
int nvert = 8;
int nelem = 12;
mesh_bb = new Mesh(1, nvert, nelem, 0, dim);
Vector v(dim);
// Vertices
v[0] = xmin[0]; v[1] = xmin[1]; v[2] = xmin[2];
mesh_bb->AddVertex(v); // 0: 000
v[0] = xmax[0]; v[1] = xmin[1]; v[2] = xmin[2];
mesh_bb->AddVertex(v); // 1: 100
v[0] = xmax[0]; v[1] = xmax[1]; v[2] = xmin[2];
mesh_bb->AddVertex(v); // 2: 110
v[0] = xmin[0]; v[1] = xmax[1]; v[2] = xmin[2];
mesh_bb->AddVertex(v); // 3: 010
v[0] = xmin[0]; v[1] = xmin[1]; v[2] = xmax[2];
mesh_bb->AddVertex(v); // 4: 001
v[0] = xmax[0]; v[1] = xmin[1]; v[2] = xmax[2];
mesh_bb->AddVertex(v); // 5: 101
v[0] = xmax[0]; v[1] = xmax[1]; v[2] = xmax[2];
mesh_bb->AddVertex(v); // 6: 111
v[0] = xmin[0]; v[1] = xmax[1]; v[2] = xmax[2];
mesh_bb->AddVertex(v); // 7: 011
// Segments
int vi[2];
// Bottom face
vi[0] = 0; vi[1] = 1; mesh_bb->AddSegment(vi);
vi[0] = 1; vi[1] = 2; mesh_bb->AddSegment(vi);
vi[0] = 2; vi[1] = 3; mesh_bb->AddSegment(vi);
vi[0] = 3; vi[1] = 0; mesh_bb->AddSegment(vi);
// Top face
vi[0] = 4; vi[1] = 5; mesh_bb->AddSegment(vi);
vi[0] = 5; vi[1] = 6; mesh_bb->AddSegment(vi);
vi[0] = 6; vi[1] = 7; mesh_bb->AddSegment(vi);
vi[0] = 7; vi[1] = 4; mesh_bb->AddSegment(vi);
// Vertical edges
vi[0] = 0; vi[1] = 4; mesh_bb->AddSegment(vi);
vi[0] = 1; vi[1] = 5; mesh_bb->AddSegment(vi);
vi[0] = 2; vi[1] = 6; mesh_bb->AddSegment(vi);
vi[0] = 3; vi[1] = 7; mesh_bb->AddSegment(vi);
mesh_bb->FinalizeMesh();
}
}
} // namespace common
} // namespace mfem
+2 -17
View File
@@ -46,8 +46,7 @@ class ParticleTrajectories
{
protected:
const ParticleSet &pset;
Mesh *mesh = nullptr; // optional edge mesh to visualize along with particles
Mesh *mesh_bb = nullptr; // optional bounding box mesh for visualization
Mesh *mesh = nullptr;
socketstream sock;
/// Track particle IDs that exist at the segment start.
@@ -91,24 +90,10 @@ public:
const char *keys_=nullptr);
/// Add a mesh to be visualized along with the particle trajectories.
void AddMeshForVisualization(Mesh *mesh_)
{
MFEM_VERIFY(mesh_->Dimension() == 1,
"Mesh dimension must be 1 to match the particle trajectory.");
mesh = mesh_;
}
void AddMeshForVisualization(Mesh *mesh_) { mesh = mesh_; }
/// Visualize the particle trajectories (and mesh if provided).
void Visualize();
/// Set the bounding box for visualization.
void SetVisualizationBoundingBox(const Vector &xmin, const Vector &xmax);
/// Destructor
~ParticleTrajectories()
{
delete mesh_bb;
}
};
+5 -7
View File
@@ -34,13 +34,11 @@ if (MFEM_USE_MPI)
EXTRA_HEADERS maxwell_solver.hpp ${MFEM_MINIAPPS_COMMON_HEADERS}
LIBRARIES mfem-common)
if (MFEM_USE_GSLIB)
add_mfem_miniapp(lorentz
MAIN lorentz.cpp
EXTRA_HEADERS ${MFEM_MINIAPPS_COMMON_HEADERS}
LIBRARIES mfem-common)
endif()
add_mfem_miniapp(lorentz
MAIN lorentz.cpp
EXTRA_HEADERS ${MFEM_MINIAPPS_COMMON_HEADERS}
LIBRARIES mfem-common)
# Add the corresponding tests to the "test" target
if (MFEM_ENABLE_TESTING)
add_test(NAME tesla_np=4
+385 -460
View File
@@ -13,8 +13,8 @@
// Lorentz Miniapp: Simple Lorentz Force Particle Mover
// -----------------------------------------------------
//
// This miniapp computes the trajectories of a set of charged particles subject
// to Lorentz forces.
// This miniapp computes the trajectory of a single charged particle subject to
// Lorentz forces.
//
// dp/dt = q (E + v x B)
//
@@ -23,14 +23,11 @@
//
// The electric and magnetic fields are read from VisItDataCollection objects
// such as those produced by the Volta and Tesla miniapps. It is notable that
// these two fields do not need to be defined on the same mesh. At least
// one of either an electric field or a magnetic field must be provided. The
// particles' locations and momenta are randomly initialized within a bounding
// box specified by command line input.
//
// This miniapp demonstrates the use of ParticleSet with FindPointsGSLIB. When
// particles leave either domains, they are subject to removal. Redistribution
// of particle data between MPI ranks is also demonstrated.
// these two fields do not need to be defined on the same mesh. Of course, the
// particle trajectory can only be computed on the intersection of the two
// domains. The starting point of the path must be chosen within in this
// intersection and the trajectory will terminate when it leaves the
// intersection or reaches a specified time duration.
//
// Note that the VisItDataCollection objects must have been stored using the
// parallel format e.g. visit_dc.SetFormat(DataCollection::PARALLEL_FORMAT);.
@@ -40,23 +37,31 @@
//
// Sample runs:
//
// Particles accelerating in a constant electric field
// mpirun -np 4 volta -m ../../data/inline-hex.mesh -dbcs '1 6' -dbcv '0 1'
// mpirun -np 4 lorentz -er Volta-AMR-Parallel -npt 100 -xmin '0.0 0.0 0.0' -xmax '1.0 1.0 1.0' -pmin '1 0 0' -pmax '1 0 0' -rdf 0 -vt 0 -nt 100
// Free particle moving with constant velocity
// mpirun -np 4 lorentz -p0 '1 1 1'
//
// Particles accelerating in a constant magnetic field
// Particle accelerating in a constant electric field
// mpirun -np 4 volta -m ../../data/inline-hex.mesh -dbcs '1 6' -dbcv '0 1'
// mpirun -np 4 lorentz -er Volta-AMR-Parallel -x0 '0.5 0.5 0.9' -p0 '1 0 0'
//
// Particle accelerating in a constant magnetic field
// mpirun -np 4 tesla -m ../../data/inline-hex.mesh -ubbc '0 0 1'
// mpirun -np 4 lorentz -br Tesla-AMR-Parallel -npt 10 -xmin '0.0 0.0 0.0' -xmax '1.0 1.0 1.0' -pmin '0 0.1 0.05' -pmax '0 0.4 0.1' -nt 1000 -rdf 0 -vt 0
// mpirun -np 4 lorentz -br Tesla-AMR-Parallel -x0 '0.1 0.5 0.1' -p0 '0 0.4 0.1' -tf 9
//
// Magnetic mirror effect near a charged sphere and a bar magnet
// mpirun -np 4 volta -m ../../data/ball-nurbs.mesh -dbcs 1 -cs '0 0 0 0.1 2e-11' -rs 2 -maxit 4
// mpirun -np 4 tesla -m ../../data/fichera.mesh -maxit 4 -rs 3 -bm '-0.1 -0.1 -0.1 0.1 0.1 0.1 0.1 -1e10'
// mpirun -np 4 lorentz -er Volta-AMR-Parallel -ec 4 -br Tesla-AMR-Parallel -bc 4 -q -10 -dt 1e-4 -nt 2000 -npt 500 -vt 10 -rdf 500 -rdm 1 -vf 10 -pmin '-8 -4 4' -pmax '-8 -4 4' -xmin '-1 -1 -1' -xmax '1 1 1'
// mpirun -np 4 lorentz -er Volta-AMR-Parallel -ec 4 -br Tesla-AMR-Parallel -bc 4 -q -10 -dt 1e-3 -npt 1 -vt 650 -rdf 500 -rdm 1 -vf 2 -pmin '-8 -4 4' -pmax '-8 -4 4' -xmin '0.8 0 0' -xmax '0.8 0 0' -nt 1300
// mpirun -np 4 lorentz -er Volta-AMR-Parallel -ec 4 -br Tesla-AMR-Parallel -bc 4 -x0 '0.8 0 0' -p0 '-8 -4 4' -q -10 -tf 0.2 -dt 1e-3 -rf 1e-6
//
// This miniapp demonstrates the use of the ParMesh::FindPoints functionality
// to evaluate field data from stored DataCollection objects. While this
// miniapp is far from a full particle-in-cell (PIC) code it does demonstrate
// some of the building blocks that might be used to construct the particle
// mover portion of a PIC code.
#include "mfem.hpp"
#include "../common/particles_extras.hpp"
#include "../common/fem_extras.hpp"
#include "../common/pfem_extras.hpp"
#include "electromagnetics.hpp"
#include <fstream>
#include <iostream>
@@ -66,176 +71,250 @@ using namespace mfem;
using namespace mfem::common;
using namespace mfem::electromagnetics;
struct LorentzContext
typedef DataCollection::FieldMapType fields_t;
/// This class implements the Boris algorithm as described in the
/// article `Why is Boris algorithm so good?` by H. Qin et al in
/// Physics of Plasmas, Volume 20 Issue 8, August 2013,
/// https://doi.org/10.1063/1.4818428.
class BorisAlgorithm
{
struct DColl
private:
real_t charge_;
real_t mass_;
ParMesh *E_pmesh_;
ParGridFunction *E_field_;
ParMesh *B_pmesh_;
ParGridFunction *B_field_;
mutable Array<int> elem_id_;
mutable Array<IntegrationPoint> ip_;
mutable Vector E_;
mutable Vector B_;
mutable Vector pxB_;
mutable Vector pm_;
mutable Vector pp_;
// Returns true if a usable V has been found. If @a pgf is NULL, V = 0 is
// returned as a default value.
bool GetValue(ParMesh *pmesh, ParGridFunction *pgf, Vector q, Vector &V)
{
string coll_name;
string field_name;
int cycle;
int pad_digits_cycle;
int pad_digits_rank;
};
DColl E{"", "E", 10, 6, 6};
DColl B{"", "B", 10, 6, 6};
DenseMatrix point(q.GetData(), 3, 1);
int ordering = 1; // 0 - byNODES, 1 - byVDIM
int npt = 1; // total number of particles
real_t q = 1.0; // particle charge
real_t m = 1.0; // particle mass
Vector x_min{-1.0,-1.0,-1.0}; // initial position min
Vector x_max{1.0,1.0,1.0}; // initial position max
Vector p_min{-1.0,-1.0,-1.0}; // initial momentum min
Vector p_max{1.0,1.0,1.0}; // initial momentum max
real_t dt = 1e-2; // time step
int nt = 1000; // number of timesteps
int redist_interval = 5; // redistribution interval
int redist_mesh = 0; // redistribution mesh: 0: E mesh, 1: B mesh
} ctx;
int pt_found =
(pmesh != NULL) ? pmesh->FindPoints(point, elem_id_, ip_, false) : -1;
// We have a mesh but the point was not found. The path must be outside
// the domain of interest.
if (pmesh != NULL && pt_found <= 0) { return false; }
int pt_root = -1;
if (pt_found > 0 && elem_id_[0] >= 0 && pgf != NULL)
{
pt_root = pmesh->GetMyRank();
pgf->GetVectorValue(elem_id_[0], ip_[0], V);
}
else
{
pt_root = 0;
V = 0.0;
}
// Determine processor which found the field point
int glb_pt_root = -1;
MPI_Allreduce(&pt_root, &glb_pt_root, 1,
MPI_INT, MPI_MAX, MPI_COMM_WORLD);
// Send the field value to the root processor
if (pmesh != NULL && elem_id_[0] >= 0 && glb_pt_root != 0)
{
MPI_Send(V.GetData(), 3, MPITypeMap<real_t>::mpi_type,
0, 1030, MPI_COMM_WORLD);
}
// Receive the field value on the root processor
if (Mpi::Root() && pmesh != NULL && glb_pt_root != 0)
{
MPI_Status status;
MPI_Recv(V.GetData(), 3, MPITypeMap<real_t>::mpi_type,
glb_pt_root, 1030, MPI_COMM_WORLD, &status);
}
return true;
}
/// This class implements the Boris algorithm as described in the article
/// `Why is Boris algorithm so good?` by H. Qin et al in Physics of Plasmas,
/// Volume 20 Issue 8, August 2013, https://doi.org/10.1063/1.4818428.
class Boris
{
public:
/// Field indices
/** Allows for convenient access to corresponding ParticleVector from
ParticleSet. */
enum Fields
BorisAlgorithm(ParGridFunction *E_gf,
ParGridFunction *B_gf,
real_t charge, real_t mass)
: charge_(charge), mass_(mass),
E_field_(E_gf),
B_field_(B_gf),
E_(3), B_(3), pxB_(3), pm_(3), pp_(3)
{
MASS, // vdim = 1
CHARGE, // vdim = 1
MOM, // vdim = dim
EFIELD, // vdim = dim
BFIELD // vdim = dim
};
protected:
/// Pointers to E and B field GridFunctions
GridFunction *E_gf = nullptr;
GridFunction *B_gf = nullptr;
E_pmesh_ = (E_field_) ? E_field_->ParFESpace()->GetParMesh() : NULL;
B_pmesh_ = (B_field_) ? B_field_->ParFESpace()->GetParMesh() : NULL;
}
/// FindPointsGSLIB objects for E and B field meshes
FindPointsGSLIB E_finder;
FindPointsGSLIB B_finder;
bool Step(Vector &q, Vector &p, real_t &t, real_t &dt)
{
// Locate current point in each mesh, evaluate the fields, and collect
// field values on the root processor.
if (!GetValue(E_pmesh_, E_field_, q, E_)) { return false; }
if (!GetValue(B_pmesh_, B_field_, q, B_)) { return false; }
/// ParticleSet of charged particles
std::unique_ptr<ParticleSet> charged_particles;
// Compute updated position and momentum using the Boris algorithm
if (Mpi::Root())
{
// Compute half of the contribution from q E
add(p, 0.5 * dt * charge_, E_, pm_);
// Temporary vectors for particle computation
mutable Vector pxB_, pm_, pp_;
// Compute the contributiobn from q p x B
const real_t B2 = B_ * B_;
/// Single particle Boris step
void ParticleStep(Particle &part, real_t &dt);
public:
// ... along pm x B
const real_t a1 = 4.0 * dt * charge_ * mass_;
pm_.cross3D(B_, pxB_);
pp_.Set(a1, pxB_);
Boris(MPI_Comm comm, GridFunction *E_gf_, GridFunction *B_gf_,
int nparticles, Ordering::Type pdata_ordering);
// ... along pm
const real_t a2 = 4.0 * mass_ * mass_ -
dt * dt * charge_ * charge_ * B2;
pp_.Add(a2, pm_);
/// Find Particles in mesh corresponding to E and B fields
void FindParticles();
// ... along B
const real_t a3 = 2.0 * dt * dt * charge_ * charge_ * (B_ * pm_);
pp_.Add(a3, B_);
/// Update E and B fields at particle locations. Must be called
/// right after FindParticles has been called.
void EvaluateFieldsAtParticles();
// scale by common denominator
const real_t a4 = 4.0 * mass_ * mass_ +
dt * dt * charge_ * charge_ * B2;
pp_ /= a4;
/// Advance particles one time step using Boris algorithm
void Step(real_t &t, real_t &dt);
// Update the momentum
add(pp_, 0.5 * dt * charge_, E_, p);
/// Remove lost particles and return their indices
Array<int> RemoveLostParticles();
// Update the position
q.Add(dt / mass_, p);
}
/// Redistribute particles based on \p redist_mesh (0 - E field, 1 - B field)
void Redistribute(int redist_mesh, Array<int> &removed_idxs);
// Update the time
t += dt;
/// Get reference to the ParticleSet of charged particles
ParticleSet& GetParticles() { return *charged_particles; }
// Broadcast the updated position
MPI_Bcast(q.GetData(), 3, MPITypeMap<real_t>::mpi_type,
0, MPI_COMM_WORLD);
/// Get reference to the E field FindPointsGSLIB object
FindPointsGSLIB& GetEFinder() { return E_finder; }
// Broadcast the updated momentum
MPI_Bcast(p.GetData(), 3, MPITypeMap<real_t>::mpi_type,
0, MPI_COMM_WORLD);
return true;
}
};
// Open the named VisItDataCollection and read the named field.
// Returns pointers to the two new objects.
int ReadGridFunction(const char * coll_name, const char * field_name,
int pad_digits_cycle, int pad_digits_rank, int cycle,
VisItDataCollection *&dc, ParGridFunction *& gf);
// By default the initial position will be the center of the intersection
// of the bounding boxes of the meshes containing the E and B fields.
void SetInitialPosition(VisItDataCollection *E_dc,
VisItDataCollection *B_dc,
Vector &x_init);
// Build a quadrilateral mesh approximating the trajectory as a
// ribbon. One edge of the ribbon follows the trajectory of the
// particle. The opposite edge is offset by the acceleration vector
// (scaled by a constant called the r_factor).
Mesh MakeTrajectoryMesh(int step, real_t m, real_t dt, real_t r_factor,
const DenseMatrix &pos_data,
const DenseMatrix &mom_data);
// Prints the program's logo to the given output stream
void display_banner(ostream & os);
// Open the named VisItDataCollection and read the named field.
// Returns pointers to the two new objects.
int ReadGridFunction(std::string coll_name, std::string field_name,
int pad_digits_cycle, int pad_digits_rank, int cycle,
std::unique_ptr<VisItDataCollection> &dc,
ParGridFunction *&gf);
// Initialize particles from user input.
void InitializeChargedParticles(ParticleSet &particles, const Vector &pos_min,
const Vector &pos_max, const Vector &x_init,
const Vector &p_init, real_t m,
real_t q);
int main(int argc, char *argv[])
{
Mpi::Init(argc, argv);
int num_ranks = Mpi::WorldSize();
int rank = Mpi::WorldRank();
Hypre::Init();
if ( Mpi::Root() ) { display_banner(cout); }
bool visualization = true; // enable visualization
int vis_tail_size = 5; // particle trajectory tail size
int vis_interval = 4; // visualization interval
const char *E_coll_name = "";
const char *E_field_name = "E";
int E_cycle = 10;
int E_pad_digits_cycle = 6;
int E_pad_digits_rank = 6;
const char *B_coll_name = "";
const char *B_field_name = "B";
int B_cycle = 10;
int B_pad_digits_cycle = 6;
int B_pad_digits_rank = 6;
real_t q = 1.0;
real_t m = 1.0;
real_t dt = 1e-2;
real_t t_init = 0.0;
real_t t_final = 1.0;
real_t r_factor = -1.0;
Vector x_init;
Vector p_init;
int visport = 19916;
bool visualization = true;
bool visit = true;
OptionsParser args(argc, argv);
args.AddOption(&ctx.E.coll_name, "-er", "--e-root-file",
args.AddOption(&E_coll_name, "-er", "--e-root-file",
"Set the VisIt data collection E field root file prefix.");
args.AddOption(&ctx.E.field_name, "-ef", "--e-field-name",
args.AddOption(&E_field_name, "-ef", "--e-field-name",
"Set the VisIt data collection E field name");
args.AddOption(&ctx.E.cycle, "-ec", "--e-cycle",
args.AddOption(&E_cycle, "-ec", "--e-cycle",
"Set the E field cycle index to read.");
args.AddOption(&ctx.E.pad_digits_cycle, "-epdc", "--e-pad-digits-cycle",
args.AddOption(&E_pad_digits_cycle, "-epdc", "--e-pad-digits-cycle",
"Number of digits in E field cycle.");
args.AddOption(&ctx.E.pad_digits_rank, "-epdr", "--e-pad-digits-rank",
args.AddOption(&E_pad_digits_rank, "-epdr", "--e-pad-digits-rank",
"Number of digits in E field MPI rank.");
args.AddOption(&ctx.B.coll_name, "-br", "--b-root-file",
args.AddOption(&B_coll_name, "-br", "--b-root-file",
"Set the VisIt data collection B field root file prefix.");
args.AddOption(&ctx.B.field_name, "-bf", "--b-field-name",
args.AddOption(&B_field_name, "-bf", "--b-field-name",
"Set the VisIt data collection B field name");
args.AddOption(&ctx.B.cycle, "-bc", "--b-cycle",
args.AddOption(&B_cycle, "-bc", "--b-cycle",
"Set the B field cycle index to read.");
args.AddOption(&ctx.B.pad_digits_cycle, "-bpdc", "--b-pad-digits-cycle",
args.AddOption(&B_pad_digits_cycle, "-bpdc", "--b-pad-digits-cycle",
"Number of digits in B field cycle.");
args.AddOption(&ctx.B.pad_digits_rank, "-bpdr", "--b-pad-digits-rank",
args.AddOption(&B_pad_digits_rank, "-bpdr", "--b-pad-digits-rank",
"Number of digits in B field MPI rank.");
args.AddOption(&ctx.redist_interval, "-rdf", "--redist-interval",
"Redistribution after this many timesteps. 0 means "
"no redistribution.");
args.AddOption(&ctx.redist_mesh, "-rdm", "--redistribution-mesh",
"Particle domain mesh for redistribution. 0 for E field mesh."
" 1 for B field mesh.");
args.AddOption(&ctx.ordering, "-o", "--ordering",
"Ordering of particle data. 0 = byNODES, 1 = byVDIM.");
args.AddOption(&ctx.npt, "-npt", "--num-particles",
"Total number of particles.");
args.AddOption(&ctx.m, "-m", "--mass", "Particles' mass.");
args.AddOption(&ctx.q, "-q", "--charge", "Particles' charge.");
args.AddOption(&ctx.x_min, "-xmin", "--x-min",
"Minimum initial particle location.");
args.AddOption(&ctx.x_max, "-xmax", "--x-max",
"Maximum initial particle location.");
args.AddOption(&ctx.p_min, "-pmin", "--p-min",
"Minimum initial particle momentum.");
args.AddOption(&ctx.p_max, "-pmax", "--p-max",
"Maximum initial particle momentum.");
args.AddOption(&ctx.dt, "-dt", "--time-step", "Time Step.");
args.AddOption(&ctx.nt, "-nt", "--num-timesteps", "Number of timesteps.");
args.AddOption(&q, "-q", "--charge",
"Particle charge.");
args.AddOption(&m, "-m", "--mass",
"Particle mass.");
args.AddOption(&dt, "-dt", "--time-step",
"Time Step.");
args.AddOption(&t_init, "-ti", "--initial-time",
"Initial Time.");
args.AddOption(&t_final, "-tf", "--final-time",
"Final Time.");
args.AddOption(&x_init, "-x0", "--initial-position",
"Initial position.");
args.AddOption(&p_init, "-p0", "--initial-momentum",
"Initial momentum.");
args.AddOption(&r_factor, "-rf", "--ribbon-factor",
"Scale factor for ribbon width (rf * (p1-p0) / (m * dt) "
"where p0 and p1 are computed momenta).");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&vis_tail_size, "-vt", "--vis-tail-size",
"GLVis visualization trajectory truncation tail size.");
args.AddOption(&vis_interval, "-vf", "--vis-interval",
"GLVis visualization update after this many timesteps. "
"0 means no visualization.");
args.AddOption(&visit, "-visit", "--visit", "-no-visit", "--no-visit",
"Enable or disable VisIt visualization.");
args.AddOption(&visport, "-p", "--send-port", "Socket for GLVis.");
args.Parse();
if (!args.Good())
{
@@ -245,310 +324,137 @@ int main(int argc, char *argv[])
}
return 1;
}
if (r_factor <= 0.0)
{
r_factor = dt;
}
if (Mpi::Root())
{
args.PrintOptions(cout);
}
std::unique_ptr<VisItDataCollection> E_dc, B_dc;
ParGridFunction *E_gf = nullptr, *B_gf = nullptr;
Vector bb_xmin, bb_xmax;
VisItDataCollection *E_dc = NULL;
ParGridFunction *E_gf = NULL;
// Read E field if provided
if (ctx.E.coll_name != "")
if (strcmp(E_coll_name, ""))
{
if (ReadGridFunction(ctx.E.coll_name, ctx.E.field_name,
ctx.E.pad_digits_cycle, ctx.E.pad_digits_rank,
ctx.E.cycle, E_dc, E_gf))
if (ReadGridFunction(E_coll_name, E_field_name, E_pad_digits_cycle,
E_pad_digits_rank, E_cycle, E_dc, E_gf))
{
mfem::err << "Error loading E field" << endl;
mfem::out << "Error loading E field" << endl;
return 1;
}
E_gf->ParFESpace()->GetParMesh()->GetBoundingBox(bb_xmin, bb_xmax, 2);
}
// Read B field if provided
if (ctx.B.coll_name != "")
VisItDataCollection *B_dc = NULL;
ParGridFunction *B_gf = NULL;
if (strcmp(B_coll_name, ""))
{
if (ReadGridFunction(ctx.B.coll_name, ctx.B.field_name,
ctx.B.pad_digits_cycle, ctx.B.pad_digits_rank,
ctx.B.cycle, B_dc, B_gf))
if (ReadGridFunction(B_coll_name, B_field_name, B_pad_digits_cycle,
B_pad_digits_rank, B_cycle, B_dc, B_gf))
{
mfem::err << "Error loading B field" << endl;
mfem::out << "Error loading B field" << endl;
return 1;
}
Vector bb_xmint, bb_xmaxt;
B_gf->ParFESpace()->GetParMesh()->GetBoundingBox(bb_xmint, bb_xmaxt, 2);
if (ctx.E.coll_name != "")
{
// compute intersection of bounding boxes
for (int d = 0; d < bb_xmin.Size(); d++)
{
bb_xmin[d] = std::max(bb_xmin[d], bb_xmint[d]);
bb_xmax[d] = std::min(bb_xmax[d], bb_xmaxt[d]);
}
}
else
{
bb_xmin = bb_xmint;
bb_xmax = bb_xmaxt;
}
}
Ordering::Type ordering_type = ctx.ordering == 0 ?
Ordering::byNODES : Ordering::byVDIM;
// Initialize particles
int num_particles = ctx.npt/num_ranks +
(rank < (ctx.npt % num_ranks) ? 1 : 0);
Boris boris(MPI_COMM_WORLD, E_gf, B_gf, num_particles, ordering_type);
InitializeChargedParticles(boris.GetParticles(), ctx.x_min, ctx.x_max,
ctx.p_min, ctx.p_max, ctx.m, ctx.q);
boris.FindParticles();
boris.EvaluateFieldsAtParticles();
real_t t = 0.0;
real_t dt = ctx.dt;
// Setup visualization
char vishost[] = "localhost";
socketstream pre_redist_sock, post_redist_sock;
std::unique_ptr<ParticleTrajectories> traj_vis;
if (visualization)
if (x_init.Size() < 3)
{
const char *keys = "baaa";
traj_vis = std::make_unique<ParticleTrajectories>(boris.GetParticles(),
vis_tail_size,
vishost, 19916,
"Trajectories",
0, 0, 600, 600, keys);
traj_vis->SetVisualizationBoundingBox(bb_xmin, bb_xmax);
SetInitialPosition(E_dc, B_dc, x_init);
}
if (p_init.Size() < 3)
{
p_init.SetSize(3); p_init = 0.0;
}
if (Mpi::Root())
{
mfem::out << "Initial position: "; x_init.Print(mfem::out);
mfem::out << "Initial momentum: "; p_init.Print(mfem::out);
}
for (int step = 1; step <= ctx.nt; step++)
BorisAlgorithm boris(E_gf, B_gf, q, m);
Vector pos(x_init);
Vector mom(p_init);
ofstream ofs("Lorentz.dat");
ofs.precision(14);
int nsteps = 1 + (int)ceil((t_final - t_init) / dt);
DenseMatrix pos_data(3, nsteps);
DenseMatrix mom_data(3, nsteps + 1);
mom_data.SetCol(0, p_init);
if (Mpi::Root())
{
mfem::out << "Maximum number of steps: " << nsteps << endl;
}
int step = -1;
real_t t = t_init;
do
{
// Step the Boris algorithm
boris.Step(t, dt);
if (Mpi::Root())
{
mfem::out << "Step: " << step << " | Time: " << t << endl;
ofs << t
<< '\t' << pos[0] << '\t' << pos[1] << '\t' << pos[2]
<< '\t' << mom[0] << '\t' << mom[1] << '\t' << mom[2]
<< '\n';
}
step++;
// Visualize trajectories
if (visualization && step % vis_interval == 0)
pos_data.SetCol(step, pos);
mom_data.SetCol(step + 1, mom);
}
while (boris.Step(pos, mom, t, dt) && step < nsteps - 1);
if (Mpi::Root() && (visit || visualization))
{
Mesh trajectory = MakeTrajectoryMesh(step, m, dt, r_factor,
pos_data, mom_data);
L2_FECollection fec_l2(0, 2);
FiniteElementSpace fes_l2(&trajectory, &fec_l2);
GridFunction traj_time(&fes_l2);
for (int i=0; i<step; i++)
{
traj_vis->Visualize();
traj_time[i] = dt * i;
}
// Remove lost particles from particle set and output
Array<int> removed_idxs = boris.RemoveLostParticles();
// Redistribute
if (ctx.redist_interval > 0 && step % ctx.redist_interval == 0 &&
boris.GetParticles().GetGlobalNParticles() > 0)
if (visit)
{
// Redistribute particles - prior to redistribution, removed any lost
// particles that were just removed from the set.
boris.Redistribute(ctx.redist_mesh, removed_idxs);
VisItDataCollection visit_dc("Lorentz", &trajectory);
visit_dc.RegisterField("Time", &traj_time);
visit_dc.SetCycle(step);
visit_dc.SetTime(step * dt);
visit_dc.Save();
}
}
}
void Boris::ParticleStep(Particle &part, real_t &dt)
{
Vector &x = part.Coords();
real_t m = part.FieldValue(MASS);
real_t q = part.FieldValue(CHARGE);
Vector &p = part.Field(MOM);
Vector &e = part.Field(EFIELD);
Vector &b = part.Field(BFIELD);
// Compute half of the contribution from q E
add(p, 0.5 * dt * q, e, pm_);
// Compute the contribution from q p x B
const real_t B2 = b * b;
// ... along pm x B
const real_t a1 = 4.0 * dt * q * m;
pm_.cross3D(b, pxB_);
pp_.Set(a1, pxB_);
// ... along pm
const real_t a2 = 4.0 * m * m -
dt * dt * q * q * B2;
pp_.Add(a2, pm_);
// ... along B
const real_t a3 = 2.0 * dt * dt * q * q * (b * pm_);
pp_.Add(a3, b);
// scale by common denominator
const real_t a4 = 4.0 * m * m +
dt * dt * q * q * B2;
pp_ /= a4;
// Update the momentum
add(pp_, 0.5 * dt * q, e, p);
// Update the position
x.Add(dt / m, p);
}
Boris::Boris(MPI_Comm comm, GridFunction *E_gf_, GridFunction *B_gf_,
int nparticles, Ordering::Type pdata_ordering)
: E_gf(E_gf_),
B_gf(B_gf_),
E_finder(comm),
B_finder(comm)
{
MFEM_VERIFY(E_gf || B_gf, "Must pass an E field or B field to Boris.");
Mesh *E_mesh = E_gf ? E_gf->FESpace()->GetMesh() : nullptr;
Mesh *B_mesh = B_gf ? B_gf->FESpace()->GetMesh() : nullptr;
if (E_mesh && B_mesh)
{
int E_dim = E_mesh->SpaceDimension();
int B_dim = B_mesh->SpaceDimension();
MFEM_VERIFY(E_dim == B_dim,
"E mesh and B mesh must have the same spatial dimension.");
}
if (E_gf)
{
E_mesh->EnsureNodes();
E_finder.Setup(*E_mesh);
}
if (B_gf)
{
B_mesh->EnsureNodes();
B_finder.Setup(*B_mesh);
}
int dim = E_mesh ? E_mesh->SpaceDimension() : B_mesh->SpaceDimension();
pxB_.SetSize(dim); pm_.SetSize(dim); pp_.SetSize(dim);
/// Create particle set:
/// 2 scalars of mass and charge,
/// 3 vectors of size space dim for momentum, e field, and b field
Array<int> field_vdims({1, 1, dim, dim, dim});
charged_particles = std::make_unique<ParticleSet>
(comm, nparticles, dim, field_vdims, 0, pdata_ordering);
}
void Boris::FindParticles()
{
ParticleVector &X = charged_particles->Coords();
// Find particles in E and B field meshes
if (E_gf)
{
E_finder.FindPoints(X); // X.GetOrdering() used internally
}
if (B_gf)
{
B_finder.FindPoints(X); // X.GetOrdering() used internally
}
}
void Boris::EvaluateFieldsAtParticles()
{
ParticleVector &E = charged_particles->Field(EFIELD);
ParticleVector &B = charged_particles->Field(BFIELD);
// Interpolate E-field + B-field onto particles
if (E_gf)
{
E_finder.Interpolate(*E_gf, E, E.GetOrdering());
}
else
{
E = 0.0;
}
if (B_gf)
{
B_finder.Interpolate(*B_gf, B, B.GetOrdering());
}
else
{
B = 0.0;
}
}
void Boris::Step(real_t &t, real_t &dt)
{
// Interpolate E and B fields onto particles
EvaluateFieldsAtParticles();
// Individually step each particle. If all ParticleSet fields are ordered
// byVDIM, we can use GetParticleRef for better performance.
if (charged_particles->IsParticleRefValid())
{
for (int i = 0; i < charged_particles->GetNParticles(); i++)
if (visualization)
{
Particle p = charged_particles->GetParticleRef(i);
ParticleStep(p, dt);
socketstream traj_sock;
traj_sock.precision(8);
char vishost[] = "localhost";
int Wx = 0, Wy = 0; // window position
int Ww = 350, Wh = 350; // window size
VisualizeField(traj_sock, vishost, visport,
traj_time, "Trajectory", Wx, Wy, Ww, Wh);
}
}
else
if (Mpi::Root())
{
for (int i = 0; i < charged_particles->GetNParticles(); i++)
{
Particle p = charged_particles->GetParticle(i);
ParticleStep(p, dt);
charged_particles->SetParticle(i, p);
}
mfem::out << "Number of steps taken: " << step << endl;
}
// Find updated particle locations in E and B field meshes
FindParticles();
// Update time
t += dt;
}
Array<int> Boris::RemoveLostParticles()
{
Array<int> lost_idxs;
const Array<int> E_lost = E_finder.GetPointsNotFoundIndices();
const Array<int> B_lost = B_finder.GetPointsNotFoundIndices();
for (const int &elem : E_lost)
{
lost_idxs.Union(elem);
}
for (const int &elem : B_lost)
{
lost_idxs.Union(elem);
}
charged_particles->RemoveParticles(lost_idxs);
return lost_idxs;
}
void Boris::Redistribute(int redist_mesh, Array<int> &removed_idxs)
{
if (redist_mesh == 0 && E_gf)
{
Array<int> proc_list = E_finder.GetProc();
proc_list.DeleteAt(removed_idxs);
charged_particles->Redistribute(proc_list);
}
else
{
Array<int> proc_list = B_finder.GetProc();
proc_list.DeleteAt(removed_idxs);
charged_particles->Redistribute(proc_list);
}
// Find particles again since ParticleSet is not yet synced with
// FindPointsGSLIB objects.
FindParticles();
// Clean up
delete E_dc;
delete B_dc;
}
// Print the Lorentz ascii logo to the given ostream
void display_banner(ostream & os)
{
os << " ____ __ "
@@ -565,22 +471,29 @@ void display_banner(ostream & os)
<< endl << flush;
}
int ReadGridFunction(std::string coll_name, std::string field_name,
int ReadGridFunction(const char * coll_name, const char * field_name,
int pad_digits_cycle, int pad_digits_rank, int cycle,
std::unique_ptr<VisItDataCollection> &dc, ParGridFunction *&gf)
VisItDataCollection *&dc, ParGridFunction *& gf)
{
dc = std::make_unique<VisItDataCollection>(MPI_COMM_WORLD, coll_name);
dc = new VisItDataCollection(MPI_COMM_WORLD, coll_name);
dc->SetPadDigitsCycle(pad_digits_cycle);
dc->SetPadDigitsRank(pad_digits_rank);
dc->Load(cycle);
if (dc->Error() != DataCollection::No_Error)
{
mfem::err << "Error loading VisIt data collection: "
mfem::out << "Error loading VisIt data collection: "
<< coll_name << endl;
return 1;
}
if (dc->GetMesh()->Dimension() < 3)
{
mfem::out << "Field must be defined on a three dimensional mesh"
<< endl;
return 1;
}
if (dc->HasField(field_name))
{
gf = dc->GetParField(field_name);
@@ -589,58 +502,70 @@ int ReadGridFunction(std::string coll_name, std::string field_name,
return 0;
}
void InitializeChargedParticles(ParticleSet &charged_particles,
const Vector &x_min, const Vector &x_max, const Vector &p_min,
const Vector &p_max, real_t m, real_t q)
void SetInitialPosition(VisItDataCollection *E_dc,
VisItDataCollection *B_dc,
Vector &x_init)
{
int dim = charged_particles.Coords().GetVDim();
int rank;
MPI_Comm_rank(charged_particles.GetComm(), &rank);
std::mt19937 gen(rank);
x_init.SetSize(3); x_init = 0.0;
// Set up uniform distribution for position
std::uniform_real_distribution<real_t> real_dist_x(0_r,1_r);
// Set up guassian distribution for momentum. Centered between p_min and
// p_max with 3-sigma range covering the box.
Vector p_center(dim);
add(0.5, p_min, p_max, p_center);
Vector dp = p_max; dp -= p_min; dp *= 1_r/6_r; // 3-sigma range
std::vector<std::normal_distribution<real_t>> norm_dist_p;
for (int d = 0; d < dim; d++)
if (E_dc != NULL || B_dc != NULL)
{
norm_dist_p.emplace_back(p_center[d], dp[d] > 0_r ? dp[d] : 1_r);
}
ParticleVector &X = charged_particles.Coords();
ParticleVector &P = charged_particles.Field(Boris::MOM);
ParticleVector &M = charged_particles.Field(Boris::MASS);
ParticleVector &Q = charged_particles.Field(Boris::CHARGE);
for (int i = 0; i < charged_particles.GetNParticles(); i++)
{
for (int d = 0; d < dim; d++)
Vector E_p_min(3); E_p_min = -infinity();
Vector E_p_max(3); E_p_max = infinity();
if (E_dc != NULL)
{
if (x_min[d] >= x_max[d]) { X(i,d) = x_min[d]; }
else
{
X(i,d) = x_min[d] + real_dist_x(gen)*(x_max[d] - x_min[d]);
}
// Initialize momentum
if (p_min[d] >= p_max[d]) { P(i,d) = p_min[d]; }
else
{
real_t p_val = norm_dist_p[d](gen);
while (p_val < p_min[d] || p_val > p_max[d])
{
p_val = norm_dist_p[d](gen);
}
P(i,d) = p_val;
}
ParMesh * E_pmesh = dynamic_cast<ParMesh*>(E_dc->GetMesh());
E_pmesh->GetBoundingBox(E_p_min, E_p_max);
}
Vector B_p_min(3); B_p_min = -infinity();
Vector B_p_max(3); B_p_max = infinity();
if (B_dc != NULL)
{
ParMesh *B_pmesh = dynamic_cast<ParMesh*>(B_dc->GetMesh());
B_pmesh->GetBoundingBox(B_p_min, B_p_max);
}
for (int d = 0; d<3; d++)
{
const real_t p_min = std::max(E_p_min[d], B_p_min[d]);
const real_t p_max = std::min(E_p_max[d], B_p_max[d]);
x_init[d] = 0.5 * (p_min + p_max);
}
// Initialize mass + charge
M(i) = m;
Q(i) = q;
}
}
Mesh MakeTrajectoryMesh(int step, real_t m, real_t dt, real_t r_factor,
const DenseMatrix &pos_data,
const DenseMatrix &mom_data)
{
Mesh trajectory(2, 2 * (step + 1), step, 0, 3);
for (int i=0; i<=step; i++)
{
trajectory.AddVertex(pos_data(0,i), pos_data(1,i), pos_data(2,i));
real_t dpx = (mom_data(0, i + 1) - mom_data(0, i)) / (m * dt);
real_t dpy = (mom_data(1, i + 1) - mom_data(1, i)) / (m * dt);
real_t dpz = (mom_data(2, i + 1) - mom_data(2, i)) / (m * dt);
trajectory.AddVertex(pos_data(0,i) + r_factor * dpx,
pos_data(1,i) + r_factor * dpy,
pos_data(2,i) + r_factor * dpz);
}
int v[4];
for (int i=0; i<step; i++)
{
v[0] = 2 * i;
v[1] = 2 * (i + 1);
v[2] = 2 * (i + 1) + 1;
v[3] = 2 * i + 1;
trajectory.AddQuad(v);
}
trajectory.FinalizeQuadMesh(1);
return trajectory;
}
+3 -8
View File
@@ -21,10 +21,7 @@ MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
SEQ_MINIAPPS =
PAR_MINIAPPS = volta tesla maxwell joule
ifeq ($(MFEM_USE_GSLIB), YES)
PAR_MINIAPPS += lorentz
endif
PAR_MINIAPPS = volta tesla maxwell joule lorentz
ifeq ($(MFEM_USE_MPI),NO)
MINIAPPS = $(SEQ_MINIAPPS)
else
@@ -54,11 +51,9 @@ all: $(MINIAPPS)
$(MFEM_CXX) $(MFEM_LINK_FLAGS) -o $@ $@.o $@_solver.o $(COMMON_LIB) \
$(MFEM_LIBS)
ifeq ($(MFEM_USE_MPI),YES)
lorentz: %: $(SRC)%.cpp $(MFEM_LIB_FILE) $(CONFIG_MK) | lib-common
$(MFEM_CXX) $(MFEM_FLAGS) -c $(<)
$(MFEM_CXX) $(MFEM_LINK_FLAGS) -o $@ $@.o $(COMMON_LIB) $(MFEM_LIBS)
endif
# Rules for compiling miniapp dependencies
$(addsuffix _solver.o,$(MINIAPPS)): \
@@ -117,10 +112,10 @@ joule-test-par: joule
lorentz-test-par: lorentz-test-1 lorentz-test-2
lorentz-test-1: lorentz volta-test-3
@$(call mfem-test,$<, $(RUN_MPI), Electromagnetic miniapp,\
-er Volta-AMR-Parallel -ec 2 -npt 100 -xmin '0.0 0.0 0.0' -xmax '1.0 1.0 1.0' -pmin '1 0 0' -pmax '1 0 0' -rdf 0 -vt 0 -nt 100')
-er Volta-AMR-Parallel -ec 2 -x0 '0.5 0.5 0.9' -p0 '1 0 0')
lorentz-test-2: lorentz tesla-test-2
@$(call mfem-test,$<, $(RUN_MPI), Electromagnetic miniapp,\
-br Tesla-AMR-Parallel -bc 2 -br Tesla-AMR-Parallel -npt 10 -xmin '0.0 0.0 0.0' -xmax '1.0 1.0 1.0' -pmin '0 0.1 0.05' -pmax '0 0.4 0.1' -nt 1000 -rdf 0 -vt 0)
-br Tesla-AMR-Parallel -bc 2 -x0 '0.1 0.5 0.1' -p0 '0 0.4 0.1' -tf 9)
# Testing: "test" target and mfem-test* variables are defined in config/test.mk
+8 -4
View File
@@ -421,18 +421,22 @@ void NavierParticles::Step(const real_t dt, const ParGridFunction &u_gf,
void NavierParticles::InterpolateUW(const ParGridFunction &u_gf,
const ParGridFunction &w_gf)
{
finder.FindPoints(X());
finder.FindPoints(X(), X().GetOrdering());
finder.Interpolate(u_gf, U(), U().GetOrdering());
finder.Interpolate(u_gf, U());
Ordering::Reorder(U(), U().GetVDim(), u_gf.ParFESpace()->GetOrdering(),
U().GetOrdering());
finder.Interpolate(w_gf, W(), W().GetOrdering());
finder.Interpolate(w_gf, W());
Ordering::Reorder(W(), W().GetVDim(), w_gf.ParFESpace()->GetOrdering(),
W().GetOrdering());
}
void NavierParticles::DeactivateLostParticles(bool findpts)
{
if (findpts)
{
finder.FindPoints(X());
finder.FindPoints(X(), X().GetOrdering());
}
const Array<unsigned int> lost_idxs = finder.GetPointsNotFoundIndices();
@@ -1,489 +0,0 @@
MFEM NURBS mesh v1.0
#
# MFEM Geometry Types (see fem/geom.hpp):
#
# SEGMENT = 1
# SQUARE = 3
# CUBE = 5
#
dimension
3
elements
1
1 5 0 1 2 3 4 5 6 7
boundary
6
1 3 2 1 0 3
1 3 4 5 6 7
1 3 0 1 5 4
1 3 1 2 6 5
1 3 2 3 7 6
1 3 3 0 4 7
edges
12
0 0 1
0 3 2
0 4 5
0 7 6
1 0 3
1 1 2
1 4 7
1 5 6
2 0 4
2 1 5
2 2 6
2 3 7
vertices
8
knotvectors
3
2 6 0 0 0 0.25 0.5 0.75 1 1 1
2 6 0 0 0 0.25 0.5 0.75 1 1 1
2 6 0 0 0 0.25 0.5 0.75 1 1 1
weights
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
FiniteElementSpace
FiniteElementCollection: NURBS2
VDim: 3
Ordering: 1
0.0116849 0.100677 0.107741
0.700841 0.44147 0.344065
0.437989 1.28285 0.264685
-0.303721 0.806601 -0.11583
-0.539413 0.0200674 0.908587
0.395933 0.494981 1.39817
0.0117652 1.43411 1.08871
-0.759897 0.891295 0.785367
0.0812991 0.127468 0.125781
0.255008 0.202244 0.188147
0.448525 0.285255 0.254138
0.625506 0.391535 0.323138
0.34086 1.23115 0.189627
0.135768 1.14038 0.0949877
-0.0571106 1.04065 -0.00204767
-0.225607 0.894706 -0.0866114
-0.3804 0.0431261 0.932761
-0.121171 0.133223 1.01883
0.0987705 0.257745 1.15354
0.299995 0.410817 1.30949
-0.0928372 1.36692 1.04319
-0.292608 1.22409 0.954962
-0.47 1.08105 0.868931
-0.663376 0.956914 0.807106
-0.0313904 0.173208 0.0241675
-0.116453 0.335806 -0.0797544
-0.197708 0.522984 -0.143019
-0.268697 0.719567 -0.147642
0.693218 0.561638 0.333951
0.642943 0.782839 0.330385
0.567332 1.00697 0.306227
0.477659 1.20209 0.287077
-0.550848 0.145891 0.873495
-0.617164 0.379504 0.811408
-0.684941 0.59523 0.782838
-0.736288 0.802334 0.780498
0.357892 0.575016 1.31896
0.285779 0.784367 1.21896
0.185794 1.01755 1.15046
0.0653891 1.28559 1.09951
-0.0537529 0.0832559 0.179502
-0.188121 0.0603262 0.356393
-0.323693 0.0343845 0.566552
-0.463087 0.0213273 0.787315
0.675777 0.435988 0.449458
0.610746 0.444971 0.684641
0.542159 0.473619 0.947334
0.451733 0.481779 1.235
0.387085 1.30441 0.330155
0.300155 1.32346 0.502159
0.197236 1.33735 0.733646
0.0778968 1.38129 0.967056
-0.364395 0.836177 -0.0380904
-0.499556 0.880443 0.175051
-0.618562 0.899155 0.419916
-0.729908 0.894583 0.658954
-0.191762 0.792322 -0.0838227
-0.0127656 0.922096 0.0198956
0.169871 1.04169 0.106131
0.374421 1.15606 0.221327
-0.107725 0.590479 -0.0701237
0.0797398 0.719117 0.052235
0.248024 0.837672 0.153949
0.4506 0.951982 0.257914
-0.014591 0.399524 -0.0133179
0.169119 0.524685 0.0962864
0.349821 0.64291 0.208715
0.547914 0.743218 0.290306
0.0485967 0.219259 0.0598748
0.22429 0.311269 0.147327
0.424681 0.420235 0.239975
0.603974 0.511425 0.314819
0.0215703 0.115709 0.220091
0.202775 0.192498 0.292796
0.404824 0.300707 0.373618
0.587761 0.396426 0.424654
-0.0813987 0.0906892 0.402963
0.122699 0.181167 0.495311
0.320206 0.283771 0.586703
0.513593 0.396941 0.659549
-0.192037 0.0641737 0.611853
0.0238142 0.153445 0.708339
0.217314 0.279993 0.795557
0.427183 0.403976 0.895618
-0.321216 0.0526643 0.821777
-0.0668994 0.132414 0.92639
0.136671 0.264444 1.03878
0.346957 0.403474 1.16556
0.652713 0.556232 0.447968
0.597625 0.792616 0.438187
0.516989 1.00686 0.408826
0.435038 1.2089 0.363187
0.570735 0.56432 0.677065
0.500648 0.786806 0.651401
0.416515 1.01355 0.598304
0.325952 1.21023 0.541362
0.493342 0.572126 0.917974
0.405211 0.78473 0.869844
0.312137 1.00883 0.809381
0.239854 1.22498 0.760211
0.407138 0.581025 1.17298
0.318312 0.777059 1.09777
0.225803 1.01555 1.0397
0.121726 1.25049 0.981676
0.290702 1.25013 0.278809
0.0701636 1.14755 0.189221
-0.119108 1.04886 0.103803
-0.289182 0.91921 0.00804675
0.185344 1.27637 0.482987
-0.0241763 1.17861 0.405915
-0.228914 1.06728 0.323432
-0.408653 0.937665 0.225694
0.0857194 1.30362 0.704401
-0.122501 1.20271 0.644137
-0.324492 1.07716 0.554526
-0.523359 0.958708 0.456627
-0.0299893 1.32937 0.937238
-0.231056 1.22639 0.854007
-0.4277 1.08064 0.77448
-0.624632 0.962435 0.693615
-0.335332 0.743245 -0.0361732
-0.264988 0.556883 -0.0142987
-0.179867 0.361039 0.028428
-0.0895202 0.170043 0.122851
-0.46712 0.787684 0.186278
-0.399567 0.586191 0.200671
-0.301046 0.368151 0.251277
-0.218868 0.158789 0.317644
-0.59138 0.804574 0.412544
-0.506716 0.60204 0.430758
-0.435083 0.376564 0.464743
-0.356472 0.147414 0.523001
-0.689157 0.806058 0.647632
-0.629068 0.593243 0.655094
-0.555738 0.371114 0.695893
-0.485726 0.146879 0.754833
-0.417228 0.17436 0.899793
-0.169866 0.253389 0.990733
0.028903 0.366525 1.11301
0.252098 0.500703 1.24433
-0.500923 0.414148 0.856523
-0.284378 0.495056 0.948388
-0.0843163 0.601653 1.05541
0.155216 0.712225 1.16544
-0.58149 0.643576 0.830741
-0.373319 0.748859 0.924034
-0.18065 0.845261 1.0198
0.0460522 0.952923 1.10623
-0.637619 0.848756 0.803888
-0.448156 0.969832 0.881928
-0.260587 1.09929 0.976527
-0.0459431 1.21968 1.05279
-0.0137524 0.209301 0.168941
0.169242 0.299016 0.260497
0.377537 0.400892 0.353247
0.570084 0.506608 0.425317
-0.080203 0.408135 0.0918812
0.105504 0.518748 0.204877
0.298886 0.634349 0.305087
0.493708 0.735607 0.396409
-0.168586 0.619614 0.0360283
0.0165872 0.734724 0.158097
0.205366 0.847514 0.254686
0.405389 0.959363 0.360348
-0.258677 0.823089 0.0161381
-0.0786817 0.945242 0.11815
0.121369 1.0512 0.208493
0.324175 1.15154 0.303941
-0.124414 0.198947 0.367495
0.0754223 0.285305 0.478578
0.271456 0.388643 0.569286
0.47429 0.500746 0.649904
-0.214578 0.415337 0.300568
-0.0214509 0.508707 0.419948
0.191156 0.607363 0.514446
0.397619 0.729213 0.61556
-0.302858 0.639951 0.251014
-0.103789 0.739391 0.357589
0.110001 0.848732 0.463479
0.318473 0.955375 0.556868
-0.385766 0.851741 0.222241
-0.192006 0.963536 0.318999
0.024575 1.07461 0.420919
0.234314 1.16838 0.513436
-0.239144 0.176376 0.576763
-0.0256959 0.27234 0.678898
0.180326 0.386088 0.77861
0.380616 0.507244 0.864894
-0.336051 0.4068 0.522487
-0.134268 0.49507 0.619439
0.0937046 0.598445 0.720534
0.295642 0.716778 0.815371
-0.413699 0.644524 0.471466
-0.217717 0.723955 0.573321
0.00422339 0.839902 0.669769
0.207017 0.943147 0.772738
-0.496887 0.853244 0.456225
-0.291644 0.956378 0.557542
-0.0924407 1.0788 0.643644
0.1296 1.17577 0.725705
-0.366417 0.168186 0.785706
-0.121768 0.262554 0.893007
0.083928 0.379718 1.0059
0.297426 0.504658 1.11483
-0.443726 0.407516 0.735498
-0.240641 0.495966 0.83729
-0.0238202 0.596216 0.939116
0.203993 0.722308 1.04178
-0.524688 0.636263 0.710265
-0.331783 0.741903 0.804337
-0.117384 0.835587 0.896008
0.103953 0.951344 0.987382
-0.599196 0.860865 0.697438
-0.398317 0.964829 0.780767
-0.195503 1.08864 0.858335
0.0177754 1.19986 0.938708
@@ -1,118 +0,0 @@
MFEM NURBS mesh v1.0
#
# MFEM Geometry Types (see fem/geom.hpp):
#
# SEGMENT = 1
# SQUARE = 3
# CUBE = 5
#
dimension
2
elements
1
1 3 0 1 2 3
boundary
4
1 1 0 1
2 1 2 3
3 1 3 0
4 1 1 2
edges
4
0 0 1
0 3 2
1 0 3
1 1 2
vertices
4
knotvectors
2
2 6 0 0 0 0.25 0.5 0.75 1 1 1
2 6 0 0 0 0.25 0.5 0.75 1 1 1
weights
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
1
FiniteElementSpace
FiniteElementCollection: NURBS2
VDim: 2
Ordering: 1
0.0163925 0.141238
0.774637 0.626247
-0.147699 1.3396
-0.757759 0.550541
0.121943 0.231571
0.272418 0.394336
0.420152 0.532036
0.635666 0.624585
-0.261202 1.30275
-0.454309 1.14438
-0.593397 0.942458
-0.710473 0.706781
-0.111803 0.190859
-0.313132 0.306672
-0.51706 0.436229
-0.67826 0.509765
0.608023 0.786507
0.372822 1.01006
0.159851 1.1653
-0.0696727 1.29923
-0.00322359 0.290759
0.158563 0.459956
0.321006 0.615434
0.509901 0.715169
-0.240232 0.408664
-0.0626107 0.581669
0.136422 0.738867
0.308415 0.910906
-0.452041 0.542364
-0.263077 0.727566
-0.0801052 0.906599
0.0851199 1.07157
-0.624784 0.659372
-0.470927 0.866487
-0.318204 1.05325
-0.134756 1.23187
+3 -16
View File
@@ -18,10 +18,6 @@
// nurbs_ex1 -m meshes/two-cubes-nurbs-rot.mesh -o 1 -r 3 -rf meshes/two-cubes.ref
// nurbs_ex1 -m meshes/two-cubes-nurbs-autoedge.mesh -o 1 -r 3 -rf meshes/two-cubes.ref
// nurbs_ex1 -m ../../data/segment-nurbs.mesh -r 2 -o 2 -lod 3
// nurbs_ex1 -m meshes/square-nurbs-deformed.mesh -o 2
// nurbs_ex1 -m meshes/square-nurbs-deformed.mesh -o 2 -no-ibp
// nurbs_ex1 -m meshes/cube-nurbs-deformed.mesh -o 2
// nurbs_ex1 -m meshes/cube-nurbs-deformed.mesh -o 2 -no-ibp
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Poisson problem
@@ -557,18 +553,9 @@ int main(int argc, char *argv[])
}
// 14. Save data in the VisIt format
if (ibp)
{
VisItDataCollection visit_dc("Example1", mesh);
visit_dc.RegisterField("solution", &x);
visit_dc.Save();
}
else
{
VisItDataCollection visit_dc("Example1_nibp", mesh);
visit_dc.RegisterField("solution", &x);
visit_dc.Save();
}
VisItDataCollection visit_dc("Example1", mesh);
visit_dc.RegisterField("solution", &x);
visit_dc.Save();
// 15. Free the used memory.
delete a;
-3
View File
@@ -43,9 +43,6 @@ add_mfem_miniapp(convert-dc
add_mfem_miniapp(lor-transfer
MAIN lor-transfer.cpp LIBRARIES mfem)
add_mfem_miniapp(compare-dc
MAIN compare-dc.cpp LIBRARIES mfem)
add_mfem_miniapp(tmop-check-metric
MAIN tmop-check-metric.cpp LIBRARIES mfem)
-166
View File
@@ -1,166 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
//
// -------------------------------------------------------------------
// Compare DC Miniapp: Compare fields saved via DataCollection classes
// -------------------------------------------------------------------
//
// This miniapp loads previously saved data and computes the l2 norm of the
// difference. Currently, only the VisItDataCollection class is supported.
//
// Compile with: make compare-dc
//
// Serial sample runs:
// > compare-dc -r0 ../../examples/Example5 -r1 ../../examples/alt/Example5
// > compare-dc -r0 Example5 -r1 alt/Example5 -tol 1e-6
//
// Parallel sample runs:
// > mpirun -np 4 compare-dc -r0 ../../examples/Example5-Parallel
// -r1 ../../examples/alt/Example5-Parallel
//
// NB: when no tolerance is provided the difference is simple reported.
// If a tolerance is provided this is compared with the symmetric
// relative difference. An error is given if difference exceeds the tolerance.
#include "mfem.hpp"
using namespace std;
using namespace mfem;
int main(int argc, char *argv[])
{
#ifdef MFEM_USE_MPI
Mpi::Init();
if (!Mpi::Root()) { mfem::out.Disable(); mfem::err.Disable(); }
Hypre::Init();
#endif
// Parse command-line options.
const char *coll_name0 = NULL;
const char *coll_name1 = NULL;
int cycle = 0;
int pad_digits_cycle = 6;
int pad_digits_rank = 6;
real_t tol = -1;
OptionsParser args(argc, argv);
args.AddOption(&coll_name0, "-r0", "--root-file_0",
"Set the VisIt data collection root file prefix.", true);
args.AddOption(&coll_name1, "-r1", "--root-file_1",
"Set the VisIt data collection root file prefix.", true);
args.AddOption(&cycle, "-c", "--cycle", "Set the cycle index to read.");
args.AddOption(&pad_digits_cycle, "-pdc", "--pad-digits-cycle",
"Number of digits in cycle.");
args.AddOption(&pad_digits_rank, "-pdr", "--pad-digits-rank",
"Number of digits in MPI rank.");
args.AddOption(&tol, "-tol", "--tolerance",
"Tolerance for checking the results.");
args.Parse();
if (!args.Good())
{
args.PrintUsage(mfem::out);
return 1;
}
args.PrintOptions(mfem::out);
#ifdef MFEM_USE_MPI
VisItDataCollection dc0(MPI_COMM_WORLD, coll_name0);
#else
VisItDataCollection dc0(coll_name0);
#endif
dc0.SetPadDigitsCycle(pad_digits_cycle);
dc0.SetPadDigitsRank(pad_digits_rank);
dc0.Load(cycle);
if (dc0.Error() != DataCollection::No_Error)
{
mfem::out << "Error loading VisIt data collection: " << coll_name0 << endl;
return 1;
}
#ifdef MFEM_USE_MPI
VisItDataCollection dc1(MPI_COMM_WORLD, coll_name1);
#else
VisItDataCollection dc1(coll_name1);
#endif
dc1.SetPadDigitsCycle(pad_digits_cycle);
dc1.SetPadDigitsRank(pad_digits_rank);
dc1.Load(cycle);
if (dc1.Error() != DataCollection::No_Error)
{
mfem::out << "Error loading VisIt data collection: " << coll_name1 << endl;
return 1;
}
typedef DataCollection::FieldMapType fields_t;
const fields_t &fields0 = dc0.GetFieldMap();
// Print the names of all fields.
bool error = false;
for (fields_t::const_iterator it0 = fields0.begin();
it0 != fields0.end() ; ++it0)
{
GridFunction *gf0 = dc0.GetField(it0->first);
if (!gf0)
{
mfem::out << "Error loading:"<<it0->first<< endl;
mfem::out << "From data collection: " << coll_name0 << endl;
return 1;
}
GridFunction *gf1 = dc1.GetField(it0->first);
if (!gf1)
{
mfem::out << "Error loading:"<<it0->first<< endl;
mfem::out << "From data collection: " << coll_name1 << endl;
return 1;
}
if (gf0->Size() != gf1->Size())
{
mfem::out << "Size error for:"<<it0->first<< endl;
mfem::out << "In data collection: " << coll_name0
<<" size is "<<gf0->Size()<< endl;
mfem::out << "In data collection: " << coll_name1
<<" size is "<<gf1->Size()<< endl;
return 1;
}
// Norm of vectors
real_t nrm0 = gf0->Norml2();
real_t nrm1 = gf1->Norml2();
// Difference
(*gf0) -= (*gf1);
real_t nrmd = gf0->Norml2();
real_t rel_sym = 2*nrmd/(nrm0 + nrm1);
if (gf0->Norml2() > rel_sym) { error = true; }
// Report
mfem::out <<"==========================================="<<std::endl;
mfem::out <<"|"<<it0->first<<"_0| = "<<nrm0<<std::endl;
mfem::out <<"|"<<it0->first<<"_1| = "<<nrm1<<std::endl;
mfem::out <<"\n|"<<it0->first<<"_0 - "<<it0->first<<"_1| = "<<nrmd <<std::endl;
mfem::out <<"\n2|"<<it0->first<<"_0 - "<<it0->first<<"_1|"<<std::endl;
mfem::out << std::setfill('-') << std::setw(15 + 2*it0->first.length())
<<" = "<<rel_sym<<std::endl;
mfem::out <<"(|"<<it0->first<<"_0| + |"<<it0->first<<"_1|)\n"<<std::endl;
}
if (error && tol > 0.0)
{
mfem::out << "Data collections: " << coll_name0
<< " & " << coll_name1 << " are outside of the tolerance!\n";
return -1;
}
return 0;
}
+2 -3
View File
@@ -21,7 +21,7 @@ MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
SEQ_MINIAPPS = display-basis load-dc convert-dc get-values lor-transfer \
tmop-check-metric tmop-metric-magnitude compare-dc
tmop-check-metric tmop-metric-magnitude
PAR_MINIAPPS = nodal-transfer plor-transfer gridfunction-bounds
@@ -78,8 +78,7 @@ RUN_MPI = $(MFEM_MPIEXEC) $(MFEM_MPIEXEC_NP) $(MFEM_MPI_NP)
# Testing: Specific execution options
# Do not test: display-basis, load-dc, convert-dc, get-values, lor-transfer, plor-transfer
NO_TEST_APPS = display-basis load-dc convert-dc get-values lor-transfer \
plor-transfer tmop-check-metric tmop-metric-magnitude gridfunction-bounds \
compare-dc
plor-transfer tmop-check-metric tmop-metric-magnitude gridfunction-bounds
$(foreach app,$(NO_TEST_APPS),$(app)-test-seq $(app)-test-par):
@true
+5
View File
@@ -31,6 +31,11 @@ function(add_benchmark name)
set_property(SOURCE ${${NAME}_BENCH_SRCS} PROPERTY LANGUAGE CUDA)
endif(MFEM_USE_CUDA)
if (MFEM_USE_HIP)
set_property(SOURCE ${${NAME}_BENCH_SRCS} PROPERTY LANGUAGE
HIP_SOURCE_PROPERTY_FORMAT TRUE)
endif(MFEM_USE_HIP)
add_executable(bench_${name} ${${NAME}_BENCH_SRCS})
target_link_libraries(bench_${name} mfem pthread)
add_dependencies(${MFEM_ALL_BENCHMARKS_TARGET_NAME} bench_${name})
+114 -229
View File
@@ -8,89 +8,23 @@
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
//
//
// This benchmark contains the implementation of the CEED's bake-off problems:
// high-order kernels/benchmarks designed to test and compare the performance
// of high-order codes.
//
// See: https://ceed.exascaleproject.org/bps
#include "bench.hpp" // IWYU pragma: keep
#include "bench.hpp"
#ifdef MFEM_USE_BENCHMARK
#include <cassert>
#include <string>
/*
This benchmark contains the implementation of the CEED's bake-off problems:
high-order kernels/benchmarks designed to test and compare the performance
of high-order codes.
#include "fem/qinterp/det.hpp" // IWYU pragma: keep
#include "fem/qinterp/grad.hpp" // IWYU pragma: keep
#include "fem/integ/lininteg_domain_kernels.hpp" // IWYU pragma: keep
#include "fem/integ/bilininteg_vecdiffusion_pa.hpp" // IWYU pragma: keep
// Custom benchmark arguments generator
static void CustomArguments(bmi::Benchmark *b) noexcept
{
constexpr int MAX_NDOFS = 16 * 1024 * (mfem_use_gpu ? 1024 : 8);
const auto orders = { 7, 6, 5, 4, 3, 2, 1 };
constexpr auto ndofs = [](int n) constexpr noexcept -> int
{
return (n + 1) * (n + 1) * (n + 1);
};
constexpr auto inc = [](int n) constexpr noexcept -> int
{
return n < 160 ? 4 : n < 240 ? 8 : n < 320 ? 16 : 32;
};
for (auto p : orders)
{
for (int n = 16; ndofs(n) <= MAX_NDOFS; n += inc(n))
{
b->Args({p, n});
}
}
}
// Register kernel specializations used in the benchmarks
static void AddKernelSpecializations()
{
using DET = QuadratureInterpolator::DetKernels;
DET::Specialization<3, 3, 2, 2>::Add();
DET::Specialization<3, 3, 2, 3>::Add();
DET::Specialization<3, 3, 2, 5>::Add();
DET::Specialization<3, 3, 2, 6>::Add();
DET::Specialization<3, 3, 5, 5>::Add();
// Others might exceed memory limits
using GRAD = QuadratureInterpolator::GradKernels;
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 2>::Add();
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 7>::Add();
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 8>::Add();
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 9>::Add();
using LIN = DomainLFIntegrator::AssembleKernels;
LIN::Specialization<3, 7, 7>::Add();
LIN::Specialization<3, 6, 6>::Add();
LIN::Specialization<3, 8, 8>::Add();
using VDIFF = VectorDiffusionIntegrator::ApplyPAKernels;
VDIFF::Specialization<3, 3, 3, 3>::Add();
VDIFF::Specialization<3, 3, 4, 4>::Add();
VDIFF::Specialization<3, 3, 5, 5>::Add();
VDIFF::Specialization<3, 3, 6, 6>::Add();
VDIFF::Specialization<3, 3, 7, 7>::Add();
VDIFF::Specialization<3, 3, 8, 8>::Add();
}
// Bake-off base class
template <int BFI, int VDIM, bool GLL>
See: ceed.exascaleproject.org/bps and github.com/CEED/benchmarks
*/
template <int VDIM, bool GLL>
struct BakeOff
{
inline static constexpr int DIM = 3;
const int p, c, q, n, nx, ny, nz;
static constexpr int DIM = 3;
const int N, p, q;
Mesh mesh;
H1_FECollection fec;
FiniteElementSpace fes;
@@ -104,15 +38,12 @@ struct BakeOff
GridFunction x, y;
BilinearForm a;
double mdofs{};
BilinearFormIntegrator *bfi;
BakeOff(int p, int side):
p(p), c(side), q(2 * p + (GLL ? -1 : 3)),
n((assert(c >= p), c / p)),
nx(n + (p * (n + 1) * p * n * p * n < c * c * c ? 1 : 0)),
ny(n + (p * (n + 1) * p * (n + 1) * p * n < c * c * c ? 1 : 0)),
nz(n),
mesh(Mesh::MakeCartesian3D(nx, ny, nz, Element::HEXAHEDRON)),
BakeOff(int p):
N(Device::IsEnabled() ? 32 : 4),
p(p),
q(2 * p + (GLL ? -1 : 3)),
mesh(Mesh::MakeCartesian3D(N, N, N, Element::HEXAHEDRON)),
fec(p, DIM, BasisType::GaussLobatto),
fes(&mesh, &fec, VDIM, VDIM == 3 ? Ordering::byVDIM : Ordering::byNODES),
geom_type(mesh.GetTypicalElementGeometry()),
@@ -127,41 +58,22 @@ struct BakeOff
a(&fes)
{
x = 0.0;
if constexpr (BFI == 1)
{
bfi = new MassIntegrator(one, ir);
}
else if constexpr (BFI == 2)
{
bfi = new VectorMassIntegrator(one, ir);
}
else if constexpr (BFI == 3 || BFI == 5)
{
bfi = new DiffusionIntegrator(one, ir);
}
else if constexpr (BFI == 4 || BFI == 6)
{
bfi = new VectorDiffusionIntegrator(one, ir);
}
else
{
static_assert(BFI >= 1 && BFI <= 6, "Invalid BilinearFormIntegrator");
}
a.AddDomainIntegrator(bfi);
}
virtual void benchmark() = 0;
[[nodiscard]] double SumMdofs() const noexcept { return mdofs; }
double SumMdofs() const { return mdofs; }
[[nodiscard]] double MDofs() const noexcept { return 1e-6 * dofs; }
double MDofs() const { return 1e-6 * dofs; }
};
// Bake-off Problems (BPs)
template <int BFI, int VDIM, bool GLL>
struct BP : public BakeOff<BFI, VDIM, GLL>
/// Bake-off Problems (BPs)
template <typename BFI, int VDIM, bool GLL>
struct Problem : public BakeOff<VDIM, GLL>
{
const int max_it = 32, print_lvl = -1;
const double rtol = 1e-12;
const int max_it = 32;
const int print_lvl = -1;
Array<int> ess_tdof_list;
Array<int> ess_bdr;
@@ -170,56 +82,44 @@ struct BP : public BakeOff<BFI, VDIM, GLL>
Vector B, X;
CGSolver cg;
using base = BakeOff<BFI, VDIM, GLL>;
using base::a;
using base::ir;
using base::one;
using base::mesh;
using base::fes;
using base::x;
using base::y;
using base::mdofs;
using base::unit_vec;
using base::bfi;
using BakeOff<VDIM, GLL>::a;
using BakeOff<VDIM, GLL>::ir;
using BakeOff<VDIM, GLL>::one;
using BakeOff<VDIM, GLL>::mesh;
using BakeOff<VDIM, GLL>::fes;
using BakeOff<VDIM, GLL>::x;
using BakeOff<VDIM, GLL>::y;
using BakeOff<VDIM, GLL>::mdofs;
BP(int p, int side) noexcept: base(p, side),
Problem(int order):
BakeOff<VDIM, GLL>(order),
ess_bdr(mesh.bdr_attributes.Max()),
b(&fes)
{
ess_bdr = 1;
fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
if constexpr (VDIM == 1)
if (VDIM == 1)
{
b.AddDomainIntegrator(new DomainLFIntegrator(one));
b.AddDomainIntegrator(new DomainLFIntegrator(this->one));
}
else
{
b.AddDomainIntegrator(new VectorDomainLFIntegrator(unit_vec));
b.AddDomainIntegrator(new VectorDomainLFIntegrator(this->unit_vec));
}
b.UseFastAssembly(true);
b.Assemble();
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
a.AddDomainIntegrator(new BFI(one, ir));
a.Assemble();
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
cg.SetRelTol(rtol);
cg.SetOperator(*A);
cg.SetAbsTol(0.0);
cg.iterative_mode = false;
{
cg.SetPrintLevel(-1);
cg.SetMaxIter(1000);
cg.SetRelTol(1e-8);
cg.Mult(B, X);
MFEM_VERIFY(cg.GetConverged(), "CG solver did not converge!");
}
cg.SetRelTol(0.0);
cg.SetMaxIter(max_it);
cg.SetPrintLevel(print_lvl);
benchmark();
mdofs = 0.0;
cg.iterative_mode = false;
MFEM_DEVICE_SYNC;
}
void benchmark() override
@@ -230,115 +130,104 @@ struct BP : public BakeOff<BFI, VDIM, GLL>
}
};
// Bake-off Kernels (BKs)
template <int BFI, int VDIM, bool GLL>
struct BK : public BakeOff<BFI, VDIM, GLL>
/// Bake-off Problems (BPs)
#define BakeOff_Problem(i, Kernel, VDIM, p_eq_q) \
static void BP##i(bm::State &state) \
{ \
Problem<Kernel##Integrator, VDIM, p_eq_q> ker(state.range(0)); \
while (state.KeepRunning()) { ker.benchmark(); } \
state.counters["MDof/s"] = \
bm::Counter(ker.SumMdofs(), bm::Counter::kIsRate); \
} \
BENCHMARK(BP##i)->DenseRange(1, 6)->Unit(bm::kMillisecond);
/// BP1: scalar PCG with mass matrix, q=p+2
BakeOff_Problem(1, Mass, 1, false)
/// BP2: vector PCG with mass matrix, q=p+2
BakeOff_Problem(2, VectorMass, 3, false)
/// BP3: scalar PCG with stiffness matrix, q=p+2
BakeOff_Problem(3, Diffusion, 1, false)
/// BP4: vector PCG with stiffness matrix, q=p+2
BakeOff_Problem(4, VectorDiffusion, 3, false)
/// BP5: scalar PCG with stiffness matrix, q=p+1
BakeOff_Problem(5, Diffusion, 1, true)
/// BP6: vector PCG with stiffness matrix, q=p+1
BakeOff_Problem(6, VectorDiffusion, 3, true)
/// Bake-off Kernels (BKs)
template <typename BFI, int VDIM, bool GLL>
struct Kernel : public BakeOff<VDIM, GLL>
{
Vector xe, ye;
using BakeOff<VDIM, GLL>::a;
using BakeOff<VDIM, GLL>::ir;
using BakeOff<VDIM, GLL>::one;
using BakeOff<VDIM, GLL>::fes;
using BakeOff<VDIM, GLL>::x;
using BakeOff<VDIM, GLL>::y;
using BakeOff<VDIM, GLL>::mdofs;
using base = BakeOff<BFI, VDIM, GLL>;
using base::ir;
using base::one;
using base::bfi;
using base::fes;
using base::mdofs;
BK(int order, int side) noexcept: base(order, side)
Kernel(int order): BakeOff<VDIM, GLL>(order)
{
bfi->AssemblePA(fes);
const Table &el2dof = fes.GetElementToDofTable();
const int e_size = el2dof.Size_of_connections()*fes.GetVDim();
const auto R = fes.GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC);
MFEM_VERIFY(e_size == R->Height(), "Input/Output E-vector size mismatch!");
xe.SetSize(R->Height());
ye.SetSize(R->Height());
xe.UseDevice(true);
ye.UseDevice(true);
xe.Randomize(1);
xe.Read();
ye = 0.0;
benchmark();
mdofs = 0.0;
x.Randomize(1);
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
a.AddDomainIntegrator(new BFI(one, ir));
a.Assemble();
a.Mult(x, y);
MFEM_DEVICE_SYNC;
}
void benchmark() override
{
bfi->AddMultPA(xe, ye);
a.Mult(x, y);
MFEM_DEVICE_SYNC;
mdofs += this->MDofs();
}
};
// Benchmarks
template <typename T>
static void Benchmark(bm::State& state) noexcept
{
T run(state.range(0), state.range(1));
while (state.KeepRunning()) { run.benchmark(); }
state.counters["Dofs"] = bm::Counter(run.dofs);
state.counters["MDof/s"] = bm::Counter(run.SumMdofs(), bm::Counter::kIsRate);
state.counters["Order"] = bm::Counter(state.range(0));
}
/// Generic CEED BKi
#define BakeOff_Kernel(i, KER, VDIM, GLL) \
static void BK##i(bm::State &state) \
{ \
Kernel<KER##Integrator, VDIM, GLL> ker(state.range(0)); \
while (state.KeepRunning()) { ker.benchmark(); } \
state.counters["MDof/s"] = \
bm::Counter(ker.SumMdofs(), bm::Counter::kIsRate); \
} \
BENCHMARK(BK##i)->DenseRange(1, 6)->Unit(bm::kMillisecond);
#define REGISTER(PK, BFI, VDIM, GLL) \
BENCHMARK_TEMPLATE(Benchmark, PK<BFI, VDIM, GLL>) \
->Name(#PK #BFI)->Apply(CustomArguments)->Unit(bm::kMillisecond)
/// BK1: scalar E-vector-to-E-vector evaluation of mass matrix, q=p+2
BakeOff_Kernel(1, Mass, 1, false)
// BP1: scalar PCG with mass matrix, q=p+2
REGISTER(BP, 1, 1, false);
/// BK2: vector E-vector-to-E-vector evaluation of mass matrix, q=p+2
BakeOff_Kernel(2, VectorMass, 3, false)
// BP2: vector PCG with mass matrix, q=p+2
REGISTER(BP, 2, 3, false);
/// BK3: scalar E-vector-to-E-vector evaluation of stiffness matrix, q=p+2
BakeOff_Kernel(3, Diffusion, 1, false)
// BP3: scalar PCG with stiffness matrix, q=p+2
REGISTER(BP, 3, 1, false);
/// BK4: vector E-vector-to-E-vector evaluation of stiffness matrix, q=p+2
BakeOff_Kernel(4, VectorDiffusion, 3, false)
// BP4: vector PCG with stiffness matrix, q=p+2
REGISTER(BP, 4, 3, false);
/// BK5: scalar E-vector-to-E-vector evaluation of stiffness matrix, q=p+1
BakeOff_Kernel(5, Diffusion, 1, true)
// BP5: scalar PCG with stiffness matrix, q=p+1
REGISTER(BP, 5, 1, true);
// BP6: vector PCG with stiffness matrix, q=p+1
REGISTER(BP, 6, 3, true);
// BK1: scalar E-vector-to-E-vector evaluation of mass matrix, q=p+2
REGISTER(BK, 1, 1, false);
// BK2: vector E-vector-to-E-vector evaluation of mass matrix, q=p+2
REGISTER(BK, 2, 3, false);
// BK3: scalar E-vector-to-E-vector evaluation of stiffness matrix, q=p+2
REGISTER(BK, 3, 1, false);
// BK4: vector E-vector-to-E-vector evaluation of stiffness matrix, q=p+2
REGISTER(BK, 4, 3, false);
// BK5: scalar E-vector-to-E-vector evaluation of stiffness matrix, q=p+1
REGISTER(BK, 5, 1, true);
// BK6: vector E-vector-to-E-vector evaluation of stiffness matrix, q=p+1
REGISTER(BK, 6, 3, true);
/// BK6: vector E-vector-to-E-vector evaluation of stiffness matrix, q=p+1
BakeOff_Kernel(6, VectorDiffusion, 3, true)
/**
* @brief CEED Bake-off Problems main entry point
* Command line options:
* --benchmark_context=device=gpu
* --benchmark_filter=BP1
* --benchmark_out_format=csv
* --benchmark_out=bp1.csv
* @brief main entry point
* --benchmark_filter=BK1/6
* --benchmark_context=device=cpu
*/
int main(int argc, char *argv[])
{
bm::ConsoleReporter CR;
bm::Initialize(&argc, argv);
AddKernelSpecializations();
// Device setup, cpu by default
std::string device_config = "cpu";
auto global_context = bmi::GetGlobalContext();
@@ -351,16 +240,12 @@ int main(int argc, char *argv[])
device_config = device->second;
}
}
Device device(device_config.c_str());
device.Print();
if (bm::ReportUnrecognizedArguments(argc, argv)) { return EXIT_FAILURE; }
if (bm::ReportUnrecognizedArguments(argc, argv)) { return 1; }
bm::RunSpecifiedBenchmarks(&CR);
bm::Shutdown();
return EXIT_SUCCESS;
return 0;
}
#endif // MFEM_USE_BENCHMARK
-1
View File
@@ -101,7 +101,6 @@ set(UNIT_TESTS_SRCS
fem/test_calcdshape.cpp
fem/test_calcshape.cpp
fem/test_calcvshape.cpp
fem/test_calchessian.cpp
fem/test_coefficient.cpp
fem/test_col_lag_der.cpp
fem/test_datacollection.cpp
+1 -5
View File
@@ -320,12 +320,8 @@ TEST_CASE("NormalTraceJumpIntegrator Element Assembly", "[AssemblyLevel][GPU]")
{
const auto fname = GENERATE(
"../../data/inline-quad.mesh",
"../../data/amr-quad.mesh",
"../../data/beam-quad-amr.mesh",
"../../data/star-q3.mesh",
"../../data/inline-hex.mesh",
"../../data/amr-hex.mesh",
"../../data/fichera-amr.mesh",
"../../data/fichera-q3.mesh"
);
const int order = GENERATE(1, 2, 3);
@@ -360,7 +356,7 @@ TEST_CASE("NormalTraceJumpIntegrator Element Assembly", "[AssemblyLevel][GPU]")
for (int f = 0; f < mesh.GetNumFaces(); ++f)
{
const Mesh::FaceInformation info = mesh.GetFaceInformation(f);
if (!info.IsInterior() || info.IsNonconformingCoarse()) { continue; }
if (!info.IsInterior()) { continue; }
const int el1 = info.element[0].index;
const int el2 = info.element[1].index;
-441
View File
@@ -1,441 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "mfem.hpp"
#include "unit_tests.hpp"
#include <iostream>
#include <cmath>
using namespace mfem;
/**
* Compute the error of the taylor series expansion of the shapefunctions, upto
* and including the hessian term:
* res = shape(xi) + dshape(xi)*eps*dx + 0.5*hessian(xi)*eps*eps*dx*dx
* - shape(xi + eps*dx)
*/
real_t TaylorSeriesError(const FiniteElement* fe,
const IntegrationPoint &ip,
const Vector &dx,
const real_t eps)
{
const int dof = fe->GetDof();
const int dim = fe->GetDim();
const int hdim = (dim*(dim+1))/2;
Vector shape(dof);
DenseMatrix dshape(dof,dim);
DenseMatrix hessian(dof,hdim);
fe->CalcShape(ip, shape);
fe->CalcDShape(ip, dshape);
fe->CalcHessian(ip, hessian);
Vector dx2(hdim);
if (dim == 1)
{
dx2[0] = dx[0]*dx[0];
}
else if (dim == 2)
{
dx2[0] = dx[0]*dx[0];
dx2[1] = 2*dx[0]*dx[1];
dx2[2] = dx[1]*dx[1];
}
else if (dim == 3)
{
dx2[0] = dx[0]*dx[0];
dx2[1] = 2*dx[0]*dx[1];
dx2[2] = 2*dx[0]*dx[2];
dx2[3] = dx[1]*dx[1];
dx2[4] = 2*dx[1]*dx[2];
dx2[5] = dx[2]*dx[2];
}
Vector res(dof);
res = shape;
dshape.AddMult(dx, res, eps);
hessian.AddMult(dx2, res, 0.5*eps*eps);
IntegrationPoint ip_eps;
Vector shape_eps(dof);
ip_eps.x = ip.x + eps*dx[0];
if (dim >= 2 ) { ip_eps.y = ip.y + eps*dx[1]; }
if (dim == 3 ) { ip_eps.z = ip.z + eps*dx[2]; }
fe->CalcShape(ip_eps, shape_eps);
res -= shape_eps;
return res.Norml2();
}
/**
* Check the convergence of the taylor series, of a given element @a fe at
* a given point @a ip in a given direction @a dx.
* For linear and quadratic elements the taylor series is exact.
* For other elements the convergence should be third order.
*/
void CheckTaylorSeries(const FiniteElement* fe,
const IntegrationPoint &ip,
const Vector &dx)
{
real_t eps = 0.1;
constexpr real_t red = 4.0;
constexpr int steps = 100;
constexpr real_t tol = 1e-8;
real_t error = TaylorSeriesError(fe, ip, dx, eps);
real_t order;
int i;
for (i = 0; i < steps; ++i)
{
eps /= red;
real_t err_new = TaylorSeriesError(fe, ip, dx, eps);
order = log(error/err_new)/log(red);
error = err_new;
if (error < tol) { break; }
}
mfem::out<<i<<" "<<error<<" "<<order<<std::endl;
if (i == 0)
{
REQUIRE(error == MFEM_Approx(0));
}
else
{
REQUIRE(order > 2.98);
}
}
/**
* Test if a given element @a fe has the correct behaviour of the taylor series.
*/
void TestCalcHessian(const FiniteElement* fe)
{
const int dim = fe->GetDim();
constexpr int check_res = 2;
int num_check_dirs = dim;
// Get a uniform grid of integration points
RefinedGeometry* ref = GlobGeometryRefiner.Refine(fe->GetGeomType(),
check_res);
const IntegrationRule& intRule = ref->RefPts;
int npoints = intRule.GetNPoints();
Vector dx(dim);
for (int i=0; i < npoints; ++i)
{
// Get the current integration point from intRule
IntegrationPoint pt = intRule.IntPoint(i);
for (int j=0; j < num_check_dirs; ++j)
{
dx[0] = sin(2*j + 0.3);
if (dim >= 2) { dx[1] = cos(5*j + 0.2); }
if (dim == 3) { dx[2] = sin(3*j + 0.1); }
CheckTaylorSeries(fe, pt, dx);
}
}
}
TEST_CASE("CalcHessian",
"[Linear1DFiniteElement]"
"[Linear2DFiniteElement]"
"[Linear3DFiniteElement]"
"[BiLinear2DFiniteElement]"
"[TriLinear3DFiniteElement]"
"[H1_SegmentElement]"
"[H1_QuadrilateralElement]"
"[H1_HexahedronElement]"
"[H1_TriangleElement]"
"[H1_TetrahedronElement]"
"[NURBS1DFiniteElement]"
"[NURBS2DFiniteElement]"
"[NURBS3DFiniteElement]")
{
// Fixed Order Elements
SECTION("Linear1DFiniteElement")
{
mfem::out<<"Linear1DFiniteElement"<<std::endl;
Linear1DFiniteElement fe;
TestCalcHessian(&fe);
}
SECTION("Linear2DFiniteElement")
{
mfem::out<<"Linear2DFiniteElement"<<std::endl;
Linear2DFiniteElement fe;
TestCalcHessian(&fe);
}
SECTION("Linear3DFiniteElement")
{
mfem::out<<"Linear3DFiniteElement"<<std::endl;
Linear3DFiniteElement fe;
TestCalcHessian(&fe);
}
SECTION("BiLinear2DFiniteElement")
{
mfem::out<<"BiLinear2DFiniteElement"<<std::endl;
BiLinear2DFiniteElement fe;
TestCalcHessian(&fe);
}
SECTION("TriLinear3DFiniteElement")
{
mfem::out<<"TriLinear3DFiniteElement"<<std::endl;
TriLinear3DFiniteElement fe;
TestCalcHessian(&fe);
}
// H1 Elements
SECTION("H1_SegmentElement")
{
int order = GENERATE(1,2,3,4,5);
mfem::out<<"H1_SegmentElement = "<<order<<std::endl;
H1_SegmentElement fe(order);
TestCalcHessian(&fe);
}
SECTION("H1_QuadrilateralElement")
{
int order = GENERATE(1,2,3,4,5);
H1_QuadrilateralElement fe(order);
mfem::out<<"H1_QuadrilateralElement = "<<order<<std::endl;
TestCalcHessian(&fe);
}
SECTION("H1_HexahedronElement")
{
int order = GENERATE(1,2,3,4,5);
mfem::out<<"H1_HexahedronElement = "<<order<<std::endl;
H1_HexahedronElement fe(order);
TestCalcHessian(&fe);
}
SECTION("H1_TriangleElement")
{
int order = GENERATE(1,2,3,4,5);
mfem::out<<"H1_TriangleElement = "<<order<<std::endl;
H1_TriangleElement fe(order);
TestCalcHessian(&fe);
}
SECTION("H1_TetrahedronElement")
{
int order = GENERATE(1,2,3,4,5);
mfem::out<<"H1_TetrahedronElement = "<<order<<std::endl;
H1_TetrahedronElement fe(order);
TestCalcHessian(&fe);
}
// NURBS Elements
SECTION("NURBS1DFiniteElement")
{
int order = GENERATE(1,2,3,4,5);
mfem::out<<"NURBS1DFiniteElement = "<<order<<std::endl;
NURBS1DFiniteElement fe(order);
Array <const KnotVector*> kv(1);
kv[0] = new KnotVector(order);
fe.KnotVectors() = kv;
int IJK[1];
IJK[0] = 0;
fe.SetIJK(IJK);
fe.SetOrder();
fe.Weights() = 1.0;
TestCalcHessian(&fe);
delete kv[0];
}
SECTION("NURBS2DFiniteElement")
{
int order = GENERATE(1,2,3,4,5);
mfem::out<<"NURBS2DFiniteElement = "<<order<<std::endl;
NURBS2DFiniteElement fe(order);
Array <const KnotVector*> kv(2);
kv[0] = new KnotVector(order);
kv[1] = new KnotVector(order);
fe.KnotVectors() = kv;
int IJK[2];
IJK[0] = IJK[1] = 0;
fe.SetIJK(IJK);
fe.SetOrder();
fe.Weights() = 1.0;
TestCalcHessian(&fe);
delete kv[0];
delete kv[1];
}
SECTION("NURBS3DFiniteElement")
{
int order = GENERATE(1,2,3,4,5);
mfem::out<<"NURBS3DFiniteElement = "<<order<<std::endl;
NURBS3DFiniteElement fe(order);
Array <const KnotVector*> kv(3);
kv[0] = new KnotVector(order);
kv[1] = new KnotVector(order);
kv[2] = new KnotVector(order);
fe.KnotVectors() = kv;
int IJK[3];
IJK[0] = IJK[1] = IJK[2] = 0;
fe.SetIJK(IJK);
fe.SetOrder();
fe.Weights() = 1.0;
TestCalcHessian(&fe);
delete kv[0];
delete kv[1];
delete kv[2];
}
}
TEST_CASE("Laplacian",
"[NURBS2DFiniteElement]"
"[NURBS3DFiniteElement]")
{
int order = 4;
std::string meshName = GENERATE("square-nurbs.mesh",
"cube-nurbs.mesh");
mfem::out<<"\nCheck laplacian for "<< meshName <<std::endl;
bool deformed = GENERATE(false,true);
if (deformed) { mfem::out<<"Mesh is deformed"<<std::endl; }
bool NURBS = GENERATE(false,true);
if (NURBS) { mfem::out<<"Using NURBS"<<std::endl; }
Mesh mesh("../../data/" + meshName, 1, 1);
const int dim = mesh.Dimension();
// Rotate mesh
DenseMatrix Rotate(dim);
if (dim == 2)
{
NURBSPatch::Get2DRotationMatrix(M_PI/7, Rotate);
}
else if (dim == 3)
{
real_t n[] = {0.0,0.0,1.0};
NURBSPatch::Get3DRotationMatrix(n, M_PI/7,M_PI/7, Rotate);
}
Vector x0(dim), x1(dim);
for (int i = 0; i <mesh.GetNodes()->Size()/dim; i++)
{
mesh.GetNode(i, x0.GetData());
Rotate.Mult(x0, x1);
mesh.SetNode(i, x1.GetData());
}
// Distort mesh
real_t distort_scale = 0.05;
if (deformed)
{
Vector dx(mesh.GetNodes()->Size());
dx.Randomize(1234);
dx *= 2.0; dx -= 1.0; dx *= distort_scale;
mesh.MoveNodes(dx);
}
if (NURBS)
{
// We need a C1 smooth mesh
mesh.DegreeElevate(1);
// Refine mesh
mesh.UniformRefinement();
// Distort mesh
distort_scale = 0.01;
if (deformed)
{
Vector dx(mesh.GetNodes()->Size());
dx.Randomize(1234);
dx *= 2.0; dx -= 1.0; dx *= distort_scale;
mesh.MoveNodes(dx);
}
}
// Create Space
FiniteElementCollection *fe_coll = nullptr;
NURBSExtension *ext = nullptr;
if (NURBS)
{
fe_coll = new NURBSFECollection (order);
ext = new NURBSExtension(mesh.NURBSext, order);
}
else
{
fe_coll = new H1_FECollection (order);
}
FiniteElementSpace fes(&mesh, ext, fe_coll);
// Compute (grad w, grad phi) + (w, laplace phi) = 0
SparseMatrix gmat(fes.GetNDofs());
Vector shape, lshape;
DenseMatrix dshape, elmat;
DofTransformation doftrans;
ElementTransformation *eltrans;
Array<int> vdofs;
for (int e = 0; e < fes.GetNE(); e++)
{
const int dof = fes.GetFE(e)->GetDof();
shape.SetSize(dof);
dshape.SetSize(dof,dim);
lshape.SetSize(dof);
elmat.SetSize(dof);
elmat = 0.0;
eltrans = fes.GetElementTransformation (e);
// Integrand involves non-polynomial mapping
const int intorder = 3*fes.GetFE(e)->GetOrder();
const IntegrationRule *ir = &IntRules.Get(fes.GetFE(e)->GetGeomType(),
intorder);
elmat = 0.0;
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
eltrans->SetIntPoint(&ip);
const real_t w = ip.weight * eltrans->Weight();
fes.GetFE(e)->CalcShape(ip, shape);
fes.GetFE(e)->CalcPhysLaplacian(*eltrans, lshape);
fes.GetFE(e)->CalcPhysDShape(*eltrans, dshape);
// Check Laplacian
AddMult_a_AAt (w, dshape, elmat);
AddMult_a_VWt (w, shape, lshape, elmat);
}
// Add to global matrix
fes.GetElementVDofs (e, vdofs);
gmat.AddSubMatrix (vdofs, vdofs, elmat, 1);
}
// Apply homogeneous essential boundary conditions on entire boundary
Array<int> ess_dofs;
fes.GetBoundaryTrueDofs(ess_dofs);
for (int i=0; i<ess_dofs.Size(); i++)
{
gmat.EliminateRowCol(ess_dofs[i], Operator::DiagonalPolicy::DIAG_ZERO);
}
gmat.Finalize (1);
mfem::out<<"Difference between matrices = "<< gmat.MaxNorm() <<std::endl;
// Tolerance can be tighter if intorder is increased
REQUIRE(gmat.MaxNorm() == MFEM_Approx(0.0, 1e-8));
delete fe_coll;
}
+3 -5
View File
@@ -47,14 +47,13 @@ TEST_CASE("GSLIBInterpolate", "[GSLIBInterpolate][GSLIB]")
int point_ordering = GENERATE(0, 1);
int ncomp = GENERATE(1, 2);
int gf_ordering = GENERATE(0, 1);
int func_out_ordering = GENERATE(0, 1);
bool href = GENERATE(true, false);
bool pref = GENERATE(true, false);
int ne = 4;
CAPTURE(space, simplex, dim, func_order, mesh_order, mesh_node_ordering,
point_ordering, ncomp, gf_ordering, func_out_ordering, href, pref);
point_ordering, ncomp, gf_ordering, href, pref);
if (ncomp == 1 && gf_ordering == 1)
{
@@ -146,8 +145,7 @@ TEST_CASE("GSLIBInterpolate", "[GSLIBInterpolate][GSLIB]")
FindPointsGSLIB finder;
finder.Setup(mesh);
finder.SetL2AvgType(FindPointsGSLIB::NONE);
finder.Interpolate(vxyz, field_vals, interp_vals, point_ordering,
func_out_ordering);
finder.Interpolate(vxyz, field_vals, interp_vals, point_ordering);
Array<unsigned int> code_out = finder.GetCode();
Vector dist_p_out = finder.GetDist();
@@ -170,7 +168,7 @@ TEST_CASE("GSLIBInterpolate", "[GSLIBInterpolate][GSLIB]")
{
if (code_out[i] < 2)
{
err = func_out_ordering == Ordering::byNODES ?
err = gf_ordering == Ordering::byNODES ?
fabs(exact_val(j) - interp_vals[i + j*pts_cnt]) :
fabs(exact_val(j) - interp_vals[i*ncomp + j]);
max_err = std::max(max_err, err);
-76
View File
@@ -464,80 +464,4 @@ TEST_CASE("QuadratureInterpolator", "[QuadratureInterpolator][GPU]")
REQUIRE(rel_error_norm == MFEM_Approx(0.0));
}
}
SECTION("Surface Determinants: 1D surface in 2D/3D and 2D surface in 3D")
{
const auto mesh_fname = GENERATE(
"../../data/diag-segment-2d.mesh", // 1D in 2D
"../../data/diag-segment-3d.mesh", // 1D in 3D
"../../data/star-surf.mesh" // 2D in 3D
);
// Using order > 1 to ensure curvature is used if supported by mesh
const int order = 3;
Mesh mesh = Mesh::LoadFromFile(mesh_fname);
const int dim = mesh.Dimension();
const int sdim = mesh.SpaceDimension();
REQUIRE(dim < sdim);
// Ensure high-order curvature for non-trivial Jacobians where possible
mesh.SetCurvature(order);
const FiniteElementSpace *fes = mesh.GetNodalFESpace();
GridFunction *nodes = mesh.GetNodes();
// Quadrature space
QuadratureSpace qs(&mesh, 2*order);
const QuadratureInterpolator *qi = fes->GetQuadratureInterpolator(qs);
qi->SetOutputLayout(QVectorLayout::byVDIM);
// Prepare E-vector from nodes
const ElementDofOrdering ordering =
(mesh.Dimension() == 1 || mesh.MeshGenerator() == 2) ?
ElementDofOrdering::LEXICOGRAPHIC : ElementDofOrdering::NATIVE;
const Operator *R = fes->GetElementRestriction(ordering);
Vector e_vec(R->Height());
R->Mult(*nodes, e_vec);
// Compute determinants (weights) via QI
// Output vector size: qs.GetSize() * 1 (since determinant is scalar)
Vector q_det(qs.GetSize());
qi->Determinants(e_vec, q_det);
// Verify against ElementTransformation::Weight()
Vector q_weights(qs.GetSize());
const int ne = qs.GetNE();
int idx_counter = 0;
for (int i = 0; i < ne; i++)
{
ElementTransformation *T = mesh.GetElementTransformation(i);
const IntegrationRule &ir = qs.GetIntRule(i);
for (int j = 0; j < ir.GetNPoints(); j++)
{
const IntegrationPoint &ip = ir.IntPoint(j);
T->SetIntPoint(&ip);
q_weights(idx_counter++) = T->Weight();
}
}
// Compare
Vector diff = q_det;
diff -= q_weights;
const real_t norm_w = q_weights.Normlinf();
const real_t norm_d = diff.Normlinf();
// If weights are effectively zero (e.g. degenerate), direct comparison might differ
// but for these valid meshes, weight should be > 0.
if (norm_w > 1e-12)
{
REQUIRE(norm_d / norm_w < 1e-12);
}
else
{
REQUIRE(norm_d < 1e-12);
}
}
}