Compare commits
117
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3ad0843586 | ||
|
|
951cf8886b | ||
|
|
9a87b34c47 | ||
|
|
80e40ace14 | ||
|
|
f5d71a2798 | ||
|
|
9981355ba2 | ||
|
|
1416665dc3 | ||
|
|
c9b2ed7a65 | ||
|
|
9eaa3cdf0a | ||
|
|
17d1afc3b7 | ||
|
|
0eaa3b521a | ||
|
|
46d05563d9 | ||
|
|
27c1ea66a4 | ||
|
|
7b69944246 | ||
|
|
fa13c848c9 | ||
|
|
8d09be7080 | ||
|
|
7f71dacae6 | ||
|
|
a7b74155f7 | ||
|
|
0c20eef8fe | ||
|
|
842514bde2 | ||
|
|
ee498a19f7 | ||
|
|
bd5f9d80b4 | ||
|
|
6ee3bbde89 | ||
|
|
92f4fe3bd0 | ||
|
|
0171b4b02d | ||
|
|
68e3a929c2 | ||
|
|
06d18956f8 | ||
|
|
60c2ac77d1 | ||
|
|
ee23534091 | ||
|
|
c0da3d6aa9 | ||
|
|
fef38a9fd2 | ||
|
|
8c68e8402f | ||
|
|
79bca13634 | ||
|
|
5048b1a219 | ||
|
|
b84988d5f1 | ||
|
|
80555fa132 | ||
|
|
9a98c2be01 | ||
|
|
268231dd09 | ||
|
|
10880e5ad2 | ||
|
|
3df2f14eb9 | ||
|
|
63a38bed84 | ||
|
|
16f9cb63a1 | ||
|
|
58da879f06 | ||
|
|
09b9e1c775 | ||
|
|
1ac86956d1 | ||
|
|
4843835f98 | ||
|
|
161630bf12 | ||
|
|
03c99b8dfe | ||
|
|
b65e7ee791 | ||
|
|
473084c8c5 | ||
|
|
2f0bbc6fc4 | ||
|
|
c7cba857af | ||
|
|
85a16c2e43 | ||
|
|
3c4d103982 | ||
|
|
38ed1e049b | ||
|
|
eeca0b4cd5 | ||
|
|
eb4fa33a7b | ||
|
|
42d7d20e43 | ||
|
|
75f8ca8cd4 | ||
|
|
36e915f226 | ||
|
|
f236d70a19 | ||
|
|
85e33c6645 | ||
|
|
7b62f035a5 | ||
|
|
81fb02389f | ||
|
|
23666fd4d8 | ||
|
|
f3a60e2f08 | ||
|
|
2f21794999 | ||
|
|
50b8f67bd1 | ||
|
|
db7da59b03 | ||
|
|
28194a6736 | ||
|
|
7a6313d725 | ||
|
|
84bbead832 | ||
|
|
dbb751d7fb | ||
|
|
58e6c4fc6d | ||
|
|
670dfb2d98 | ||
|
|
4167f0027c | ||
|
|
b59ccd206d | ||
|
|
76b033ad36 | ||
|
|
0cd0a98198 | ||
|
|
411a361ebf | ||
|
|
0ee57ac12d | ||
|
|
dee971c308 | ||
|
|
bb12710561 | ||
|
|
f3276a0d5d | ||
|
|
426b5dc6cd | ||
|
|
0dd81462c0 | ||
|
|
25514d6e8e | ||
|
|
9f01e61a57 | ||
|
|
90fdd7e762 | ||
|
|
cfa6594977 | ||
|
|
08bf7f991b | ||
|
|
0503cbd41c | ||
|
|
74d9671ec8 | ||
|
|
74f64934e8 | ||
|
|
8ff5affe45 | ||
|
|
2e7a6d745c | ||
|
|
9cb3c1e1a6 | ||
|
|
0f449eb906 | ||
|
|
cdca79060b | ||
|
|
3505a5a354 | ||
|
|
d62ffe149a | ||
|
|
fd36e1b177 | ||
|
|
d390f9a7d1 | ||
|
|
128f650dd3 | ||
|
|
79733572d7 | ||
|
|
6b5d1e55de | ||
|
|
6b5eb2f92f | ||
|
|
49c346233e | ||
|
|
e81352715e | ||
|
|
159f1873d6 | ||
|
|
fa07a503dd | ||
|
|
d65fcc5d8c | ||
|
|
1696197f54 | ||
|
|
19089ac132 | ||
|
|
2b3657c76b | ||
|
|
63d45eb194 | ||
|
|
9e5b93a532 |
@@ -29,16 +29,12 @@ Runs a number of static repository-level sanity checks.
|
||||
|
||||
- `branch-history` guards against accidental commits of large files using the `--history` option of the `config/githooks/pre-push` script.
|
||||
|
||||
## `mfem-analysis.yml` (`build-analysis`)
|
||||
|
||||
Checks if the code builds and satisfies minimal requirements.
|
||||
|
||||
- `gitignore` builds hypre, METIS, and MFEM using `mfem/github-actions/build-hypre`, `mfem/github-actions/build-metis`, and `mfem/github-actions/build-mfem` and checks for correct `.gitignore` settings by running the `tests/scripts/gitignore` script.
|
||||
|
||||
## `builds-and-tests.yml`
|
||||
|
||||
Runs a matrix of builds and tests runs with different compilers, OS, mfem/hypre settings, etc. Also processes and upload Codecov reports.
|
||||
|
||||
One matrix job runs `tests/scripts/gitignore` after `make test-noclean` to check generated artifacts against `.gitignore`.
|
||||
|
||||
Uses the following GitHub Actions from <https://github.com/mfem/github-actions>:
|
||||
|
||||
- `mfem/github-actions/build-hypre`
|
||||
|
||||
@@ -111,6 +111,7 @@ jobs:
|
||||
build-system: make
|
||||
hypre-target: int64
|
||||
precision: fp64
|
||||
gitignore-check: YES
|
||||
- os: ubuntu-latest
|
||||
target: opt
|
||||
codecov: NO
|
||||
@@ -317,7 +318,13 @@ jobs:
|
||||
- name: tests
|
||||
if: matrix.build-system == 'make' && (matrix.target == 'opt' || matrix.os == 'ubuntu-latest')
|
||||
run: |
|
||||
cd ${{ env.MFEM_TOP_DIR }} && make test
|
||||
cd ${{ env.MFEM_TOP_DIR }}
|
||||
if [[ "${{ matrix.gitignore-check }}" == "YES" ]]; then
|
||||
make test-noclean
|
||||
else
|
||||
make test
|
||||
fi
|
||||
shell: bash
|
||||
|
||||
- name: cmake checks
|
||||
if: matrix.build-system == 'cmake' && matrix.target == 'dbg'
|
||||
@@ -369,3 +376,9 @@ jobs:
|
||||
directories: "fem general linalg mesh"
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
|
||||
- name: gitignore
|
||||
if: matrix.gitignore-check == 'YES'
|
||||
run: |
|
||||
cd ${{ env.MFEM_TOP_DIR }}/tests/scripts
|
||||
./runtest gitignore
|
||||
|
||||
@@ -14,9 +14,19 @@ name: "Static Analysis"
|
||||
on:
|
||||
push:
|
||||
branches: ["master", "next"]
|
||||
paths-ignore: &docs-only-paths
|
||||
- "**/*.md"
|
||||
- "doc/**"
|
||||
- ".binder/**"
|
||||
- "CITATION.cff"
|
||||
- "LICENSE"
|
||||
- "NOTICE"
|
||||
- "CHANGELOG"
|
||||
- "INSTALL"
|
||||
pull_request:
|
||||
# The branches below must be a subset of the branches above
|
||||
branches: ["master"]
|
||||
paths-ignore: *docs-only-paths
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
|
||||
@@ -1,102 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
name: "Build Analysis"
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
- next
|
||||
pull_request:
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
HYPRE_ARCHIVE: v2.19.0.tar.gz
|
||||
HYPRE_TOP_DIR: hypre-2.19.0
|
||||
METIS_ARCHIVE: metis-4.0.3.tar.gz
|
||||
METIS_TOP_DIR: metis-4.0.3
|
||||
COVERAGE_ENV: mfem-coverage
|
||||
MFEM_ACTIONS_VERSION: v2.7
|
||||
|
||||
jobs:
|
||||
gitignore:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
steps:
|
||||
- name: checkout MFEM
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
path: mfem
|
||||
|
||||
- name: Get MPI (Linux)
|
||||
run: |
|
||||
sudo apt-get install openmpi-bin libopenmpi-dev
|
||||
export OMPI_MCA_rmaps_base_oversubscribe=1
|
||||
|
||||
- name: Cache Hypre Install
|
||||
id: hypre-cache
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-int32-fp64-${{ env.MFEM_ACTIONS_VERSION }}
|
||||
|
||||
- name: Get Hypre
|
||||
if: steps.hypre-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-hypre@v2.7
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
target: int32
|
||||
precision: fp64
|
||||
|
||||
- name: Cache Metis Install
|
||||
id: metis-cache
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.METIS_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-${{ env.MFEM_ACTIONS_VERSION }}
|
||||
|
||||
- name: Install Metis
|
||||
if: steps.metis-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.7
|
||||
with:
|
||||
archive: ${{ env.METIS_ARCHIVE }}
|
||||
dir: ${{ env.METIS_TOP_DIR }}
|
||||
|
||||
# MFEM build and test
|
||||
- name: build-mfem
|
||||
uses: mfem/github-actions/build-mfem@v2.7
|
||||
with:
|
||||
os: ${{ runner.os }}
|
||||
target: opt
|
||||
codecov: NO
|
||||
mpi: par
|
||||
build-system: make
|
||||
hypre-dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
metis-dir: ${{ env.METIS_TOP_DIR }}
|
||||
mfem-dir: mfem
|
||||
|
||||
- name: test (no clean)
|
||||
run: |
|
||||
cd mfem && make test-noclean
|
||||
|
||||
- name: gitignore
|
||||
run: |
|
||||
cd mfem/tests/scripts
|
||||
./runtest gitignore
|
||||
@@ -17,7 +17,17 @@ permissions:
|
||||
on:
|
||||
push:
|
||||
branches: ["master", "next"]
|
||||
paths-ignore: &docs-only-paths
|
||||
- "**/*.md"
|
||||
- "doc/**"
|
||||
- ".binder/**"
|
||||
- "CITATION.cff"
|
||||
- "LICENSE"
|
||||
- "NOTICE"
|
||||
- "CHANGELOG"
|
||||
- "INSTALL"
|
||||
pull_request:
|
||||
paths-ignore: *docs-only-paths
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
|
||||
@@ -15,6 +15,10 @@ Version 4.9.1 (development)
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Improved FindPointsGSLIB surface mesh capability with support for simplices
|
||||
and an option to specify axis-aligned bounding box padding for near-surface
|
||||
point queries.
|
||||
|
||||
- Added GPU-enabled partial assembly for simplicial Bernstein H1 basis based on
|
||||
ragged tensor algorithms (see DOI: 10.1137/11082539X) for mass and diffusion
|
||||
integrators.
|
||||
@@ -66,6 +70,8 @@ Linear and nonlinear solvers
|
||||
|
||||
GPU computing
|
||||
-------------
|
||||
- Added device assembly support for 3D H(curl) VectorFEDomainLFIntegrator.
|
||||
|
||||
- Added NVIDIA cuDSS library interface. Implementation examples have been
|
||||
added to ex1 and ex1p. See https://developer.nvidia.com/cudss for more
|
||||
details. Supported versions >= 0.6.0.
|
||||
|
||||
@@ -6,7 +6,6 @@
|
||||
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-blue.svg"></a>
|
||||
<a href="https://github.com/mfem/mfem/releases/latest"><img alt="GitHub release" src="https://img.shields.io/github/v/release/mfem/mfem"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions/workflows/repo-check.yml?query=branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml?query=branch%3Amaster"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml?query=branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://ci.appveyor.com/project/mfem/mfem"><img alt="Build Status" src="https://ci.appveyor.com/api/projects/status/19non9sqm6msi2wy?svg=true"></a>
|
||||
<a href="https://docs.mfem.org/html/index.html"><img alt="Documentation" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
|
||||
|
||||
@@ -22,15 +22,15 @@ include(MfemCmakeUtilities)
|
||||
mfem_find_package(SuiteSparse SuiteSparse SuiteSparse_DIR "" "" "" ""
|
||||
"Paths to headers required by SuiteSparse."
|
||||
"Libraries required by SuiteSparse."
|
||||
ADD_COMPONENT "UMFPACK" "include;suitesparse" umfpack.h "lib" umfpack
|
||||
ADD_COMPONENT "KLU" "include;suitesparse" klu.h "lib" klu
|
||||
ADD_COMPONENT "AMD" "include;suitesparse" amd.h "lib" amd
|
||||
ADD_COMPONENT "BTF" "include;suitesparse" btf.h "lib" btf
|
||||
ADD_COMPONENT "CHOLMOD" "include;suitesparse" cholmod.h "lib" cholmod
|
||||
ADD_COMPONENT "COLAMD" "include;suitesparse" colamd.h "lib" colamd
|
||||
ADD_COMPONENT "CAMD" "include;suitesparse" camd.h "lib" camd
|
||||
ADD_COMPONENT "CCOLAMD" "include;suitesparse" ccolamd.h "lib" ccolamd
|
||||
ADD_COMPONENT "config" "include;suitesparse" SuiteSparse_config.h "lib"
|
||||
ADD_COMPONENT "UMFPACK" "include;include/suitesparse;suitesparse" umfpack.h "lib" umfpack
|
||||
ADD_COMPONENT "KLU" "include;include/suitesparse;suitesparse" klu.h "lib" klu
|
||||
ADD_COMPONENT "AMD" "include;include/suitesparse;suitesparse" amd.h "lib" amd
|
||||
ADD_COMPONENT "BTF" "include;include/suitesparse;suitesparse" btf.h "lib" btf
|
||||
ADD_COMPONENT "CHOLMOD" "include;include/suitesparse;suitesparse" cholmod.h "lib" cholmod
|
||||
ADD_COMPONENT "COLAMD" "include;include/suitesparse;suitesparse" colamd.h "lib" colamd
|
||||
ADD_COMPONENT "CAMD" "include;include/suitesparse;suitesparse" camd.h "lib" camd
|
||||
ADD_COMPONENT "CCOLAMD" "include;include/suitesparse;suitesparse" ccolamd.h "lib" ccolamd
|
||||
ADD_COMPONENT "config" "include;include/suitesparse;suitesparse" SuiteSparse_config.h "lib"
|
||||
suitesparseconfig)
|
||||
|
||||
if (SuiteSparse_FOUND AND METIS_VERSION_5)
|
||||
|
||||
+2
-1
@@ -133,7 +133,7 @@ set(SRCS
|
||||
tmop/assemble/diag2.cpp
|
||||
tmop/assemble/grad2_limit.cpp
|
||||
tmop/assemble/grad2.cpp
|
||||
tmop/assemble/diag3_limit.cpp
|
||||
tmop/assemble/diag3_limit.cpp
|
||||
tmop/assemble/diag3.cpp
|
||||
tmop/assemble/grad3_limit.cpp
|
||||
tmop/assemble/grad3.cpp
|
||||
@@ -311,6 +311,7 @@ set(HDRS
|
||||
tmop_tools.hpp
|
||||
tmop_amr.hpp
|
||||
gslib.hpp
|
||||
gslib/gslib_kernel_helpers.hpp
|
||||
transfer.hpp
|
||||
hyperbolic.hpp
|
||||
integrator.hpp
|
||||
|
||||
@@ -54,6 +54,8 @@ void Coefficient::Project(QuadratureFunction &qf)
|
||||
QuadratureSpaceBase &qspace = *qf.GetSpace();
|
||||
const int ne = qspace.GetNE();
|
||||
Vector values;
|
||||
// GetValues makes a reference, but we need it to be valid on Host
|
||||
qf.HostWrite();
|
||||
for (int iel = 0; iel < ne; ++iel)
|
||||
{
|
||||
qf.GetValues(iel, values);
|
||||
@@ -327,6 +329,8 @@ void VectorCoefficient::Project(QuadratureFunction &qf)
|
||||
const int ne = qspace.GetNE();
|
||||
DenseMatrix values;
|
||||
Vector col;
|
||||
// GetValues makes a reference, but we need it to be valid on Host
|
||||
qf.HostWrite();
|
||||
for (int iel = 0; iel < ne; ++iel)
|
||||
{
|
||||
qf.GetValues(iel, values);
|
||||
@@ -695,6 +699,8 @@ void MatrixCoefficient::Project(QuadratureFunction &qf, bool transpose)
|
||||
QuadratureSpaceBase &qspace = *qf.GetSpace();
|
||||
const int ne = qspace.GetNE();
|
||||
DenseMatrix values, matrix;
|
||||
// GetValues makes a reference, but we need it to be valid on Host
|
||||
qf.HostWrite();
|
||||
for (int iel = 0; iel < ne; ++iel)
|
||||
{
|
||||
qf.GetValues(iel, values);
|
||||
|
||||
+1231
-726
File diff suppressed because it is too large
Load Diff
+161
-46
@@ -12,6 +12,9 @@
|
||||
#ifndef MFEM_GSLIB
|
||||
#define MFEM_GSLIB
|
||||
|
||||
#include <map>
|
||||
#include <vector>
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include "pgridfunc.hpp"
|
||||
@@ -119,6 +122,11 @@ protected:
|
||||
// IntegrationRules for simplex->Quad/Hex and to project to p_max in-case of
|
||||
// p-refinement.
|
||||
Array<IntegrationRule *> ir_split;
|
||||
/// Integration rules built at the field polynomial order (only for surface
|
||||
/// meshes when mesh order is not the same as gridfunction order).
|
||||
Array<IntegrationRule *> ir_split_sol;
|
||||
/// Order at which #ir_split_sol was built; -1 means not built.
|
||||
int ir_split_sol_order = -1;
|
||||
Array<FiniteElementSpace *> fes_rst_map; //FESpaces to map Quad/Hex->Simplex
|
||||
Array<GridFunction *> gf_rst_map; // GridFunctions to map Quad/Hex->Simplex
|
||||
FiniteElementCollection *fec_map_lin;
|
||||
@@ -134,6 +142,8 @@ protected:
|
||||
AvgType avgtype; // average type used for L2 functions
|
||||
Array<int> split_element_map;
|
||||
Array<int> split_element_index;
|
||||
// Geometry::Type (as int) of the original element for each split quad.
|
||||
Array<int> split_element_geom;
|
||||
int NE_split_total; // total number of elements after mesh splitting
|
||||
int mesh_points_cnt; // number of mesh nodes
|
||||
// Tolerance to ignore points found beyond the mesh boundary.
|
||||
@@ -141,6 +151,12 @@ protected:
|
||||
double bdr_tol;
|
||||
// Use CPU functions for Mesh/GridFunction on device for gslib1.0.7
|
||||
bool gpu_to_cpu_fallback = false;
|
||||
// Check if a point is inside the oriented bounding box of an
|
||||
// element before the Newton iteration.
|
||||
// Note: only used in MFEM implementation (not in gslib) which currently
|
||||
// supports GPU kernels for area meshes in 2D, volume meshes in 3D,
|
||||
// and surface meshes in 1D/2D/3D.
|
||||
bool obb_check = true;
|
||||
|
||||
// Device specific data used for FindPoints
|
||||
struct DEV_STRUCT
|
||||
@@ -162,11 +178,16 @@ protected:
|
||||
mutable double surf_dist_tol;
|
||||
} DEV;
|
||||
|
||||
/// Use GSLIB for communication and interpolation
|
||||
// Helper function to setup and free gslib's crystal router.
|
||||
void SetupCrystal(); // Called inside Setup and SetupSurf_base
|
||||
void FreeCrystal(); // Called inside FreeData
|
||||
|
||||
/// Use GSLIB for communication and interpolation. Updates field_out on
|
||||
/// host.
|
||||
virtual void InterpolateH1(const GridFunction &field_in, Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
/// Uses GSLIB Crystal Router for communication followed by MFEM's
|
||||
/// interpolation functions
|
||||
/// interpolation functions. Updates field_out on host.
|
||||
virtual void InterpolateGeneral(const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
@@ -181,12 +202,26 @@ protected:
|
||||
IntegrationRule *irule,
|
||||
int order);
|
||||
|
||||
/** @brief Build integration rules at the given @a order for each split mesh
|
||||
* and store them in @a ir_out. Requires that \ref SetupSplitMeshes has
|
||||
* already been called. */
|
||||
virtual void SetupIntegrationRules(const int order,
|
||||
Array<IntegrationRule *> &ir_out);
|
||||
|
||||
/** @brief Helper function that calls \ref SetupSplitMeshes and
|
||||
* \ref SetupIntegrationRuleForSplitMesh. */
|
||||
* \ref SetupIntegrationRules. */
|
||||
virtual void SetupSplitMeshesAndIntegrationRules(const int order);
|
||||
|
||||
/// Get GridFunction value at the points expected by GSLIB.
|
||||
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals) const;
|
||||
/** @brief Get GridFunction value at the points expected by GSLIB.
|
||||
* @param[in] gf_in Grid function to evaluate.
|
||||
* @param[out] node_vals Output values.
|
||||
* @param[in] ir_in If non-null, use these rules instead of #ir_split.
|
||||
* @param[in] by_element If true, output has element-major layout
|
||||
* [nel][vdim][ndofs]; otherwise component-major
|
||||
* layout [vdim][total_pts]. */
|
||||
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals,
|
||||
const Array<IntegrationRule *> *ir_in = nullptr,
|
||||
bool by_element = false) const;
|
||||
|
||||
/** @brief Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For
|
||||
* simplices, find the original element number (that was split into
|
||||
@@ -293,9 +328,10 @@ protected:
|
||||
const unsigned n,
|
||||
const uint nel,
|
||||
const unsigned m,
|
||||
const double bbox_tol,
|
||||
const double bbox_rel_size_inc,
|
||||
const uint local_hash_size,
|
||||
const uint global_hash_size);
|
||||
const uint global_hash_size,
|
||||
const Vector *aabb_sz_inc);
|
||||
|
||||
/// Preprocess 3D surface mesh needed for FindPoints.
|
||||
void findptssurf_setup_3(DEV_STRUCT &devs,
|
||||
@@ -303,17 +339,47 @@ protected:
|
||||
const unsigned n,
|
||||
const uint nel,
|
||||
const unsigned m,
|
||||
const double bbox_tol,
|
||||
const double bbox_rel_size_inc,
|
||||
const uint local_hash_size,
|
||||
const uint global_hash_size,
|
||||
const int rD);
|
||||
const int rD,
|
||||
const Vector *aabb_sz_inc);
|
||||
|
||||
/** @brief Shared implementation for the public surface-setup methods.
|
||||
*
|
||||
* @details Initializes the surface-search data structures, builds the
|
||||
* split-element representation expected by gslib, and constructs the
|
||||
* element bounding boxes used by the MFEM surface kernels.
|
||||
*
|
||||
* If @a aabb_sz_inc is null, the setup stores the default oriented
|
||||
* bounding boxes and uses @a bbox_rel_size_inc as their relative size
|
||||
* increase factor.
|
||||
*
|
||||
* If @a aabb_sz_inc is non-null, the setup stores axis-aligned bounding
|
||||
* boxes only, applies the requested absolute AABB expansion in each
|
||||
* physical direction, and adjusts the tolerance @a bdr_tol so points
|
||||
* found in the expanded region are classified as border points.
|
||||
*
|
||||
* @param[in] m Input surface mesh.
|
||||
* @param[in] bbox_rel_size_inc Relative size increase applied when
|
||||
* expanding each element bounding box during
|
||||
* setup.
|
||||
* @param[in] aabb_sz_inc Optional total absolute AABB expansion
|
||||
* applied to the stored axis-aligned
|
||||
* bounding boxes after construction.
|
||||
* @param[in] newt_tol Newton tolerance for the point-search
|
||||
* kernels.
|
||||
*/
|
||||
void SetupSurf_Base(Mesh &m,
|
||||
const double bbox_rel_size_inc,
|
||||
const Vector *aabb_sz_inc,
|
||||
const double newt_tol);
|
||||
public:
|
||||
/// Serial constructor
|
||||
FindPointsGSLIB();
|
||||
|
||||
/// Serial constructor + setup with given Mesh (see \ref Setup)
|
||||
FindPointsGSLIB(Mesh &mesh_in, const double bb_t = 0.1,
|
||||
FindPointsGSLIB(Mesh &mesh_in, const double bbox_rel_size_inc = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
@@ -322,7 +388,7 @@ public:
|
||||
FindPointsGSLIB(MPI_Comm comm_);
|
||||
|
||||
/// Constructor + setup with given ParMesh (see \ref Setup)
|
||||
FindPointsGSLIB(ParMesh &mesh_in, const double bb_t = 0.1,
|
||||
FindPointsGSLIB(ParMesh &mesh_in, const double bbox_rel_size_inc = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
#endif
|
||||
@@ -338,23 +404,59 @@ public:
|
||||
Note: not tested with periodic (L2).
|
||||
Note: the input mesh \p m must have Nodes set.
|
||||
|
||||
@param[in] m Input mesh.
|
||||
@param[in] bb_t (Optional) Relative size of bounding box around
|
||||
each element.
|
||||
@param[in] newt_tol (Optional) Newton tolerance for the gslib
|
||||
search methods.
|
||||
@param[in] npt_max (Optional) Number of points for simultaneous
|
||||
iteration. This alters performance and
|
||||
memory footprint.
|
||||
@param[in] m Input mesh.
|
||||
@param[in] bbox_rel_size_inc (Optional) Relative size increase applied
|
||||
when expanding each element bounding box.
|
||||
@param[in] newt_tol (Optional) Newton tolerance for the gslib
|
||||
search methods.
|
||||
@param[in] npt_max (Optional) Number of points for
|
||||
simultaneous iteration. This alters
|
||||
performance and memory footprint.
|
||||
*/
|
||||
void Setup(Mesh &m, const double bb_t = 0.1, const double newt_tol = 1.0e-12,
|
||||
void Setup(Mesh &m, const double bbox_rel_size_inc = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
/// Preprocess the surface mesh to compute data for FindPoints.
|
||||
void SetupSurf(Mesh &m,
|
||||
const double bb_t = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
const double bbox_rel_size_inc = 0.1,
|
||||
const double newt_tol = 1.0e-12);
|
||||
|
||||
/** @brief Preprocess the surface mesh to compute data for FindPoints using
|
||||
* absolute AABB expansion.
|
||||
*
|
||||
* @details This method computes only axis-aligned bounding boxes and
|
||||
* increases their total length by a user-specified amount in each
|
||||
* physical direction. The absolute AABB expansion is applied
|
||||
* symmetrically to the lower and upper bounds.
|
||||
*
|
||||
* The size of @a aabb_sz_inc determines how the expansion values are
|
||||
* interpreted:
|
||||
* - `1`: one expansion value used in every direction for every element
|
||||
* - `NElements`: one expansion value per element, reused in x/y/z
|
||||
* directions
|
||||
* - `SpaceDim`: one expansion value per physical direction, reused for
|
||||
* every element
|
||||
* - `NElements*SpaceDim`: one expansion value per element and direction,
|
||||
* ordered as `(dx1,dy1,dz1, ... dxN,dyN,dzN)`
|
||||
*
|
||||
* This method disables the oriented bounding-box precheck because the
|
||||
* stored boxes are modified only in their axis-aligned representation.
|
||||
*
|
||||
* @param[in] m Input surface mesh.
|
||||
* @param[in] aabb_sz_inc Total absolute AABB expansion applied in
|
||||
* each physical direction to the stored
|
||||
* axis-aligned bounding boxes.
|
||||
* @param[in] newt_tol Newton tolerance for the point-search
|
||||
* kernels.
|
||||
*
|
||||
* @note We disable the oriented bounding box check with this setup.
|
||||
* @a bdr_tol is also adjusted so that all points in the AABBs can
|
||||
* be found.
|
||||
*/
|
||||
void SetupSurfWithAABBExpansion(Mesh &m, const Vector &aabb_sz_inc,
|
||||
const double newt_tol = 1.0e-12);
|
||||
|
||||
|
||||
/** @brief Searches positions given in physical space by \p point_pos.
|
||||
|
||||
@@ -401,7 +503,8 @@ public:
|
||||
/// Setup FindPoints and search positions
|
||||
void FindPoints(Mesh &m, const Vector &point_pos,
|
||||
const int point_pos_ordering = Ordering::byNODES,
|
||||
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
|
||||
const double bbox_rel_size_inc = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
/** @brief Interpolation of field values at prescribed reference space
|
||||
@@ -413,7 +516,11 @@ public:
|
||||
mesh that was given to Setup().
|
||||
@param[out] field_out Interpolated values. For points that are not found
|
||||
the value is set to #default_interp_value.
|
||||
The output ordering is determined from field_in.*/
|
||||
The output ordering is determined from field_in.
|
||||
|
||||
@note: field_out is moved to device if field_in is on device. Otherwise,
|
||||
field_out memory allocation is not changed.
|
||||
*/
|
||||
virtual void Interpolate(const GridFunction &field_in, Vector &field_out);
|
||||
|
||||
/// Interpolation of field values, with output ordering specification.
|
||||
@@ -468,7 +575,12 @@ public:
|
||||
* @details When using FindPoints, gslib may return points as found on the
|
||||
* boundary even when they are slightly outside the domain. This tolerance
|
||||
* is used to filter such points based on the distance^2 value and mark them
|
||||
* as not found.*/
|
||||
* as not found.
|
||||
*
|
||||
* @note When the SetupSurfWithAABBExpansion method is used for surface
|
||||
* meshes, this tolerance is automatically computed based on the size of
|
||||
* expanded AABBs. Using this method will override that computed tolerance.
|
||||
* */
|
||||
virtual void SetDistanceToleranceForPointsFoundOnBoundary(double bdr_tol_)
|
||||
{
|
||||
bdr_tol = bdr_tol_;
|
||||
@@ -603,25 +715,28 @@ public:
|
||||
Note: not tested with periodic meshes (L2).
|
||||
Note: the input mesh \p m must have Nodes set.
|
||||
|
||||
@param[in] m Input mesh.
|
||||
@param[in] meshid A unique # for each overlapping mesh. This id is
|
||||
used to make sure that points being searched are not
|
||||
looked for in the mesh that they belong to.
|
||||
@param[in] gfmax (Optional) GridFunction in H1 that is used as a
|
||||
discriminator when one point is located in multiple
|
||||
meshes. The mesh that maximizes gfmax is chosen.
|
||||
For example, using the distance field based on the
|
||||
overlapping boundaries is helpful for convergence
|
||||
during Schwarz iterations.
|
||||
@param[in] bb_t (Optional) Relative size of bounding box around
|
||||
each element.
|
||||
@param[in] newt_tol (Optional) Newton tolerance for the gslib
|
||||
search methods.
|
||||
@param[in] npt_max (Optional) Number of points for simultaneous
|
||||
iteration. This alters performance and
|
||||
memory footprint.*/
|
||||
void Setup(Mesh &m, const int meshid, GridFunction *gfmax = NULL,
|
||||
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
|
||||
@param[in] m Input mesh.
|
||||
@param[in] meshid A unique # for each overlapping mesh.
|
||||
This id is used to make sure that points
|
||||
being searched are not looked for in the
|
||||
mesh that they belong to.
|
||||
@param[in] gfmax (Optional) GridFunction in H1 that is used
|
||||
as a discriminator when one point is
|
||||
located in multiple meshes. The mesh that
|
||||
maximizes gfmax is chosen. For example,
|
||||
using the distance field based on the
|
||||
overlapping boundaries is helpful for
|
||||
convergence during Schwarz iterations.
|
||||
@param[in] bbox_rel_size_inc (Optional) Relative size increase applied
|
||||
when expanding each element bounding box.
|
||||
@param[in] newt_tol (Optional) Newton tolerance for the gslib
|
||||
search methods.
|
||||
@param[in] npt_max (Optional) Number of points for
|
||||
simultaneous iteration. This alters
|
||||
performance and memory footprint.*/
|
||||
void Setup(Mesh &m, const int meshid, GridFunction *gfmax = nullptr,
|
||||
const double bbox_rel_size_inc = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
/** Searches positions given in physical space by \p point_pos. All output
|
||||
@@ -677,7 +792,7 @@ class GSOPGSLIB
|
||||
protected:
|
||||
struct gslib::crystal *cr; // gslib's internal data
|
||||
struct gslib::comm *gsl_comm; // gslib's internal data
|
||||
struct gslib::gs_data *gsl_data = NULL;
|
||||
struct gslib::gs_data *gsl_data = nullptr;
|
||||
int num_ids;
|
||||
|
||||
public:
|
||||
|
||||
+64
-170
@@ -11,7 +11,7 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/kernels.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -27,8 +27,6 @@
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
|
||||
#include <climits>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
#if GSLIB_RELEASE_VERSION >= 10009
|
||||
@@ -54,127 +52,14 @@ struct findptsElementGPT_t
|
||||
double x[DIM], jac[DIM * DIM], hes[4];
|
||||
};
|
||||
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[DIM], A[DIM * DIM];
|
||||
dbl_range_t x[DIM];
|
||||
};
|
||||
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[DIM];
|
||||
double fac[DIM];
|
||||
unsigned int *offset;
|
||||
int max;
|
||||
};
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first derivative at x.
|
||||
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2 * (x - z[j]);
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p0[pN+i] = 2.0 * lCoeff[i] * u1;
|
||||
}
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first and second derivative at x.
|
||||
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2 * (x - z[j]);
|
||||
u2 = d_j * u2 + u1;
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p0[pN+i] = 2.0 * lCoeff[i] * u1;
|
||||
p0[2*pN+i] = 8.0 * lCoeff[i] * u2;
|
||||
}
|
||||
|
||||
// Axis-aligned bounding box test.
|
||||
static MFEM_HOST_DEVICE inline double AABB_test(const obbox_t *const b,
|
||||
const double x[2])
|
||||
{
|
||||
double test = 1;
|
||||
for (int d = 0; d < 2; ++d)
|
||||
{
|
||||
double b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
|
||||
test = test < 0 ? test : b_d;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
|
||||
// Axis-aligned bounding box test followed by oriented bounding-box test.
|
||||
static MFEM_HOST_DEVICE inline double bbox_test(const obbox_t *const b,
|
||||
const double x[2])
|
||||
{
|
||||
const double bxyz = AABB_test(b, x);
|
||||
if (bxyz < 0)
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
else
|
||||
{
|
||||
double dxyz[2];
|
||||
for (int d = 0; d < 2; ++d)
|
||||
{
|
||||
dxyz[d] = x[d] - b->c0[d];
|
||||
}
|
||||
double test = 1;
|
||||
for (int d = 0; d < 2; ++d)
|
||||
{
|
||||
double rst = 0;
|
||||
for (int e = 0; e < 2; ++e)
|
||||
{
|
||||
rst += b->A[d * 2 + e] * dxyz[e];
|
||||
}
|
||||
double brst = (rst + 1) * (1 - rst);
|
||||
test = test < 0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
}
|
||||
|
||||
// Element index corresponding to hash mesh that the point is located in.
|
||||
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
|
||||
const double x[2])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d = 2 - 1; d >= 0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
|
||||
sum += i < 0 ? 0 : (n - 1 < i ? n - 1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
using dbl_range_t = gslib::dbl_range_t;
|
||||
using obbox_t = gslib::obbox_t<DIM>;
|
||||
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<DIM>;
|
||||
using gslib::bbox_test;
|
||||
using gslib::hash_index;
|
||||
using gslib::l2norm2;
|
||||
using gslib::lag_eval_first_der;
|
||||
using gslib::lag_eval_second_der;
|
||||
|
||||
/*Solve Ax=y. A is row-major */
|
||||
static MFEM_HOST_DEVICE inline void lin_solve_2(double x[2], const double A[4],
|
||||
@@ -185,12 +70,6 @@ static MFEM_HOST_DEVICE inline void lin_solve_2(double x[2], const double A[4],
|
||||
x[1] = idet*(A[0]*y[1] - A[2]*y[0]);
|
||||
}
|
||||
|
||||
/* L2 norm squared. */
|
||||
static MFEM_HOST_DEVICE inline double l2norm2(const double x[2])
|
||||
{
|
||||
return x[0] * x[0] + x[1] * x[1];
|
||||
}
|
||||
|
||||
/* the bit structure of flags is CSSRR
|
||||
the C bit --- 1<<4 --- is set when the point is converged
|
||||
RR is 0 = 00b if r is unconstrained,
|
||||
@@ -352,7 +231,7 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *res,
|
||||
const findptsElementPoint_t *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = l2norm2(resid);
|
||||
const double dist2 = l2norm2<2>(resid);
|
||||
const double decr = p->dist2 - dist2;
|
||||
const double pred = p->dist2p;
|
||||
for (int d = 0; d < 2; ++d)
|
||||
@@ -695,25 +574,25 @@ static MFEM_HOST_DEVICE double tensor_ig2_j(double *g_partials,
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void FindPointsLocal2D_Kernel(const int npt,
|
||||
const double tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
static void FindPointsLocal2DKernel(const int npt,
|
||||
const double tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
{
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
@@ -1175,30 +1054,45 @@ void FindPointsGSLIB::FindPointsLocal2(const Vector &point_pos,
|
||||
switch (DEV.dof1d)
|
||||
{
|
||||
case 2:
|
||||
return FindPointsLocal2D_Kernel<2>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
FindPointsLocal2DKernel<2>(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 3:
|
||||
return FindPointsLocal2D_Kernel<3>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
FindPointsLocal2DKernel<3>(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 4:
|
||||
return FindPointsLocal2D_Kernel<4>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
FindPointsLocal2DKernel<4>(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 5:
|
||||
return FindPointsLocal2D_Kernel<5>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
FindPointsLocal2DKernel<5>(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
default:
|
||||
return FindPointsLocal2D_Kernel(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx,
|
||||
plhm, plhf, plho, pcode, pelem,
|
||||
pref, pdist, pgll1d, plc, DEV.dof1d);
|
||||
FindPointsLocal2DKernel(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc, DEV.dof1d);
|
||||
break;
|
||||
}
|
||||
}
|
||||
#undef DIM2
|
||||
|
||||
+29
-157
@@ -11,9 +11,7 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/kernels.hpp"
|
||||
|
||||
#include <climits>
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -59,128 +57,15 @@ struct findptsElemPt
|
||||
double x[DIM], jac[DIM * DIM], hes[18];
|
||||
};
|
||||
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[DIM], A[DIM * DIM];
|
||||
dbl_range_t x[DIM];
|
||||
};
|
||||
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[DIM];
|
||||
double fac[DIM];
|
||||
unsigned int *offset;
|
||||
// int max;
|
||||
};
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first derivative at x.
|
||||
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2*(x-z[j]);
|
||||
u1 = d_j*u1+u0;
|
||||
u0 = d_j*u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i]*u0;
|
||||
p0[pN+i] = 2.0*lCoeff[i]*u1;
|
||||
}
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first and second derivative at x.
|
||||
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2*(x-z[j]);
|
||||
u2 = d_j*u2+u1;
|
||||
u1 = d_j*u1+u0;
|
||||
u0 = d_j*u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i]*u0;
|
||||
p0[pN+i] = 2.0*lCoeff[i]*u1;
|
||||
p0[2*pN+i] = 8.0*lCoeff[i]*u2;
|
||||
}
|
||||
|
||||
// Axis-aligned bounding box test.
|
||||
static MFEM_HOST_DEVICE inline double AABB_test(const obbox_t *const b,
|
||||
const double x[3])
|
||||
{
|
||||
double b_d;
|
||||
for (int d = 0; d < 3; ++d)
|
||||
{
|
||||
b_d = (x[d]-b->x[d].min)*(b->x[d].max-x[d]);
|
||||
if (b_d < 0) { return b_d; }
|
||||
}
|
||||
return b_d;
|
||||
}
|
||||
|
||||
// Axis-aligned bounding box test followed by oriented bounding-box test.
|
||||
static MFEM_HOST_DEVICE inline double bbox_test(const obbox_t *const b,
|
||||
const double x[3])
|
||||
{
|
||||
const double bxyz = AABB_test(b, x);
|
||||
if (bxyz < 0)
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
else
|
||||
{
|
||||
double dxyz[3];
|
||||
for (int d = 0; d < 3; ++d)
|
||||
{
|
||||
dxyz[d] = x[d]-b->c0[d];
|
||||
}
|
||||
double test = 1;
|
||||
for (int d = 0; d < 3; ++d)
|
||||
{
|
||||
double rst = 0;
|
||||
for (int e = 0; e < 3; ++e)
|
||||
{
|
||||
rst += b->A[d*3+e]*dxyz[e];
|
||||
}
|
||||
double brst = (rst+1)*(1-rst);
|
||||
test = test < 0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
}
|
||||
|
||||
// Element index corresponding to hash mesh that the point is located in.
|
||||
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
|
||||
const double x[3])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d = 3-1; d >= 0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
int i = (int)floor((x[d]-p->bnd[d].min)*p->fac[d]);
|
||||
sum += i < 0 ? 0 : (n-1 < i ? n-1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
using dbl_range_t = gslib::dbl_range_t;
|
||||
using obbox_t = gslib::obbox_t<DIM>;
|
||||
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<DIM>;
|
||||
using gslib::bbox_test;
|
||||
using gslib::hash_index;
|
||||
using gslib::l2norm2;
|
||||
using gslib::lag_eval_first_der;
|
||||
using gslib::lag_eval_second_der;
|
||||
using gslib::lin_solve_sym_2;
|
||||
|
||||
// Solve Ax=y. A is row-major.
|
||||
static MFEM_HOST_DEVICE inline void lin_solve_3(double x[3], const double A[9],
|
||||
@@ -199,22 +84,6 @@ static MFEM_HOST_DEVICE inline void lin_solve_3(double x[3], const double A[9],
|
||||
x[2] = idet*(inv6*y[0]+inv7*y[1]+inv8*y[2]);
|
||||
}
|
||||
|
||||
// Solve Ax=y. A is a symmetric 2x2 matrix.
|
||||
static MFEM_HOST_DEVICE inline void lin_solve_sym_2(double x[2],
|
||||
const double A[3],
|
||||
const double y[2])
|
||||
{
|
||||
const double idet = 1 / (A[0]*A[2]-A[1]*A[1]);
|
||||
x[0] = idet*(A[2]*y[0]-A[1]*y[1]);
|
||||
x[1] = idet*(A[0]*y[1]-A[1]*y[0]);
|
||||
}
|
||||
|
||||
// L2 norm.
|
||||
static MFEM_HOST_DEVICE inline double l2norm2(const double x[3])
|
||||
{
|
||||
return x[0]*x[0]+x[1]*x[1]+x[2]*x[2];
|
||||
}
|
||||
|
||||
/* the bit structure of flags is CTTSSRR
|
||||
the C bit --- 1<<6 --- is set when the point is converged
|
||||
RR is 0 = 00b if r is unconstrained,
|
||||
@@ -459,7 +328,7 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsPt *res,
|
||||
const findptsPt *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = l2norm2(resid);
|
||||
const double dist2 = l2norm2<3>(resid);
|
||||
const double decr = p->dist2-dist2;
|
||||
const double pred = p->dist2p;
|
||||
for (int d = 0; d < 3; ++d)
|
||||
@@ -1809,33 +1678,36 @@ void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
|
||||
{
|
||||
case 2:
|
||||
FindPointsLocal3DKernel<2>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
pgslm, NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
break;
|
||||
case 3:
|
||||
FindPointsLocal3DKernel<3>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
pgslm, NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
break;
|
||||
case 4:
|
||||
FindPointsLocal3DKernel<4>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
pgslm, NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
break;
|
||||
case 5:
|
||||
FindPointsLocal3DKernel<5>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
pgslm, NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
break;
|
||||
default:
|
||||
FindPointsLocal3DKernel(npt, DEV.newt_tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc,
|
||||
FindPointsLocal3DKernel(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist, pgll1d, plc,
|
||||
DEV.dof1d);
|
||||
break;
|
||||
}
|
||||
}
|
||||
#undef pMax
|
||||
|
||||
+107
-176
@@ -11,6 +11,7 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -52,113 +53,14 @@ struct findptsElementGPT_t
|
||||
double x[sDIM], jac[sDIM*rDIM], hes[sDIM*rDIM];
|
||||
};
|
||||
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[sDIM], A[sDIM*sDIM];
|
||||
dbl_range_t x[sDIM];
|
||||
};
|
||||
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[sDIM];
|
||||
double fac[sDIM];
|
||||
unsigned int *offset;
|
||||
};
|
||||
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2 * (x-z[j]);
|
||||
u2 = d_j * u2 + u1;
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
double *p1 = p0 + pN, *p2 = p0 + 2 * pN;
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p1[i] = 2.0 * lCoeff[i] * u1;
|
||||
p2[i] = 8.0 * lCoeff[i] * u2;
|
||||
}
|
||||
|
||||
/* positive when possibly inside */
|
||||
static MFEM_HOST_DEVICE inline double obbox_axis_test(const obbox_t *const b,
|
||||
const double x[sDIM])
|
||||
{
|
||||
double b_d;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
|
||||
if (b_d < 0) // if outside in any dimension
|
||||
{
|
||||
return b_d;
|
||||
}
|
||||
}
|
||||
return b_d; // only positive if inside
|
||||
}
|
||||
|
||||
/* positive when given point is possibly inside given obbox b */
|
||||
static MFEM_HOST_DEVICE inline double obbox_test(const obbox_t *const b,
|
||||
const double x[sDIM])
|
||||
{
|
||||
const double bxyz = obbox_axis_test(b,x);
|
||||
if (bxyz<0) // test if point is in AABB
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
else // test OBB only if inside AABB
|
||||
{
|
||||
double dxyz[sDIM];
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
dxyz[d] = x[d] - b->c0[d];
|
||||
}
|
||||
double test = 1;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
double rst = 0;
|
||||
for (int e=0; e<sDIM; ++e)
|
||||
{
|
||||
rst += b->A[d*2 + e] * dxyz[e];
|
||||
}
|
||||
double brst = (rst+1)*(1-rst);
|
||||
test = test<0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
}
|
||||
|
||||
/* Hash index in the hash table to the elements that possibly contain the point x */
|
||||
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
|
||||
const double x[2])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d=sDIM-1; d>=0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
|
||||
sum += i<0 ? 0 : (n-1 < i ? n-1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline double l2norm2(const double x[2])
|
||||
{
|
||||
return x[0] * x[0] + x[1] * x[1];
|
||||
}
|
||||
using dbl_range_t = gslib::dbl_range_t;
|
||||
using obbox_t = gslib::obbox_t<sDIM>;
|
||||
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<sDIM>;
|
||||
using gslib::AABB_test;
|
||||
using gslib::bbox_test;
|
||||
using gslib::hash_index;
|
||||
using gslib::l2norm2;
|
||||
using gslib::lag_eval_second_der;
|
||||
|
||||
/* the bit structure of flags is CRR
|
||||
the C bit --- 1<<2 --- is set when the point is converged
|
||||
@@ -187,29 +89,29 @@ static MFEM_HOST_DEVICE inline int point_index(const int x)
|
||||
/* check reduction in objective against prediction, and adjust
|
||||
trust region radius (p->tr) accordingly;
|
||||
may reject the prior step, returning 1; otherwise returns 0
|
||||
sets out->dist2, out->index, out->x, out->oldr in any event,
|
||||
leaving out->r, out->dr, out->flags to be set when returning 0 */
|
||||
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
|
||||
sets out_pt->dist2, out_pt->index, out_pt->x, out_pt->oldr in any event,
|
||||
leaving out_pt->r, out_pt->dr, out_pt->flags to be set when returning 0 */
|
||||
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out_pt,
|
||||
const double resid[2],
|
||||
const findptsElementPoint_t *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = l2norm2(resid);
|
||||
const double dist2 = l2norm2<2>(resid);
|
||||
const double decr = p->dist2 - dist2;
|
||||
const double pred = p->dist2p;
|
||||
out->x[0] = p->x[0];
|
||||
out->x[1] = p->x[1];
|
||||
out->oldr = p->r;
|
||||
out->dist2 = dist2;
|
||||
out_pt->x[0] = p->x[0];
|
||||
out_pt->x[1] = p->x[1];
|
||||
out_pt->oldr = p->r;
|
||||
out_pt->dist2 = dist2;
|
||||
if (decr >= 0.01*pred)
|
||||
{
|
||||
if (decr >= 0.9*pred) // very good iteration
|
||||
{
|
||||
out->tr = p->tr*2;
|
||||
out_pt->tr = p->tr*2;
|
||||
}
|
||||
else // somewhat good iteration
|
||||
{
|
||||
out->tr = p->tr;
|
||||
out_pt->tr = p->tr;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -220,21 +122,21 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
|
||||
"very good iteration" --- this doubles the trust radius,
|
||||
which is why we divide by 4 below */
|
||||
double v0 = fabs(p->r - p->oldr);
|
||||
out->tr = v0/4.0;
|
||||
out->dist2 = p->dist2;
|
||||
out->r = p->oldr;
|
||||
out->flags = p->flags>>3;
|
||||
out->dist2p = -HUGE_VAL;
|
||||
out_pt->tr = v0/4.0;
|
||||
out_pt->dist2 = p->dist2;
|
||||
out_pt->r = p->oldr;
|
||||
out_pt->flags = p->flags>>3;
|
||||
out_pt->dist2p = -HUGE_VAL;
|
||||
if (pred < dist2*tol)
|
||||
{
|
||||
out->flags |= CONVERGED_FLAG;
|
||||
out_pt->flags |= CONVERGED_FLAG;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline void newton_edge( findptsElementPoint_t *const
|
||||
out,
|
||||
out_pt,
|
||||
const double jac[2],
|
||||
const double rhess,
|
||||
const double resid[2],
|
||||
@@ -304,9 +206,9 @@ newton_edge_fin:
|
||||
{
|
||||
new_flags |= CONVERGED_FLAG;
|
||||
}
|
||||
out->r = newr;
|
||||
out->dist2p = -v;
|
||||
out->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
|
||||
out_pt->r = newr;
|
||||
out_pt->dist2p = -v;
|
||||
out_pt->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE void seed_j( const double *elx[sDIM],
|
||||
@@ -332,26 +234,27 @@ static MFEM_HOST_DEVICE void seed_j( const double *elx[sDIM],
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void FindPointsEdgeLocal2D_Kernel( const int npt,
|
||||
const double tol,
|
||||
const double dist2tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0 )
|
||||
static void FindPointsEdgeLocal2DKernel( const int npt,
|
||||
const double tol,
|
||||
const double dist2tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const bool obb_check,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0 )
|
||||
{
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
@@ -412,22 +315,34 @@ static void FindPointsEdgeLocal2D_Kernel( const int npt,
|
||||
{
|
||||
const unsigned int el = *elp;
|
||||
|
||||
const int n_box_ents = obb_check ? (3*sDIM + sDIM2) : (2*sDIM);
|
||||
bool pass_bb = true;
|
||||
obbox_t box;
|
||||
int n_box_ents = 3*sDIM + sDIM2;
|
||||
|
||||
for (int idx = 0; idx < sDIM; ++idx)
|
||||
if (obb_check)
|
||||
{
|
||||
box.c0[idx] = boxinfo[n_box_ents*el + idx];
|
||||
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
|
||||
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
|
||||
for (int idx = 0; idx < sDIM; ++idx)
|
||||
{
|
||||
box.c0[idx] = boxinfo[n_box_ents*el + idx];
|
||||
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
|
||||
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
|
||||
}
|
||||
for (int idx = 0; idx < sDIM2; ++idx)
|
||||
{
|
||||
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
|
||||
}
|
||||
pass_bb = (bbox_test(&box, x_i) >= 0);
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int d = 0; d < sDIM; ++d)
|
||||
{
|
||||
box.x[d].min = boxinfo[n_box_ents*el + d];
|
||||
box.x[d].max = boxinfo[n_box_ents*el + sDIM + d];
|
||||
}
|
||||
pass_bb = (AABB_test(&box, x_i) >= 0);
|
||||
}
|
||||
|
||||
for (int idx = 0; idx < sDIM2; ++idx)
|
||||
{
|
||||
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
|
||||
}
|
||||
|
||||
if (obbox_test(&box,x_i)>=0)
|
||||
if (pass_bb)
|
||||
{
|
||||
//------------ findpts_local ------------------
|
||||
{
|
||||
@@ -516,11 +431,14 @@ static void FindPointsEdgeLocal2D_Kernel( const int npt,
|
||||
double *hess = jac + sDIM*rDIM;
|
||||
|
||||
findptsElementGEdge_t edge;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d] = constraint_workspace + d*D1D;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d] = constraint_workspace + d*D1D;
|
||||
edge.x[d][j] = elx[d][j];
|
||||
}
|
||||
}
|
||||
@@ -681,28 +599,41 @@ void FindPointsGSLIB::FindPointsEdgeLocal2( const Vector &point_pos,
|
||||
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
|
||||
auto plc = DEV.lagcoeff.Read(use_dev);
|
||||
double dist2tol = DEV.surf_dist_tol;
|
||||
const bool obb_chk = obb_check;
|
||||
switch (DEV.dof1d)
|
||||
{
|
||||
case 2:
|
||||
return FindPointsEdgeLocal2D_Kernel<2>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
FindPointsEdgeLocal2DKernel<2>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 3:
|
||||
return FindPointsEdgeLocal2D_Kernel<3>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
FindPointsEdgeLocal2DKernel<3>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 4:
|
||||
return FindPointsEdgeLocal2D_Kernel<4>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
FindPointsEdgeLocal2DKernel<4>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
default:
|
||||
return FindPointsEdgeLocal2D_Kernel(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc, DEV.dof1d);
|
||||
FindPointsEdgeLocal2DKernel(npt, DEV.newt_tol, dist2tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc, DEV.dof1d);
|
||||
break;
|
||||
}
|
||||
}
|
||||
#undef sDIM
|
||||
|
||||
+109
-181
@@ -11,6 +11,7 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -54,117 +55,14 @@ struct findptsElementGPT_t
|
||||
double x[sDIM], jac[sDIM], hes[sDIM*(1+1)];
|
||||
};
|
||||
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[sDIM], A[sDIM*sDIM];
|
||||
dbl_range_t x[sDIM];
|
||||
};
|
||||
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[sDIM];
|
||||
double fac[sDIM];
|
||||
unsigned int *offset;
|
||||
};
|
||||
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j=0; j<pN; ++j)
|
||||
{
|
||||
if (i!=j)
|
||||
{
|
||||
double d_j = 2 * (x-z[j]);
|
||||
u2 = d_j * u2 + u1;
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
double *p1 = p0 + pN, *p2 = p0 + 2 * pN;
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p1[i] = 2.0 * lCoeff[i] * u1;
|
||||
p2[i] = 8.0 * lCoeff[i] * u2;
|
||||
}
|
||||
|
||||
/* positive when possibly inside */
|
||||
static MFEM_HOST_DEVICE inline double obbox_axis_test(const obbox_t *const b,
|
||||
const double x[sDIM])
|
||||
{
|
||||
double b_d;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
|
||||
if (b_d < 0) // if outside in any dimension
|
||||
{
|
||||
return b_d;
|
||||
}
|
||||
}
|
||||
return b_d; // only positive if inside in all dimensions
|
||||
}
|
||||
|
||||
/* positive when possibly inside */
|
||||
static MFEM_HOST_DEVICE inline double obbox_test(const obbox_t *const b,
|
||||
const double x[sDIM])
|
||||
{
|
||||
const double bxyz = obbox_axis_test(b, x);
|
||||
if (bxyz<0)
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
else
|
||||
{
|
||||
double dxyz[3];
|
||||
// dxyz: distance of the point from the center of the OBB
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
dxyz[d] = x[d] - b->c0[d];
|
||||
}
|
||||
// transform dxyz to the local coordinate system of the OBB,
|
||||
// and check if the point is inside the OBB [-1,1]^sDIM
|
||||
double test = 1;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
double rst = 0;
|
||||
for (int e=0; e<sDIM; ++e)
|
||||
{
|
||||
rst += b->A[d*sDIM + e] * dxyz[e];
|
||||
}
|
||||
double brst = (rst+1)*(1-rst);
|
||||
test = test<0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
}
|
||||
|
||||
/* Hash index in the hash table to the elements that possibly contain the point x */
|
||||
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
|
||||
const double x[sDIM])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d=sDIM-1; d>=0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
|
||||
sum += i<0 ? 0 : (n-1 < i ? n-1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
|
||||
static MFEM_HOST_DEVICE inline double norm2(const double x[sDIM])
|
||||
{
|
||||
return ( x[0]*x[0] + x[1]*x[1] + x[2]*x[2] );
|
||||
}
|
||||
using dbl_range_t = gslib::dbl_range_t;
|
||||
using obbox_t = gslib::obbox_t<sDIM>;
|
||||
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<sDIM>;
|
||||
using gslib::AABB_test;
|
||||
using gslib::bbox_test;
|
||||
using gslib::hash_index;
|
||||
using gslib::l2norm2;
|
||||
using gslib::lag_eval_second_der;
|
||||
|
||||
/* the bit structure of flags is CRR
|
||||
the C bit --- 1<<2 --- is set when the point is converged
|
||||
@@ -175,47 +73,46 @@ static MFEM_HOST_DEVICE inline double norm2(const double x[sDIM])
|
||||
#define CONVERGED_FLAG (1u<<2)
|
||||
#define FLAG_MASK 0x07u
|
||||
|
||||
/* returns the number of constrained reference coordinates, max 2
|
||||
/* returns the number of constrained reference coordinates, max 1
|
||||
*/
|
||||
static MFEM_HOST_DEVICE inline int num_constrained(const int flags)
|
||||
{
|
||||
const int y = (flags | flags>>1);
|
||||
return (y & 1u) + (y>>2 & 1u);
|
||||
return ((flags | flags>>1) & 1u);
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline int point_index(const int x)
|
||||
{
|
||||
return ((x>>1)&1u) | ((x>>2)&2u);
|
||||
return ((x>>1)&1u);
|
||||
}
|
||||
|
||||
/* check reduction in objective against prediction, and adjust
|
||||
trust region radius (p->tr) accordingly;
|
||||
may reject the prior step, returning 1; otherwise returns 0
|
||||
sets out->dist2, out->index, out->x, out->oldr in any event,
|
||||
leaving out->r, out->dr, out->flags to be set when returning 0 */
|
||||
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
|
||||
sets out_pt->dist2, out_pt->index, out_pt->x, out_pt->oldr in any event,
|
||||
leaving out_pt->r, out_pt->dr, out_pt->flags to be set when returning 0 */
|
||||
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out_pt,
|
||||
const double resid[3],
|
||||
const findptsElementPoint_t *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = norm2(resid);
|
||||
const double dist2 = l2norm2<sDIM>(resid);
|
||||
const double decr = p->dist2 - dist2;
|
||||
const double pred = p->dist2p;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
out->x[d] = p->x[d];
|
||||
out_pt->x[d] = p->x[d];
|
||||
}
|
||||
out->oldr = p->r;
|
||||
out->dist2 = dist2;
|
||||
out_pt->oldr = p->r;
|
||||
out_pt->dist2 = dist2;
|
||||
if (decr>=0.01*pred)
|
||||
{
|
||||
if (decr>=0.9*pred) // very good iteration
|
||||
{
|
||||
out->tr = 2*p->tr;
|
||||
out_pt->tr = 2*p->tr;
|
||||
}
|
||||
else // good iteration
|
||||
{
|
||||
out->tr = p->tr;
|
||||
out_pt->tr = p->tr;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -226,21 +123,21 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
|
||||
"very good iteration" --- this doubles the trust radius,
|
||||
which is why we divide by 4 below */
|
||||
double v0 = fabs(p->r - p->oldr);
|
||||
out->tr = v0/4.0;
|
||||
out->dist2 = p->dist2;
|
||||
out->r = p->oldr;
|
||||
out->flags = p->flags>>3;
|
||||
out->dist2p = -HUGE_VAL;
|
||||
out_pt->tr = v0/4.0;
|
||||
out_pt->dist2 = p->dist2;
|
||||
out_pt->r = p->oldr;
|
||||
out_pt->flags = p->flags>>3;
|
||||
out_pt->dist2p = -HUGE_VAL;
|
||||
if (pred<dist2*tol)
|
||||
{
|
||||
out->flags |= CONVERGED_FLAG;
|
||||
out_pt->flags |= CONVERGED_FLAG;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline void newton_edge(findptsElementPoint_t *const
|
||||
out,
|
||||
out_pt,
|
||||
const double jac[sDIM*rDIM],
|
||||
const double rhes,
|
||||
const double resid[sDIM],
|
||||
@@ -314,9 +211,9 @@ newton_edge_fin:
|
||||
{
|
||||
new_flags |= CONVERGED_FLAG;
|
||||
}
|
||||
out->r = nr;
|
||||
out->dist2p = -v;
|
||||
out->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
|
||||
out_pt->r = nr;
|
||||
out_pt->dist2p = -v;
|
||||
out_pt->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
|
||||
#undef EVAL
|
||||
}
|
||||
|
||||
@@ -338,31 +235,32 @@ static MFEM_HOST_DEVICE void seed_j(const double *elx[sDIM],
|
||||
{
|
||||
dx[d] = x[d] - elx[d][ir];
|
||||
}
|
||||
dist2[ir] = norm2(dx);;
|
||||
dist2[ir] = l2norm2(dx);
|
||||
r[ir] = z[ir];
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void FindPointsEdgeLocal3D_Kernel(const int npt,
|
||||
const double tol,
|
||||
const double dist2tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
static void FindPointsEdgeLocal3DKernel(const int npt,
|
||||
const double tol,
|
||||
const double dist2tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const bool obb_check,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
{
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
@@ -419,21 +317,35 @@ static void FindPointsEdgeLocal3D_Kernel(const int npt,
|
||||
for (; elp!=ele; ++elp)
|
||||
{
|
||||
const unsigned int el = *elp;
|
||||
|
||||
const int n_box_ents = obb_check ? (3*sDIM + sDIM2) : (2*sDIM);
|
||||
bool pass_bb = true;
|
||||
obbox_t box;
|
||||
int n_box_ents = 3*sDIM + sDIM2;
|
||||
|
||||
for (int idx = 0; idx < sDIM; ++idx)
|
||||
if (obb_check)
|
||||
{
|
||||
box.c0[idx] = boxinfo[n_box_ents*el + idx];
|
||||
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
|
||||
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
|
||||
for (int idx = 0; idx < sDIM; ++idx)
|
||||
{
|
||||
box.c0[idx] = boxinfo[n_box_ents*el + idx];
|
||||
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
|
||||
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
|
||||
}
|
||||
for (int idx = 0; idx < sDIM2; ++idx)
|
||||
{
|
||||
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
|
||||
}
|
||||
pass_bb = (bbox_test(&box, x_i) >= 0);
|
||||
}
|
||||
for (int idx = 0; idx < sDIM2; ++idx)
|
||||
else
|
||||
{
|
||||
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
|
||||
for (int d = 0; d < sDIM; ++d)
|
||||
{
|
||||
box.x[d].min = boxinfo[n_box_ents*el + d];
|
||||
box.x[d].max = boxinfo[n_box_ents*el + sDIM + d];
|
||||
}
|
||||
pass_bb = (AABB_test(&box, x_i) >= 0);
|
||||
}
|
||||
|
||||
if (obbox_test(&box, x_i)>=0)
|
||||
if (pass_bb)
|
||||
{
|
||||
//// findpts_local ////
|
||||
{
|
||||
@@ -521,11 +433,14 @@ static void FindPointsEdgeLocal3D_Kernel(const int npt,
|
||||
double *hess = jac + sDIM*rDIM;
|
||||
|
||||
findptsElementGEdge_t edge;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d] = constraint_workspace + d*D1D;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d] = constraint_workspace + d*D1D;
|
||||
edge.x[d][j] = elx[d][j];
|
||||
}
|
||||
}
|
||||
@@ -688,28 +603,41 @@ void FindPointsGSLIB::FindPointsEdgeLocal3(const Vector &point_pos,
|
||||
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
|
||||
auto plc = DEV.lagcoeff.Read(use_dev);
|
||||
double dist2tol = DEV.surf_dist_tol;
|
||||
const bool obb_chk = obb_check;
|
||||
switch (DEV.dof1d)
|
||||
{
|
||||
case 2:
|
||||
return FindPointsEdgeLocal3D_Kernel<2>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
FindPointsEdgeLocal3DKernel<2>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 3:
|
||||
return FindPointsEdgeLocal3D_Kernel<3>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
FindPointsEdgeLocal3DKernel<3>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 4:
|
||||
return FindPointsEdgeLocal3D_Kernel<4>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
FindPointsEdgeLocal3DKernel<4>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
default:
|
||||
return FindPointsEdgeLocal3D_Kernel(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc, DEV.dof1d);
|
||||
FindPointsEdgeLocal3DKernel(npt, DEV.newt_tol, dist2tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc, DEV.dof1d);
|
||||
break;
|
||||
}
|
||||
}
|
||||
#undef rDIM2
|
||||
|
||||
+131
-206
@@ -11,6 +11,7 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
@@ -51,124 +52,15 @@ struct findptsElementGPT_t
|
||||
double x[sDIM], jac[sDIM*rDIM], hes[sDIM*(rDIM+1)];
|
||||
};
|
||||
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[sDIM], A[sDIM*sDIM];
|
||||
dbl_range_t x[sDIM];
|
||||
};
|
||||
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[sDIM];
|
||||
double fac[sDIM];
|
||||
unsigned int *offset;
|
||||
};
|
||||
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2 * (x - z[j]);
|
||||
u2 = d_j * u2 + u1;
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p0[pN+i] = 2.0 * lCoeff[i] * u1;
|
||||
p0[2*pN+i] = 8.0 * lCoeff[i] * u2;
|
||||
}
|
||||
|
||||
/* positive when possibly inside */
|
||||
static MFEM_HOST_DEVICE inline double AABB_test(const obbox_t *const b,
|
||||
const double x[sDIM])
|
||||
{
|
||||
double b_d;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
|
||||
if (b_d < 0) // if outside in any dimension
|
||||
{
|
||||
return b_d;
|
||||
}
|
||||
}
|
||||
return b_d; // only positive if inside in all dimensions
|
||||
}
|
||||
|
||||
/* positive when possibly inside */
|
||||
static MFEM_HOST_DEVICE inline double bbox_test(const obbox_t *const b,
|
||||
const double x[sDIM])
|
||||
{
|
||||
const double bxyz = AABB_test(b, x);
|
||||
if (bxyz<0)
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
else
|
||||
{
|
||||
double dxyz[3];
|
||||
// dxyz: distance of the point from the center of the OBB
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
dxyz[d] = x[d] - b->c0[d];
|
||||
}
|
||||
// tranform dxyz to the local coordinate system of the OBB,
|
||||
// and check if the point is inside the OBB [-1,1]^sDIM
|
||||
double test = 1;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
double rst = 0;
|
||||
for (int e=0; e<sDIM; ++e)
|
||||
{
|
||||
rst += b->A[d*sDIM + e] * dxyz[e];
|
||||
}
|
||||
double brst = (rst+1)*(1-rst);
|
||||
test = test<0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
}
|
||||
|
||||
/* Hash index in the hash table to the elements that possibly contain the point x */
|
||||
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
|
||||
const double x[sDIM])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d=sDIM-1; d>=0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
|
||||
sum += i<0 ? 0 : (n-1 < i ? n-1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline void lin_solve_sym_2(double x[2],
|
||||
const double A[3],
|
||||
const double y[2])
|
||||
{
|
||||
const double idet = 1 / (A[0] * A[2] - A[1] * A[1]);
|
||||
x[0] = idet * (A[2] * y[0] - A[1] * y[1]);
|
||||
x[1] = idet * (A[0] * y[1] - A[1] * y[0]);
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline double l2norm2(const double x[sDIM])
|
||||
{
|
||||
return ( x[0]*x[0] + x[1]*x[1] + x[2]*x[2]);
|
||||
}
|
||||
using dbl_range_t = gslib::dbl_range_t;
|
||||
using obbox_t = gslib::obbox_t<sDIM>;
|
||||
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<sDIM>;
|
||||
using gslib::AABB_test;
|
||||
using gslib::bbox_test;
|
||||
using gslib::hash_index;
|
||||
using gslib::l2norm2;
|
||||
using gslib::lag_eval_second_der;
|
||||
using gslib::lin_solve_sym_2;
|
||||
|
||||
/* the bit structure of flags is CSSRR
|
||||
the C bit --- 1<<4 --- is set when the point is converged
|
||||
@@ -219,18 +111,10 @@ static MFEM_HOST_DEVICE inline int point_index(const int x)
|
||||
return ((x>>1)&1u) | ((x>>2)&2u);
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline findptsElementGEdge_t
|
||||
static MFEM_HOST_DEVICE inline void
|
||||
get_edge(const double *elx[3], const double *wtend, int ei,
|
||||
double *workspace, int &side_init, int jidx, int pN)
|
||||
int &side_init, int jidx, int pN, findptsElementGEdge_t &edge)
|
||||
{
|
||||
findptsElementGEdge_t edge;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d] = workspace + d*pN;
|
||||
edge.dxdn[d] = workspace + sDIM*pN + d*pN;
|
||||
edge.d2xdn[d] = workspace + 2*sDIM*pN + d*pN;
|
||||
}
|
||||
|
||||
// given edge index, compute normal and tangential directions
|
||||
const int dn = ei>>1, //0 for rmin/rmax, 1 for smin/smax
|
||||
de = plus_1_mod_2(dn); // 1 for rmin/rmax, 0 for smin/smax
|
||||
@@ -256,7 +140,6 @@ get_edge(const double *elx[3], const double *wtend, int ei,
|
||||
edge.d2xdn[dd][jj] = sums_k[1];
|
||||
#undef ELX
|
||||
}
|
||||
return edge;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline findptsElementGPT_t get_pt(const double *elx[3],
|
||||
@@ -312,34 +195,34 @@ static MFEM_HOST_DEVICE inline findptsElementGPT_t get_pt(const double *elx[3],
|
||||
/* check reduction in objective against prediction, and adjust
|
||||
trust region radius (p->tr) accordingly;
|
||||
may reject the prior step, returning 1; otherwise returns 0
|
||||
sets out->dist2, out->index, out->x, out->oldr in any event,
|
||||
leaving out->r, out->dr, out->flags to be set when returning 0 */
|
||||
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
|
||||
sets out_pt->dist2, out_pt->index, out_pt->x, out_pt->oldr in any event,
|
||||
leaving out_pt->r, out_pt->dr, out_pt->flags to be set when returning 0 */
|
||||
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out_pt,
|
||||
const double resid[3],
|
||||
const findptsElementPoint_t *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = l2norm2(resid);
|
||||
const double dist2 = l2norm2<sDIM>(resid);
|
||||
const double decr = p->dist2 - dist2;
|
||||
const double pred = p->dist2p;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
out->x[d] = p->x[d];
|
||||
out_pt->x[d] = p->x[d];
|
||||
}
|
||||
for (int d=0; d<rDIM; ++d)
|
||||
{
|
||||
out->oldr[d] = p->r[d];
|
||||
out_pt->oldr[d] = p->r[d];
|
||||
}
|
||||
out->dist2 = dist2;
|
||||
out_pt->dist2 = dist2;
|
||||
if (decr>=0.01*pred)
|
||||
{
|
||||
if (decr>=0.9*pred) // very good iteration
|
||||
{
|
||||
out->tr = 2*p->tr;
|
||||
out_pt->tr = 2*p->tr;
|
||||
}
|
||||
else // good iteration
|
||||
{
|
||||
out->tr = p->tr;
|
||||
out_pt->tr = p->tr;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -351,17 +234,17 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
|
||||
which is why we divide by 4 below */
|
||||
double v0 = fabs(p->r[0] - p->oldr[0]),
|
||||
v1 = fabs(p->r[1] - p->oldr[1]);
|
||||
out->tr = ( v0>v1 ? v0 : v1 )/4;
|
||||
out->dist2 = p->dist2;
|
||||
out->flags = p->flags >> 5;
|
||||
out->dist2p = -HUGE_VAL;
|
||||
out_pt->tr = ( v0>v1 ? v0 : v1 )/4;
|
||||
out_pt->dist2 = p->dist2;
|
||||
out_pt->flags = p->flags >> 5;
|
||||
out_pt->dist2p = -HUGE_VAL;
|
||||
for (int d=0; d<rDIM; ++d)
|
||||
{
|
||||
out->r[d] = p->oldr[d];
|
||||
out_pt->r[d] = p->oldr[d];
|
||||
}
|
||||
if (pred<dist2*tol)
|
||||
{
|
||||
out->flags |= CONVERGED_FLAG;
|
||||
out_pt->flags |= CONVERGED_FLAG;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -369,7 +252,7 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
|
||||
|
||||
/* minimize ||resid - jac * dr||_2, with |dr| <= tr, |r0+dr|<=1
|
||||
(exact solution of trust region problem) */
|
||||
static MFEM_HOST_DEVICE void newton_face( findptsElementPoint_t *const out,
|
||||
static MFEM_HOST_DEVICE void newton_face( findptsElementPoint_t *const out_pt,
|
||||
const double jac[sDIM*rDIM],
|
||||
const double rhes[3],
|
||||
const double resid[sDIM],
|
||||
@@ -540,19 +423,19 @@ newton_face_constrained:
|
||||
}
|
||||
|
||||
newton_face_fin:
|
||||
out->dist2p = -2*v;
|
||||
out_pt->dist2p = -2*v;
|
||||
dr[0] = r[0] - p->r[0];
|
||||
dr[1] = r[1] - p->r[1];
|
||||
if ( fabs(dr[0])+fabs(dr[1]) < tol)
|
||||
{
|
||||
new_flags |= CONVERGED_FLAG;
|
||||
}
|
||||
out->r[0] = r[0], out->r[1] = r[1];
|
||||
out->flags = new_flags | ((p->flags & FLAG_MASK)<<5);
|
||||
out_pt->r[0] = r[0], out_pt->r[1] = r[1];
|
||||
out_pt->flags = new_flags | ((p->flags & FLAG_MASK)<<5);
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline void newton_edge(findptsElementPoint_t *const
|
||||
out,
|
||||
out_pt,
|
||||
const double jac[sDIM*rDIM],
|
||||
const double rhes,
|
||||
const double resid[sDIM],
|
||||
@@ -637,10 +520,10 @@ newton_edge_fin:
|
||||
{
|
||||
new_flags |= CONVERGED_FLAG;
|
||||
}
|
||||
out->r[de] = nr;
|
||||
out->r[dn] = p->r[dn];
|
||||
out->dist2p = -v;
|
||||
out->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<5);
|
||||
out_pt->r[de] = nr;
|
||||
out_pt->r[dn] = p->r[dn];
|
||||
out_pt->dist2p = -v;
|
||||
out_pt->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<5);
|
||||
#undef EVAL
|
||||
}
|
||||
|
||||
@@ -676,26 +559,27 @@ static MFEM_HOST_DEVICE void seed_j(const double *elx[sDIM],
|
||||
// global memory access of element coordinates.
|
||||
// Are the structs being stored in "local memory" or registers?
|
||||
template<int T_D1D = 0>
|
||||
static void FindPointsSurfLocal3D_Kernel(const int npt,
|
||||
const double tol,
|
||||
const double dist2tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
static void FindPointsSurfLocal3DKernel(const int npt,
|
||||
const double tol,
|
||||
const double dist2tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const bool obb_check,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
{
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
@@ -753,22 +637,36 @@ static void FindPointsSurfLocal3D_Kernel(const int npt,
|
||||
{
|
||||
const unsigned int el = *elp;
|
||||
|
||||
// construct obbox on the fly
|
||||
const int n_box_ents = obb_check ? (3*sDIM + sDIM2) : (2*sDIM);
|
||||
bool pass_bb = true;
|
||||
obbox_t box;
|
||||
int n_box_ents = 3*sDIM + sDIM2;
|
||||
for (int idx = 0; idx < sDIM; ++idx)
|
||||
if (obb_check)
|
||||
{
|
||||
box.c0[idx] = boxinfo[n_box_ents*el + idx];
|
||||
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
|
||||
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
|
||||
// construct obbox on the fly
|
||||
for (int idx = 0; idx < sDIM; ++idx)
|
||||
{
|
||||
box.c0[idx] = boxinfo[n_box_ents*el + idx];
|
||||
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
|
||||
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
|
||||
}
|
||||
|
||||
for (int idx = 0; idx < sDIM2; ++idx)
|
||||
{
|
||||
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
|
||||
}
|
||||
pass_bb = (bbox_test(&box, x_i) >= 0);
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int d = 0; d < sDIM; ++d)
|
||||
{
|
||||
box.x[d].min = boxinfo[n_box_ents*el + d];
|
||||
box.x[d].max = boxinfo[n_box_ents*el + sDIM + d];
|
||||
}
|
||||
pass_bb = (AABB_test(&box, x_i) >= 0);
|
||||
}
|
||||
|
||||
for (int idx = 0; idx < sDIM2; ++idx)
|
||||
{
|
||||
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
|
||||
}
|
||||
|
||||
if (bbox_test(&box, x_i) < 0) { continue; }
|
||||
if (!pass_bb) { continue; }
|
||||
|
||||
//// findpts_local ////
|
||||
{
|
||||
@@ -968,13 +866,19 @@ static void FindPointsSurfLocal3D_Kernel(const int npt,
|
||||
double *hes_T = jac + sDIM*rDIM;
|
||||
double *hes = hes_T + hes_count*sDIM;
|
||||
findptsElementGEdge_t edge;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d] = constraint_workspace + d*D1D;
|
||||
edge.dxdn[d] = constraint_workspace + d*D1D
|
||||
+ sDIM*D1D;
|
||||
edge.d2xdn[d] = constraint_workspace + d*D1D
|
||||
+ 2*sDIM*D1D;
|
||||
}
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,D1D*sDIM)
|
||||
{
|
||||
// utilized first D1D threads
|
||||
edge = get_edge(elx, wtend, ei,
|
||||
constraint_workspace, edge_init, j,
|
||||
D1D);
|
||||
// One thread per physical component and edge DOF.
|
||||
get_edge(elx, wtend, ei, edge_init, j, D1D, edge);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
@@ -1045,7 +949,15 @@ static void FindPointsSurfLocal3D_Kernel(const int npt,
|
||||
steep *= tmp->r[dn];
|
||||
if (steep<0)
|
||||
{
|
||||
newton_face( fpt,jac,hes,resid,tmp->flags&CONVERGED_FLAG,tmp,tol);
|
||||
double face_hes[3] =
|
||||
{
|
||||
dn == 0 ? hes[2] : hes[0],
|
||||
hes[1],
|
||||
dn == 0 ? hes[0] : hes[2]
|
||||
};
|
||||
newton_face(fpt, jac, face_hes, resid,
|
||||
tmp->flags & CONVERGED_FLAG,
|
||||
tmp, tol);
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -1211,29 +1123,42 @@ void FindPointsGSLIB::FindPointsSurfLocal3(const Vector &point_pos,
|
||||
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
|
||||
auto plc = DEV.lagcoeff.Read(use_dev);
|
||||
double dist2tol = DEV.surf_dist_tol;
|
||||
const bool obb_chk = obb_check;
|
||||
|
||||
switch (DEV.dof1d)
|
||||
{
|
||||
case 2:
|
||||
return FindPointsSurfLocal3D_Kernel<2>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
FindPointsSurfLocal3DKernel<2>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 3:
|
||||
return FindPointsSurfLocal3D_Kernel<3>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
FindPointsSurfLocal3DKernel<3>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 4:
|
||||
return FindPointsSurfLocal3D_Kernel<4>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
FindPointsSurfLocal3DKernel<4>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
default:
|
||||
return FindPointsSurfLocal3D_Kernel(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc, DEV.dof1d);
|
||||
FindPointsSurfLocal3DKernel(npt, DEV.newt_tol, dist2tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc, DEV.dof1d);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,190 @@
|
||||
#ifndef MFEM_GSLIB_KERNEL_HELPERS_HPP
|
||||
#define MFEM_GSLIB_KERNEL_HELPERS_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
|
||||
#include <cmath>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace gslib
|
||||
{
|
||||
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
|
||||
template <int SDIM>
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[SDIM], A[SDIM * SDIM];
|
||||
dbl_range_t x[SDIM];
|
||||
};
|
||||
|
||||
template <int SDIM>
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[SDIM];
|
||||
double fac[SDIM];
|
||||
unsigned int *offset;
|
||||
};
|
||||
|
||||
// Eval the ith Lagrange interpolant at x.
|
||||
MFEM_HOST_DEVICE inline void lagrange_eval(double *p0, double x,
|
||||
int i, int p_Nq,
|
||||
double *z, double *lagrangeCoeff)
|
||||
{
|
||||
double p_i = (1 << (p_Nq - 1));
|
||||
for (int j = 0; j < p_Nq; ++j)
|
||||
{
|
||||
const double d_j = x - z[j];
|
||||
p_i *= j == i ? 1 : d_j;
|
||||
}
|
||||
p0[i] = lagrangeCoeff[i] * p_i;
|
||||
}
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first derivative at x.
|
||||
MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
const double d_j = 2 * (x - z[j]);
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p0[pN + i] = 2.0 * lCoeff[i] * u1;
|
||||
}
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first and second derivative at x.
|
||||
MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
const double d_j = 2 * (x - z[j]);
|
||||
u2 = d_j * u2 + u1;
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p0[pN + i] = 2.0 * lCoeff[i] * u1;
|
||||
p0[2 * pN + i] = 8.0 * lCoeff[i] * u2;
|
||||
}
|
||||
|
||||
// Solve Ax=y where A is a symmetric 2x2 matrix packed as {a00, a01, a11}.
|
||||
MFEM_HOST_DEVICE inline void lin_solve_sym_2(double x[2],
|
||||
const double A[3],
|
||||
const double y[2])
|
||||
{
|
||||
const double idet = 1 / (A[0] * A[2] - A[1] * A[1]);
|
||||
x[0] = idet * (A[2] * y[0] - A[1] * y[1]);
|
||||
x[1] = idet * (A[0] * y[1] - A[1] * y[0]);
|
||||
}
|
||||
|
||||
// Positive when the point is inside the axis-aligned bounding box.
|
||||
template <int SDIM>
|
||||
MFEM_HOST_DEVICE inline double AABB_test(const obbox_t<SDIM> *const b,
|
||||
const double (&x)[SDIM])
|
||||
{
|
||||
double test = 1.0;
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
const double b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
|
||||
test = test < 0.0 ? test : b_d;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
|
||||
// Positive when the point is inside the oriented bounding box.
|
||||
template <int SDIM>
|
||||
MFEM_HOST_DEVICE inline double bbox_test(const obbox_t<SDIM> *const b,
|
||||
const double (&x)[SDIM])
|
||||
{
|
||||
const double bxyz = AABB_test(b, x);
|
||||
if (bxyz < 0.0)
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
|
||||
double dxyz[SDIM];
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
dxyz[d] = x[d] - b->c0[d];
|
||||
}
|
||||
|
||||
double test = 1.0;
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
double rst = 0.0;
|
||||
for (int e = 0; e < SDIM; ++e)
|
||||
{
|
||||
rst += b->A[d * SDIM + e] * dxyz[e];
|
||||
}
|
||||
const double brst = (rst + 1.0) * (1.0 - rst);
|
||||
test = test < 0.0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
|
||||
// Hash index in the hash table for the point x.
|
||||
template <int SDIM>
|
||||
MFEM_HOST_DEVICE inline int hash_index(
|
||||
const findptsLocalHashData_t<SDIM> *const p,
|
||||
const double (&x)[SDIM])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d = SDIM - 1; d >= 0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
const int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
|
||||
sum += i < 0 ? 0 : (n - 1 < i ? n - 1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
// Squared Euclidean norm.
|
||||
template <int SDIM>
|
||||
MFEM_HOST_DEVICE inline double l2norm2(const double (&x)[SDIM])
|
||||
{
|
||||
double sum = 0.0;
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
sum += x[d] * x[d];
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
template <int SDIM>
|
||||
MFEM_HOST_DEVICE inline double l2norm2(const double *x)
|
||||
{
|
||||
double sum = 0.0;
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
sum += x[d] * x[d];
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
} // namespace gslib
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
@@ -11,7 +11,7 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/kernels.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -33,17 +33,7 @@ namespace mfem
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
|
||||
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
|
||||
int i, int p_Nq,
|
||||
double *z, double *lagrangeCoeff)
|
||||
{
|
||||
double p_i = (1 << (p_Nq - 1));
|
||||
for (int j=0; j<p_Nq; ++j)
|
||||
{
|
||||
p_i *= j==i ? 1 : x-z[j];
|
||||
}
|
||||
p0[i] = lagrangeCoeff[i] * p_i;
|
||||
}
|
||||
using gslib::lagrange_eval;
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void InterpolateLocal1DKernel(const double *const gf_in,
|
||||
@@ -123,21 +113,26 @@ void FindPointsGSLIB::InterpolateLocal1( const Vector &field_in,
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
|
||||
switch (dof1Dsol)
|
||||
{
|
||||
case 2: return InterpolateLocal1DKernel<2>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 3: return InterpolateLocal1DKernel<3>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 4: return InterpolateLocal1DKernel<4>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 5: return InterpolateLocal1DKernel<5>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
default: return InterpolateLocal1DKernel(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf, dof1Dsol);
|
||||
case 2:
|
||||
InterpolateLocal1DKernel<2>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 3:
|
||||
InterpolateLocal1DKernel<3>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 4:
|
||||
InterpolateLocal1DKernel<4>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 5:
|
||||
InterpolateLocal1DKernel<5>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
default:
|
||||
InterpolateLocal1DKernel(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf, dof1Dsol);
|
||||
break;
|
||||
}
|
||||
}
|
||||
#undef CODE_INTERNAL
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -32,18 +33,7 @@ namespace mfem
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
|
||||
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
|
||||
int i, int p_Nq,
|
||||
double *z, double *lagrangeCoeff)
|
||||
{
|
||||
double p_i = (1 << (p_Nq - 1));
|
||||
for (int j = 0; j < p_Nq; ++j)
|
||||
{
|
||||
double d_j = x - z[j];
|
||||
p_i *= j == i ? 1 : d_j;
|
||||
}
|
||||
p0[i] = lagrangeCoeff[i] * p_i;
|
||||
}
|
||||
using gslib::lagrange_eval;
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void InterpolateLocal2DKernel(const double *const gf_in,
|
||||
@@ -132,21 +122,26 @@ void FindPointsGSLIB::InterpolateLocal2(const Vector &field_in,
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
|
||||
switch (dof1Dsol)
|
||||
{
|
||||
case 2: return InterpolateLocal2DKernel<2>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 3: return InterpolateLocal2DKernel<3>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 4: return InterpolateLocal2DKernel<4>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 5: return InterpolateLocal2DKernel<5>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
default: return InterpolateLocal2DKernel(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf, dof1Dsol);
|
||||
case 2:
|
||||
InterpolateLocal2DKernel<2>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 3:
|
||||
InterpolateLocal2DKernel<3>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 4:
|
||||
InterpolateLocal2DKernel<4>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 5:
|
||||
InterpolateLocal2DKernel<5>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
default:
|
||||
InterpolateLocal2DKernel(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf, dof1Dsol);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -32,18 +33,7 @@ namespace mfem
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
|
||||
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
|
||||
int i, int p_Nq,
|
||||
double *z, double *lagrangeCoeff)
|
||||
{
|
||||
double p_i = (1 << (p_Nq - 1));
|
||||
for (int j = 0; j < p_Nq; ++j)
|
||||
{
|
||||
double d_j = x - z[j];
|
||||
p_i *= j == i ? 1 : d_j;
|
||||
}
|
||||
p0[i] = lagrangeCoeff[i] * p_i;
|
||||
}
|
||||
using gslib::lagrange_eval;
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void InterpolateLocal3DKernel(const double *const gf_in,
|
||||
@@ -135,21 +125,26 @@ void FindPointsGSLIB::InterpolateLocal3(const Vector &field_in,
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
|
||||
switch (dof1Dsol)
|
||||
{
|
||||
case 2: return InterpolateLocal3DKernel<2>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 3: return InterpolateLocal3DKernel<3>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 4: return InterpolateLocal3DKernel<4>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 5: return InterpolateLocal3DKernel<5>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
default: return InterpolateLocal3DKernel(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf, dof1Dsol);
|
||||
case 2:
|
||||
InterpolateLocal3DKernel<2>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 3:
|
||||
InterpolateLocal3DKernel<3>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 4:
|
||||
InterpolateLocal3DKernel<4>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 5:
|
||||
InterpolateLocal3DKernel<5>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
default:
|
||||
InterpolateLocal3DKernel(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf, dof1Dsol);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -307,6 +307,506 @@ DomainLFIntegrator::AssembleKernels::Kernel()
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void HdivDLFAssemble2D(const int ne, const Array<int> &markers,
|
||||
const Vector &jac, const Array<real_t> &weights,
|
||||
const Array<real_t> &testBO,
|
||||
const Array<real_t> &testBC, const Vector &coeff,
|
||||
Vector &y, const int d, const int q)
|
||||
{
|
||||
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
|
||||
"Problem size too large.");
|
||||
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
|
||||
"Problem size too large.");
|
||||
MFEM_VERIFY(y.Size() == 2 * (d - 1) * d * ne, "");
|
||||
|
||||
constexpr int vdim = 2;
|
||||
const auto F = coeff.Read();
|
||||
const auto M = markers.Read();
|
||||
const auto BO = Reshape(testBO.Read(), q, d-1);
|
||||
const auto BC = Reshape(testBC.Read(), q, d);
|
||||
const auto J = Reshape(jac.Read(), q, q, vdim, vdim, ne);
|
||||
const auto W = Reshape(weights.Read(), q, q);
|
||||
const bool cst = coeff.Size() == vdim;
|
||||
const auto C = cst ? Reshape(F,vdim,1,1,1) : Reshape(F,vdim,q,q,ne);
|
||||
auto Y = y.ReadWrite();
|
||||
|
||||
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
constexpr int vdim = 2;
|
||||
if (M[e] == 0) { return; } // ignore
|
||||
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HDIV_MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HDIV_MAX_D1D;
|
||||
|
||||
MFEM_SHARED real_t sBot[Q*D];
|
||||
MFEM_SHARED real_t sBct[Q*D];
|
||||
MFEM_SHARED real_t sQQ[vdim*Q*Q];
|
||||
MFEM_SHARED real_t sQD[vdim*Q*D];
|
||||
|
||||
// Bo and Bc into shared memory
|
||||
const DeviceMatrix Bot(sBot, d-1, q);
|
||||
kernels::internal::LoadB<D,Q>(d-1, q, BO, sBot);
|
||||
const DeviceMatrix Bct(sBct, d, q);
|
||||
kernels::internal::LoadB<D,Q>(d, q, BC, sBct);
|
||||
|
||||
const DeviceCube QQ(sQQ, q, q, vdim);
|
||||
const DeviceCube QD(sQD, q, d, vdim);
|
||||
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const real_t cst_val_0 = C(0,0,0,0);
|
||||
const real_t cst_val_1 = C(1,0,0,0);
|
||||
MFEM_FOREACH_THREAD(y,y,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(x,x,q)
|
||||
{
|
||||
const real_t J0 = J(x,y,0,vd,e);
|
||||
const real_t J1 = J(x,y,1,vd,e);
|
||||
const real_t C0 = cst ? cst_val_0 : C(0,x,y,e);
|
||||
const real_t C1 = cst ? cst_val_1 : C(1,x,y,e);
|
||||
QQ(x,y,vd) = W(x,y)*(J0*C0 + J1*C1);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d : d-1;
|
||||
DeviceMatrix Btx = (vd == 0) ? Bct : Bot;
|
||||
MFEM_FOREACH_THREAD(qy,y,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,nx)
|
||||
{
|
||||
real_t qd = 0.0;
|
||||
for (int qx = 0; qx < q; ++qx)
|
||||
{
|
||||
qd += QQ(qx,qy,vd) * Btx(dx,qx);
|
||||
}
|
||||
QD(dx,qy,vd) = qd;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d : d-1;
|
||||
const int ny = (vd == 1) ? d : d-1;
|
||||
DeviceMatrix Bty = (vd == 1) ? Bct : Bot;
|
||||
DeviceTensor<4> Yxy(Y, nx, ny, vdim, ne);
|
||||
MFEM_FOREACH_THREAD(dy,y,ny)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,nx)
|
||||
{
|
||||
real_t dd = 0.0;
|
||||
for (int qy = 0; qy < q; ++qy)
|
||||
{
|
||||
dd += QD(dx,qy,vd) * Bty(dy,qy);
|
||||
}
|
||||
Yxy(dx,dy,vd,e) += dd;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
});
|
||||
}
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void HdivDLFAssemble3D(const int ne, const Array<int> &markers,
|
||||
const Vector &jac, const Array<real_t> &weights,
|
||||
const Array<real_t> &testBO,
|
||||
const Array<real_t> &testBC, const Vector &coeff,
|
||||
Vector &y, const int d, const int q)
|
||||
{
|
||||
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
|
||||
"Problem size too large.");
|
||||
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
|
||||
"Problem size too large.");
|
||||
MFEM_VERIFY(y.Size() == 3 * (d - 1) * (d - 1) * d * ne, "y wrong length");
|
||||
|
||||
constexpr int vdim = 3;
|
||||
const auto F = coeff.Read();
|
||||
const auto M = markers.Read();
|
||||
const auto BO = Reshape(testBO.Read(), q, d-1);
|
||||
const auto BC = Reshape(testBC.Read(), q, d);
|
||||
const auto J = Reshape(jac.Read(), q, q, q, vdim, vdim, ne);
|
||||
const auto W = Reshape(weights.Read(), q, q, q);
|
||||
const bool cst = coeff.Size() == vdim;
|
||||
const auto C = cst ? Reshape(F,vdim,1,1,1,1) : Reshape(F,vdim,q,q,q,ne);
|
||||
auto Y = y.ReadWrite();
|
||||
|
||||
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
constexpr int vdim = 3;
|
||||
if (M[e] == 0) { return; } // ignore
|
||||
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HDIV_MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HDIV_MAX_D1D;
|
||||
|
||||
MFEM_SHARED real_t sBot[Q*D];
|
||||
MFEM_SHARED real_t sBct[Q*D];
|
||||
|
||||
// Bo and Bc into shared memory
|
||||
const DeviceMatrix Bot(sBot, d-1, q);
|
||||
kernels::internal::LoadB<D,Q>(d-1, q, BO, sBot);
|
||||
const DeviceMatrix Bct(sBct, d, q);
|
||||
kernels::internal::LoadB<D,Q>(d, q, BC, sBct);
|
||||
|
||||
MFEM_SHARED real_t sm0[vdim*Q*Q*Q];
|
||||
MFEM_SHARED real_t sm1[vdim*Q*Q*Q];
|
||||
DeviceTensor<4> QQQ(sm1, q, q, q, vdim);
|
||||
DeviceTensor<4> DQQ(sm0, d, q, q, vdim);
|
||||
DeviceTensor<4> DDQ(sm1, d, d, q, vdim);
|
||||
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const real_t cst_val_0 = C(0,0,0,0,0);
|
||||
const real_t cst_val_1 = C(1,0,0,0,0);
|
||||
const real_t cst_val_2 = C(2,0,0,0,0);
|
||||
MFEM_FOREACH_THREAD(y,y,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(x,x,q)
|
||||
{
|
||||
for (int z = 0; z < q; ++z)
|
||||
{
|
||||
const real_t J0 = J(x,y,z,0,vd,e);
|
||||
const real_t J1 = J(x,y,z,1,vd,e);
|
||||
const real_t J2 = J(x,y,z,2,vd,e);
|
||||
const real_t C0 = cst ? cst_val_0 : C(0,x,y,z,e);
|
||||
const real_t C1 = cst ? cst_val_1 : C(1,x,y,z,e);
|
||||
const real_t C2 = cst ? cst_val_2 : C(2,x,y,z,e);
|
||||
QQQ(x,y,z,vd) = W(x,y,z)*(J0*C0 + J1*C1 + J2*C2);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// Apply Bt operator
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d : d-1;
|
||||
DeviceMatrix Btx = (vd == 0) ? Bct : Bot;
|
||||
MFEM_FOREACH_THREAD(qy,y,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,nx)
|
||||
{
|
||||
real_t u[Q];
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz) { u[qz] = 0.0; }
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qx = 0; qx < q; ++qx)
|
||||
{
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
u[qz] += QQQ(qx,qy,qz,vd) * Btx(dx,qx);
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz) { DQQ(dx,qy,qz,vd) = u[qz]; }
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d : d-1;
|
||||
const int ny = (vd == 1) ? d : d-1;
|
||||
DeviceMatrix Bty = (vd == 1) ? Bct : Bot;
|
||||
MFEM_FOREACH_THREAD(dy,y,ny)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,nx)
|
||||
{
|
||||
real_t u[Q];
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz) { u[qz] = 0.0; }
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qy = 0; qy < q; ++qy)
|
||||
{
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
u[qz] += DQQ(dx,qy,qz,vd) * Bty(dy,qy);
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz) { DDQ(dx,dy,qz,vd) = u[qz]; }
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d : d-1;
|
||||
const int ny = (vd == 1) ? d : d-1;
|
||||
const int nz = (vd == 2) ? d : d-1;
|
||||
DeviceTensor<5> Yxyz(Y, nx, ny, nz, vdim, ne);
|
||||
DeviceMatrix Btz = (vd == 2) ? Bct : Bot;
|
||||
MFEM_FOREACH_THREAD(dy,y,ny)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,nx)
|
||||
{
|
||||
real_t u[D];
|
||||
MFEM_UNROLL(D)
|
||||
for (int dz = 0; dz < nz; ++dz) { u[dz] = 0.0; }
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
MFEM_UNROLL(D)
|
||||
for (int dz = 0; dz < nz; ++dz)
|
||||
{
|
||||
u[dz] += DDQ(dx,dy,qz,vd) * Btz(dz,qz);
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(D)
|
||||
for (int dz = 0; dz < nz; ++dz) { Yxyz(dx,dy,dz,vd,e) += u[dz]; }
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
});
|
||||
}
|
||||
|
||||
/// @param ne number of elements
|
||||
/// @param markers array where entry markers[e] == 0 to skip assembly over
|
||||
/// element e element
|
||||
/// @param jac Spatial Jacobians evaluated at all quadrature points
|
||||
/// @param weights 1D quadrature weights
|
||||
/// @param testBO 1D open basis test functions
|
||||
/// @param testBC 1D closed basis test functions
|
||||
/// @param coeff coefficient values evaluated at quadrature points, possibly
|
||||
/// compressed.
|
||||
/// @param d number of 1D closed dofs
|
||||
/// @param q number of 1D quadrature points
|
||||
/// @tparam T_D1D maximum number of dofs along any direction, or 0
|
||||
/// @tparam T_Q1D maximum number of quadrature points along any direction, or 0
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void HcurlDLFAssemble3D(const int ne, const Array<int> &markers,
|
||||
const Vector &jac, const Array<real_t> &weights,
|
||||
const Array<real_t> &testBO,
|
||||
const Array<real_t> &testBC, const Vector &coeff,
|
||||
Vector &y, const int d, const int q)
|
||||
{
|
||||
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HCURL_MAX_D1D,
|
||||
"Problem size too large.");
|
||||
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HCURL_MAX_Q1D,
|
||||
"Problem size too large.");
|
||||
MFEM_VERIFY(y.Size() == 3 * (d - 1) * d * d * ne, "y wrong length");
|
||||
|
||||
constexpr int vdim = 3;
|
||||
const auto F = coeff.Read();
|
||||
const auto M = markers.Read();
|
||||
const auto BO = Reshape(testBO.Read(), q, d-1);
|
||||
const auto BC = Reshape(testBC.Read(), q, d);
|
||||
const auto J = Reshape(jac.Read(), q, q, q, vdim, vdim, ne);
|
||||
const auto W = Reshape(weights.Read(), q, q, q);
|
||||
const bool cst = coeff.Size() == vdim;
|
||||
const auto C = cst ? Reshape(F,vdim,1,1,1,1) : Reshape(F,vdim,q,q,q,ne);
|
||||
auto Y = y.ReadWrite();
|
||||
|
||||
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
if (M[e] == 0)
|
||||
{
|
||||
// ignore
|
||||
return;
|
||||
}
|
||||
|
||||
constexpr int vdim = 3;
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HCURL_MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HCURL_MAX_D1D;
|
||||
|
||||
MFEM_SHARED real_t sBot[Q * D];
|
||||
MFEM_SHARED real_t sBct[Q * D];
|
||||
|
||||
// Bo and Bc into shared memory
|
||||
const DeviceMatrix Bot(sBot, d - 1, q);
|
||||
kernels::internal::LoadB<D, Q>(d - 1, q, BO, sBot);
|
||||
const DeviceMatrix Bct(sBct, d, q);
|
||||
kernels::internal::LoadB<D, Q>(d, q, BC, sBct);
|
||||
|
||||
MFEM_SHARED real_t sm0[vdim * Q * Q * Q];
|
||||
MFEM_SHARED real_t sm1[vdim * Q * Q * Q];
|
||||
DeviceTensor<4> QQQ(sm1, q, q, q, vdim);
|
||||
DeviceTensor<4> DQQ(sm0, d, q, q, vdim);
|
||||
DeviceTensor<4> DDQ(sm1, d, d, q, vdim);
|
||||
|
||||
const real_t cst_val_0 = C(0, 0, 0, 0, 0);
|
||||
const real_t cst_val_1 = C(1, 0, 0, 0, 0);
|
||||
const real_t cst_val_2 = C(2, 0, 0, 0, 0);
|
||||
|
||||
MFEM_FOREACH_THREAD(vd, z, vdim)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(y, y, q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(x, x, q)
|
||||
{
|
||||
for (int z = 0; z < q; ++z)
|
||||
{
|
||||
real_t curr[3];
|
||||
curr[0] = cst ? cst_val_0 : C(0, x, y, z, e);
|
||||
curr[1] = cst ? cst_val_1 : C(1, x, y, z, e);
|
||||
curr[2] = cst ? cst_val_2 : C(2, x, y, z, e);
|
||||
|
||||
const real_t J11 = J(x, y, z, 0, 0, e);
|
||||
const real_t J21 = J(x, y, z, 1, 0, e);
|
||||
const real_t J31 = J(x, y, z, 2, 0, e);
|
||||
const real_t J12 = J(x, y, z, 0, 1, e);
|
||||
const real_t J22 = J(x, y, z, 1, 1, e);
|
||||
const real_t J32 = J(x, y, z, 2, 1, e);
|
||||
const real_t J13 = J(x, y, z, 0, 2, e);
|
||||
const real_t J23 = J(x, y, z, 1, 2, e);
|
||||
const real_t J33 = J(x, y, z, 2, 2, e);
|
||||
// adj(J)
|
||||
const real_t A11 = (J22 * J33) - (J23 * J32);
|
||||
const real_t A12 = (J32 * J13) - (J12 * J33);
|
||||
const real_t A13 = (J12 * J23) - (J22 * J13);
|
||||
const real_t A21 = (J31 * J23) - (J21 * J33);
|
||||
const real_t A22 = (J11 * J33) - (J13 * J31);
|
||||
const real_t A23 = (J21 * J13) - (J11 * J23);
|
||||
const real_t A31 = (J21 * J32) - (J31 * J22);
|
||||
const real_t A32 = (J31 * J12) - (J11 * J32);
|
||||
const real_t A33 = (J11 * J22) - (J12 * J21);
|
||||
const real_t A[9] = {A11, A12, A13, A21, A22,
|
||||
A23, A31, A32, A33
|
||||
};
|
||||
QQQ(x, y, z, vd) = W(x, y, z) * (A[vd * vdim] * curr[0] +
|
||||
A[vd * vdim + 1] * curr[1] +
|
||||
A[vd * vdim + 2] * curr[2]);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// Apply Bt operator
|
||||
MFEM_FOREACH_THREAD(vd, z, vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d - 1 : d;
|
||||
DeviceMatrix Btx = (vd == 0) ? Bot : Bct;
|
||||
MFEM_FOREACH_THREAD(qy, y, q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, nx)
|
||||
{
|
||||
real_t u[Q];
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
u[qz] = 0.0;
|
||||
}
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qx = 0; qx < q; ++qx)
|
||||
{
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
u[qz] += QQQ(qx, qy, qz, vd) * Btx(dx, qx);
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
DQQ(dx, qy, qz, vd) = u[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(vd, z, vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d - 1 : d;
|
||||
const int ny = (vd == 1) ? d - 1 : d;
|
||||
DeviceMatrix Bty = (vd == 1) ? Bot : Bct;
|
||||
MFEM_FOREACH_THREAD(dy, y, ny)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, nx)
|
||||
{
|
||||
real_t u[Q];
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
u[qz] = 0.0;
|
||||
}
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qy = 0; qy < q; ++qy)
|
||||
{
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
u[qz] += DQQ(dx, qy, qz, vd) * Bty(dy, qy);
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
DDQ(dx, dy, qz, vd) = u[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(vd, z, vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d - 1 : d;
|
||||
const int ny = (vd == 1) ? d - 1 : d;
|
||||
const int nz = (vd == 2) ? d - 1 : d;
|
||||
DeviceTensor<5> Yxyz(Y, nx, ny, nz, vdim, ne);
|
||||
DeviceMatrix Btz = (vd == 2) ? Bot : Bct;
|
||||
MFEM_FOREACH_THREAD(dy, y, ny)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, nx)
|
||||
{
|
||||
real_t u[D];
|
||||
MFEM_UNROLL(D)
|
||||
for (int dz = 0; dz < nz; ++dz)
|
||||
{
|
||||
u[dz] = 0.0;
|
||||
}
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
MFEM_UNROLL(D)
|
||||
for (int dz = 0; dz < nz; ++dz)
|
||||
{
|
||||
u[dz] += DDQ(dx, dy, qz, vd) * Btz(dz, qz);
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(D)
|
||||
for (int dz = 0; dz < nz; ++dz)
|
||||
{
|
||||
Yxyz(dx, dy, dz, vd, e) += u[dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
});
|
||||
}
|
||||
|
||||
template <FiniteElement::DerivType TestType, int DIM, int TEST_D1D, int Q1D>
|
||||
VectorFEDomainLFIntegrator::AssembleKernelType
|
||||
VectorFEDomainLFIntegrator::AssembleKernels::Kernel()
|
||||
{
|
||||
if constexpr (TestType == FiniteElement::DIV)
|
||||
{
|
||||
if constexpr (DIM == 2)
|
||||
{
|
||||
return HdivDLFAssemble2D<TEST_D1D, Q1D>;
|
||||
}
|
||||
if constexpr (DIM == 3)
|
||||
{
|
||||
return HdivDLFAssemble3D<TEST_D1D, Q1D>;
|
||||
}
|
||||
}
|
||||
if constexpr (TestType == FiniteElement::CURL)
|
||||
{
|
||||
if constexpr (DIM == 3)
|
||||
{
|
||||
return HcurlDLFAssemble3D<TEST_D1D, Q1D>;
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -13,317 +13,76 @@
|
||||
#include "../../fem/kernels.hpp"
|
||||
#include "../fem.hpp"
|
||||
|
||||
#include "lininteg_domain_kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void HdivDLFAssemble2D(
|
||||
const int ne, const int d, const int q, const int *markers, const real_t *bo,
|
||||
const real_t *bc, const real_t *j, const real_t *weights,
|
||||
const Vector &coeff, real_t *y)
|
||||
VectorFEDomainLFIntegrator::Kernels::Kernels()
|
||||
{
|
||||
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
|
||||
"Problem size too large.");
|
||||
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
|
||||
"Problem size too large.");
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 1, 1>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 2, 2>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 3, 3>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 4, 4>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 5, 5>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 6, 6>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 7, 7>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 8, 8>();
|
||||
|
||||
static constexpr int vdim = 2;
|
||||
const auto F = coeff.Read();
|
||||
const auto M = Reshape(markers, ne);
|
||||
const auto BO = Reshape(bo, q, d-1);
|
||||
const auto BC = Reshape(bc, q, d);
|
||||
const auto J = Reshape(j, q, q, vdim, vdim, ne);
|
||||
const auto W = Reshape(weights, q, q);
|
||||
const bool cst = coeff.Size() == vdim;
|
||||
const auto C = cst ? Reshape(F,vdim,1,1,1) : Reshape(F,vdim,q,q,ne);
|
||||
auto Y = Reshape(y, 2*(d-1)*d, ne);
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 1, 1>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 2, 2>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 3, 3>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 4, 4>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 5, 5>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 6, 6>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 7, 7>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 8, 8>();
|
||||
|
||||
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
if (M(e) == 0) { return; } // ignore
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 1, 1>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 2, 2>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 3, 3>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 4, 4>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 5, 5>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 6, 6>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 7, 7>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 8, 8>();
|
||||
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HDIV_MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HDIV_MAX_D1D;
|
||||
|
||||
MFEM_SHARED real_t sBot[Q*D];
|
||||
MFEM_SHARED real_t sBct[Q*D];
|
||||
MFEM_SHARED real_t sQQ[vdim*Q*Q];
|
||||
MFEM_SHARED real_t sQD[vdim*Q*D];
|
||||
|
||||
// Bo and Bc into shared memory
|
||||
const DeviceMatrix Bot(sBot, d-1, q);
|
||||
kernels::internal::LoadB<D,Q>(d-1, q, BO, sBot);
|
||||
const DeviceMatrix Bct(sBct, d, q);
|
||||
kernels::internal::LoadB<D,Q>(d, q, BC, sBct);
|
||||
|
||||
const DeviceCube QQ(sQQ, q, q, vdim);
|
||||
const DeviceCube QD(sQD, q, d, vdim);
|
||||
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const real_t cst_val_0 = C(0,0,0,0);
|
||||
const real_t cst_val_1 = C(1,0,0,0);
|
||||
MFEM_FOREACH_THREAD(y,y,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(x,x,q)
|
||||
{
|
||||
const real_t J0 = J(x,y,0,vd,e);
|
||||
const real_t J1 = J(x,y,1,vd,e);
|
||||
const real_t C0 = cst ? cst_val_0 : C(0,x,y,e);
|
||||
const real_t C1 = cst ? cst_val_1 : C(1,x,y,e);
|
||||
QQ(x,y,vd) = W(x,y)*(J0*C0 + J1*C1);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d : d-1;
|
||||
DeviceMatrix Btx = (vd == 0) ? Bct : Bot;
|
||||
MFEM_FOREACH_THREAD(qy,y,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,nx)
|
||||
{
|
||||
real_t qd = 0.0;
|
||||
for (int qx = 0; qx < q; ++qx)
|
||||
{
|
||||
qd += QQ(qx,qy,vd) * Btx(dx,qx);
|
||||
}
|
||||
QD(dx,qy,vd) = qd;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d : d-1;
|
||||
const int ny = (vd == 1) ? d : d-1;
|
||||
DeviceMatrix Bty = (vd == 1) ? Bct : Bot;
|
||||
DeviceTensor<4> Yxy(Y, nx, ny, vdim, ne);
|
||||
MFEM_FOREACH_THREAD(dy,y,ny)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,nx)
|
||||
{
|
||||
real_t dd = 0.0;
|
||||
for (int qy = 0; qy < q; ++qy)
|
||||
{
|
||||
dd += QD(dx,qy,vd) * Bty(dy,qy);
|
||||
}
|
||||
Yxy(dx,dy,vd,e) += dd;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
});
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 1, 2>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 2, 3>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 3, 4>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 4, 5>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 5, 6>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 6, 7>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 7, 8>();
|
||||
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 8, 9>();
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void HdivDLFAssemble3D(
|
||||
const int ne, const int d, const int q, const int *markers, const real_t *bo,
|
||||
const real_t *bc, const real_t *j, const real_t *weights,
|
||||
const Vector &coeff, real_t *y)
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
VectorFEDomainLFIntegrator::AssembleKernelType
|
||||
VectorFEDomainLFIntegrator::AssembleKernels::Fallback(
|
||||
FiniteElement::DerivType TestType, int DIM, int, int)
|
||||
{
|
||||
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
|
||||
"Problem size too large.");
|
||||
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
|
||||
"Problem size too large.");
|
||||
|
||||
static constexpr int vdim = 3;
|
||||
const auto F = coeff.Read();
|
||||
const auto M = Reshape(markers, ne);
|
||||
const auto BO = Reshape(bo, q, d-1);
|
||||
const auto BC = Reshape(bc, q, d);
|
||||
const auto J = Reshape(j, q, q, q, vdim, vdim, ne);
|
||||
const auto W = Reshape(weights, q, q, q);
|
||||
const bool cst = coeff.Size() == vdim;
|
||||
const auto C = cst ? Reshape(F,vdim,1,1,1,1) : Reshape(F,vdim,q,q,q,ne);
|
||||
auto Y = Reshape(y, 2*(d-1)*(d-1)*d, ne);
|
||||
|
||||
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE (int e)
|
||||
if (TestType == FiniteElement::DIV)
|
||||
{
|
||||
if (M(e) == 0) { return; } // ignore
|
||||
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HDIV_MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HDIV_MAX_D1D;
|
||||
|
||||
MFEM_SHARED real_t sBot[Q*D];
|
||||
MFEM_SHARED real_t sBct[Q*D];
|
||||
|
||||
// Bo and Bc into shared memory
|
||||
const DeviceMatrix Bot(sBot, d-1, q);
|
||||
kernels::internal::LoadB<D,Q>(d-1, q, BO, sBot);
|
||||
const DeviceMatrix Bct(sBct, d, q);
|
||||
kernels::internal::LoadB<D,Q>(d, q, BC, sBct);
|
||||
|
||||
MFEM_SHARED real_t sm0[vdim*Q*Q*Q];
|
||||
MFEM_SHARED real_t sm1[vdim*Q*Q*Q];
|
||||
DeviceTensor<4> QQQ(sm1, q, q, q, vdim);
|
||||
DeviceTensor<4> DQQ(sm0, d, q, q, vdim);
|
||||
DeviceTensor<4> DDQ(sm1, d, d, q, vdim);
|
||||
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
if (DIM == 2)
|
||||
{
|
||||
const real_t cst_val_0 = C(0,0,0,0,0);
|
||||
const real_t cst_val_1 = C(1,0,0,0,0);
|
||||
const real_t cst_val_2 = C(2,0,0,0,0);
|
||||
MFEM_FOREACH_THREAD(y,y,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(x,x,q)
|
||||
{
|
||||
for (int z = 0; z < q; ++z)
|
||||
{
|
||||
const real_t J0 = J(x,y,z,0,vd,e);
|
||||
const real_t J1 = J(x,y,z,1,vd,e);
|
||||
const real_t J2 = J(x,y,z,2,vd,e);
|
||||
const real_t C0 = cst ? cst_val_0 : C(0,x,y,z,e);
|
||||
const real_t C1 = cst ? cst_val_1 : C(1,x,y,z,e);
|
||||
const real_t C2 = cst ? cst_val_2 : C(2,x,y,z,e);
|
||||
QQQ(x,y,z,vd) = W(x,y,z)*(J0*C0 + J1*C1 + J2*C2);
|
||||
}
|
||||
}
|
||||
}
|
||||
return HdivDLFAssemble2D<0, 0>;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// Apply Bt operator
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
if (DIM == 3)
|
||||
{
|
||||
const int nx = (vd == 0) ? d : d-1;
|
||||
DeviceMatrix Btx = (vd == 0) ? Bct : Bot;
|
||||
MFEM_FOREACH_THREAD(qy,y,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,nx)
|
||||
{
|
||||
real_t u[Q];
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz) { u[qz] = 0.0; }
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qx = 0; qx < q; ++qx)
|
||||
{
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
u[qz] += QQQ(qx,qy,qz,vd) * Btx(dx,qx);
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz) { DQQ(dx,qy,qz,vd) = u[qz]; }
|
||||
}
|
||||
}
|
||||
return HdivDLFAssemble3D<0, 0>;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d : d-1;
|
||||
const int ny = (vd == 1) ? d : d-1;
|
||||
DeviceMatrix Bty = (vd == 1) ? Bct : Bot;
|
||||
MFEM_FOREACH_THREAD(dy,y,ny)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,nx)
|
||||
{
|
||||
real_t u[Q];
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz) { u[qz] = 0.0; }
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qy = 0; qy < q; ++qy)
|
||||
{
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
u[qz] += DQQ(dx,qy,qz,vd) * Bty(dy,qy);
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz) { DDQ(dx,dy,qz,vd) = u[qz]; }
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(vd,z,vdim)
|
||||
{
|
||||
const int nx = (vd == 0) ? d : d-1;
|
||||
const int ny = (vd == 1) ? d : d-1;
|
||||
const int nz = (vd == 2) ? d : d-1;
|
||||
DeviceTensor<5> Yxyz(Y, nx, ny, nz, vdim, ne);
|
||||
DeviceMatrix Btz = (vd == 2) ? Bct : Bot;
|
||||
MFEM_FOREACH_THREAD(dy,y,ny)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,nx)
|
||||
{
|
||||
real_t u[D];
|
||||
MFEM_UNROLL(D)
|
||||
for (int dz = 0; dz < nz; ++dz) { u[dz] = 0.0; }
|
||||
MFEM_UNROLL(Q)
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
MFEM_UNROLL(D)
|
||||
for (int dz = 0; dz < nz; ++dz)
|
||||
{
|
||||
u[dz] += DDQ(dx,dy,qz,vd) * Btz(dz,qz);
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(D)
|
||||
for (int dz = 0; dz < nz; ++dz) { Yxyz(dx,dy,dz,vd,e) += u[dz]; }
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
});
|
||||
}
|
||||
|
||||
static void HdivDLFAssemble(const FiniteElementSpace &fes,
|
||||
const IntegrationRule *ir,
|
||||
const Array<int> &markers,
|
||||
const Vector &coeff,
|
||||
Vector &y)
|
||||
{
|
||||
Mesh &mesh = *fes.GetMesh();
|
||||
const int dim = mesh.Dimension();
|
||||
const FiniteElement *el = fes.GetTypicalFE();
|
||||
const auto *vel = dynamic_cast<const VectorTensorFiniteElement *>(el);
|
||||
MFEM_VERIFY(vel != nullptr, "Must be VectorTensorFiniteElement");
|
||||
const MemoryType mt = Device::GetDeviceMemoryType();
|
||||
const DofToQuad &maps_o = vel->GetDofToQuadOpen(*ir, DofToQuad::TENSOR);
|
||||
const DofToQuad &maps_c = vel->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int d = maps_c.ndof, q = maps_c.nqpt;
|
||||
constexpr int flags = GeometricFactors::JACOBIANS;
|
||||
const GeometricFactors *geom = mesh.GetGeometricFactors(*ir, flags, mt);
|
||||
decltype(&HdivDLFAssemble2D<>) ker =
|
||||
dim == 2 ? HdivDLFAssemble2D<> : HdivDLFAssemble3D<>;
|
||||
|
||||
if (dim==2)
|
||||
{
|
||||
if (d==1 && q==1) { ker=HdivDLFAssemble2D<1,1>; }
|
||||
if (d==2 && q==2) { ker=HdivDLFAssemble2D<2,2>; }
|
||||
if (d==3 && q==3) { ker=HdivDLFAssemble2D<3,3>; }
|
||||
if (d==4 && q==4) { ker=HdivDLFAssemble2D<4,4>; }
|
||||
if (d==5 && q==5) { ker=HdivDLFAssemble2D<5,5>; }
|
||||
if (d==6 && q==6) { ker=HdivDLFAssemble2D<6,6>; }
|
||||
if (d==7 && q==7) { ker=HdivDLFAssemble2D<7,7>; }
|
||||
if (d==8 && q==8) { ker=HdivDLFAssemble2D<8,8>; }
|
||||
}
|
||||
|
||||
if (dim==3)
|
||||
else if (TestType == FiniteElement::CURL)
|
||||
{
|
||||
if (d==2 && q==2) { ker=HdivDLFAssemble3D<2,2>; }
|
||||
if (d==3 && q==3) { ker=HdivDLFAssemble3D<3,3>; }
|
||||
if (d==4 && q==4) { ker=HdivDLFAssemble3D<4,4>; }
|
||||
if (d==5 && q==5) { ker=HdivDLFAssemble3D<5,5>; }
|
||||
if (d==6 && q==6) { ker=HdivDLFAssemble3D<6,6>; }
|
||||
if (d==7 && q==7) { ker=HdivDLFAssemble3D<7,7>; }
|
||||
if (d==8 && q==8) { ker=HdivDLFAssemble3D<8,8>; }
|
||||
if (DIM == 3)
|
||||
{
|
||||
return HcurlDLFAssemble3D<0, 0>;
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_VERIFY(ker, "No kernel ndof " << d << " nqpt " << q);
|
||||
|
||||
const int ne = mesh.GetNE();
|
||||
const int *M = markers.Read();
|
||||
const real_t *Bo = maps_o.B.Read();
|
||||
const real_t *Bc = maps_c.B.Read();
|
||||
const real_t *J = geom->J.Read();
|
||||
const real_t *W = ir->GetWeights().Read();
|
||||
real_t *Y = y.ReadWrite();
|
||||
ker(ne, d, q, M, Bo, Bc, J, W, coeff, Y);
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
void VectorFEDomainLFIntegrator::AssembleDevice(const FiniteElementSpace &fes,
|
||||
const Array<int> &markers,
|
||||
@@ -337,15 +96,23 @@ void VectorFEDomainLFIntegrator::AssembleDevice(const FiniteElementSpace &fes,
|
||||
QuadratureSpace qs(*fes.GetMesh(), *ir);
|
||||
CoefficientVector coeff(QF, qs, CoefficientStorage::COMPRESSED);
|
||||
|
||||
const int fe_type = fe.GetDerivType();
|
||||
if (fe_type == FiniteElement::DIV)
|
||||
{
|
||||
HdivDLFAssemble(fes, ir, markers, coeff, b);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Not implemented.");
|
||||
}
|
||||
const FiniteElement::DerivType fe_type =
|
||||
static_cast<FiniteElement::DerivType>(fe.GetDerivType());
|
||||
|
||||
Mesh &mesh = *fes.GetMesh();
|
||||
const int dim = mesh.Dimension();
|
||||
const FiniteElement *el = fes.GetTypicalFE();
|
||||
const auto *vel = dynamic_cast<const VectorTensorFiniteElement *>(el);
|
||||
MFEM_VERIFY(vel != nullptr, "Must be VectorTensorFiniteElement");
|
||||
const MemoryType mt = Device::GetDeviceMemoryType();
|
||||
const DofToQuad &maps_o = vel->GetDofToQuadOpen(*ir, DofToQuad::TENSOR);
|
||||
const DofToQuad &maps_c = vel->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int d = maps_c.ndof, q = maps_c.nqpt;
|
||||
constexpr int flags = GeometricFactors::JACOBIANS;
|
||||
const GeometricFactors *geom = mesh.GetGeometricFactors(*ir, flags, mt);
|
||||
|
||||
AssembleKernels::Run(fe_type, dim, d, q, mesh.GetNE(), markers, geom->J,
|
||||
ir->GetWeights(), maps_o.B, maps_c.B, coeff, b, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -471,6 +471,13 @@ void VectorBoundaryLFIntegrator::AssembleRHSElementVect(
|
||||
}
|
||||
}
|
||||
|
||||
VectorFEDomainLFIntegrator::VectorFEDomainLFIntegrator(
|
||||
VectorCoefficient &F, const IntegrationRule *ir)
|
||||
: DeltaLFIntegrator(F, ir), QF(F)
|
||||
{
|
||||
static Kernels kernels{};
|
||||
}
|
||||
|
||||
void VectorFEDomainLFIntegrator::AssembleRHSElementVect(
|
||||
const FiniteElement &el, ElementTransformation &Tr, Vector &elvect)
|
||||
{
|
||||
|
||||
+36
-2
@@ -369,8 +369,8 @@ private:
|
||||
Vector vec;
|
||||
|
||||
public:
|
||||
VectorFEDomainLFIntegrator(VectorCoefficient &F)
|
||||
: DeltaLFIntegrator(F), QF(F) { }
|
||||
VectorFEDomainLFIntegrator(VectorCoefficient &F,
|
||||
const IntegrationRule *ir = nullptr);
|
||||
|
||||
void AssembleRHSElementVect(const FiniteElement &el,
|
||||
ElementTransformation &Tr,
|
||||
@@ -387,6 +387,40 @@ public:
|
||||
Vector &b) override;
|
||||
|
||||
using LinearFormIntegrator::AssembleRHSElementVect;
|
||||
|
||||
/// @param ne number of elements
|
||||
/// @param markers array where entry markers[e] == 0 to skip assembly over
|
||||
/// element e element
|
||||
/// @param jac Spatial Jacobians evaluated at all quadrature points
|
||||
/// @param weights 1D quadrature weights
|
||||
/// @param testBO 1D open basis test functions
|
||||
/// @param testBC 1D closed basis test functions
|
||||
/// @param coeff coefficient values evaluated at quadrature points, possibly
|
||||
/// compressed.
|
||||
/// @param d number of 1D closed dofs
|
||||
/// @param q number of 1D quadrature points
|
||||
using AssembleKernelType = void (*)(const int NE, const Array<int> &markers,
|
||||
const Vector &jac,
|
||||
const Array<real_t> &weights,
|
||||
const Array<real_t> &testBO,
|
||||
const Array<real_t> &testBC,
|
||||
const Vector &coeff, Vector &y,
|
||||
const int testd1d, const int q1d);
|
||||
|
||||
/// parameters: test_fetype, ndims, test_d1d, q1d
|
||||
MFEM_REGISTER_KERNELS(AssembleKernels, AssembleKernelType,
|
||||
(FiniteElement::DerivType, int, int, int));
|
||||
|
||||
struct Kernels
|
||||
{
|
||||
Kernels();
|
||||
};
|
||||
|
||||
template <FiniteElement::DerivType TestType, int DIM, int TEST_D1D, int Q1D>
|
||||
static void AddSpecialization()
|
||||
{
|
||||
AssembleKernels::Specialization<TestType, DIM, TEST_D1D, Q1D>::Add();
|
||||
}
|
||||
};
|
||||
|
||||
/// $ (Q, \mathrm{curl}(v))_{\Omega} $ for Nedelec Elements
|
||||
|
||||
+354
-63
@@ -10,6 +10,7 @@
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "particleset.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
|
||||
#if defined(MFEM_USE_MPI) && defined(MFEM_USE_GSLIB)
|
||||
|
||||
@@ -225,6 +226,7 @@ void ParticleSet::AddParticles(const Array<IDType> &new_ids,
|
||||
}
|
||||
}
|
||||
// Add new ids
|
||||
ids.HostReadWrite();
|
||||
ids.Append(new_ids);
|
||||
|
||||
// Update data
|
||||
@@ -244,6 +246,102 @@ void ParticleSet::AddParticles(const Array<IDType> &new_ids,
|
||||
#if defined(MFEM_USE_MPI) && defined(MFEM_USE_GSLIB)
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
// Static helper: gather selected particle-vector entries into a compact buffer.
|
||||
// nvcc does not allow extended host/device lambdas in non-public members.
|
||||
static void GatherParticleVectorDevice(const ParticleVector &pv,
|
||||
const Array<int> &send_idxs,
|
||||
Vector &send_data,
|
||||
int nsend)
|
||||
{
|
||||
const int vdim = pv.GetVDim();
|
||||
const int ordering = pv.GetOrdering();
|
||||
const int num_particles = pv.GetNumParticles();
|
||||
const MemoryClass device_mc = Device::GetDeviceMemoryClass();
|
||||
send_data.SetSize(nsend*vdim);
|
||||
real_t *d_send_data =
|
||||
send_data.GetMemory().Write(device_mc, send_data.Size());
|
||||
const real_t *d_src = pv.GetMemory().Read(device_mc, pv.Size());
|
||||
const int *d_send_idxs = send_idxs.GetMemory().Read(device_mc, nsend);
|
||||
|
||||
mfem::forall(nsend, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
const int p = d_send_idxs[i];
|
||||
const int offset = (ordering == Ordering::byVDIM) ? p * vdim : p;
|
||||
const int stride = (ordering == Ordering::byVDIM) ? 1 : num_particles;
|
||||
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
d_send_data[i*vdim + c] = d_src[offset + c*stride];
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Static helper: gather selected tag values into a compact buffer.
|
||||
// nvcc does not allow extended host/device lambdas in non-public members.
|
||||
static void GatherParticleTagsDevice(const Array<int> &tag,
|
||||
const Array<int> &send_idxs,
|
||||
Array<int> &send_tag,
|
||||
int nsend)
|
||||
{
|
||||
const MemoryClass device_mc = Device::GetDeviceMemoryClass();
|
||||
send_tag.SetSize(nsend);
|
||||
int *d_send_tag = send_tag.GetMemory().Write(device_mc, nsend);
|
||||
const int *d_tag = tag.GetMemory().Read(device_mc, tag.Size());
|
||||
const int *d_send_idxs = send_idxs.GetMemory().Read(device_mc, nsend);
|
||||
|
||||
mfem::forall(nsend, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_send_tag[i] = d_tag[d_send_idxs[i]];
|
||||
});
|
||||
}
|
||||
|
||||
// Static helper: scatter compact particle-vector entries to particle storage.
|
||||
// nvcc does not allow extended host/device lambdas in non-public members.
|
||||
static void ScatterParticleVectorDevice(ParticleVector &pv,
|
||||
const Vector &recv_data,
|
||||
const Array<int> &recv_locs,
|
||||
int nrecv)
|
||||
{
|
||||
const int vdim = pv.GetVDim();
|
||||
const int ordering = pv.GetOrdering();
|
||||
const int num_particles = pv.GetNumParticles();
|
||||
const MemoryClass device_mc = Device::GetDeviceMemoryClass();
|
||||
const real_t *d_recv_data =
|
||||
recv_data.GetMemory().Read(device_mc, recv_data.Size());
|
||||
const int *d_recv_locs = recv_locs.GetMemory().Read(device_mc, nrecv);
|
||||
real_t *d_dst = pv.GetMemory().ReadWrite(device_mc, pv.Size());
|
||||
|
||||
mfem::forall(nrecv, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
const int p = d_recv_locs[i];
|
||||
const int offset = (ordering == Ordering::byVDIM) ? p * vdim : p;
|
||||
const int stride = (ordering == Ordering::byVDIM) ? 1 : num_particles;
|
||||
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
d_dst[offset + c*stride] = d_recv_data[i*vdim + c];
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Static helper: scatter compact tag values to particle storage.
|
||||
// nvcc does not allow extended host/device lambdas in non-public members.
|
||||
static void ScatterParticleTagsDevice(Array<int> &tag,
|
||||
const Array<int> &recv_tag,
|
||||
const Array<int> &recv_locs,
|
||||
int nrecv)
|
||||
{
|
||||
const MemoryClass device_mc = Device::GetDeviceMemoryClass();
|
||||
const int *d_recv_tag = recv_tag.GetMemory().Read(device_mc, nrecv);
|
||||
const int *d_recv_locs = recv_locs.GetMemory().Read(device_mc, nrecv);
|
||||
int *d_tag = tag.GetMemory().ReadWrite(device_mc, tag.Size());
|
||||
|
||||
mfem::forall(nrecv, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_tag[d_recv_locs[i]] = d_recv_tag[i];
|
||||
});
|
||||
}
|
||||
|
||||
template<size_t NBytes>
|
||||
void ParticleSet::TransferParticlesImpl(ParticleSet &pset,
|
||||
const Array<int> &send_idxs,
|
||||
@@ -266,37 +364,108 @@ void ParticleSet::TransferParticlesImpl(ParticleSet &pset,
|
||||
array_init(parr_t, &gsl_arr, send_idxs.Size());
|
||||
pdata_arr = (parr_t*) gsl_arr.ptr;
|
||||
|
||||
int nparticles = pset.GetNParticles();
|
||||
int nsend = send_idxs.Size();
|
||||
gsl_arr.n = send_idxs.Size();
|
||||
|
||||
const int *h_send_idxs_initial = send_idxs.HostRead();
|
||||
const IDType *h_ids = pset.GetIDs().HostRead();
|
||||
for (int i = 0; i < send_idxs.Size(); i++)
|
||||
{
|
||||
parr_t &pdata = pdata_arr[i];
|
||||
pdata.id = pset.GetIDs()[send_idxs[i]];
|
||||
pdata.id = h_ids[h_send_idxs_initial[i]];
|
||||
}
|
||||
|
||||
// Copy particle data directly into pdata
|
||||
size_t counter = 0;
|
||||
for (int f = -1; f < pset.GetNFields(); f++)
|
||||
// Pack coords and fields into the GSLIB send buffer. Device-resident data
|
||||
// is first gathered into a compact device buffer so that only selected
|
||||
// particles are copied back to host. Host-resident data is packed directly.
|
||||
int max_vdim = pset.Coords().GetVDim();
|
||||
for (int f = 0; f < pset.GetNFields(); f++)
|
||||
{
|
||||
int f_vdim = pset.Field(f).GetVDim();
|
||||
if (f_vdim > max_vdim) { max_vdim = f_vdim; }
|
||||
}
|
||||
Vector send_data;
|
||||
Array<int> send_tag;
|
||||
if (Device::IsEnabled())
|
||||
{
|
||||
send_data.SetSize(nsend * max_vdim); // allocate max size over all fields
|
||||
send_tag.SetSize(nsend);
|
||||
}
|
||||
|
||||
size_t counter = 0;
|
||||
for (int f = -1; f < pset.GetNFields(); f++)
|
||||
{
|
||||
const ParticleVector &pv = f == -1 ? pset.Coords() : pset.Field(f);
|
||||
const int vdim = pv.GetVDim();
|
||||
const int ordering = pv.GetOrdering();
|
||||
const int num_particles = pv.GetNumParticles();
|
||||
const bool use_dev = Device::IsEnabled() && pv.UseDevice();
|
||||
|
||||
if (use_dev)
|
||||
{
|
||||
ParticleVector &pv = (f == -1 ? pset.Coords() : pset.Field(f));
|
||||
for (int c = 0; c < pv.GetVDim(); c++)
|
||||
GatherParticleVectorDevice(pv, send_idxs, send_data, nsend);
|
||||
|
||||
const real_t *h_send_data = send_data.HostRead();
|
||||
for (int i = 0; i < nsend; i++)
|
||||
{
|
||||
std::memcpy(pdata.data.data() + counter, &pv(send_idxs[i], c),
|
||||
sizeof(real_t));
|
||||
counter += sizeof(real_t);
|
||||
std::memcpy(pdata_arr[i].data.data() + counter,
|
||||
h_send_data + i*vdim, vdim * sizeof(real_t));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
const real_t *h_src = pv.HostRead();
|
||||
const int *h_send_idxs = send_idxs.HostRead();
|
||||
for (int i = 0; i < nsend; i++)
|
||||
{
|
||||
parr_t &pdata = pdata_arr[i];
|
||||
const int p = h_send_idxs[i];
|
||||
const int offset = (ordering == Ordering::byVDIM) ? p * vdim : p;
|
||||
const int stride = (ordering == Ordering::byVDIM) ? 1 :
|
||||
num_particles;
|
||||
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
std::memcpy(pdata.data.data() + counter + c*sizeof(real_t),
|
||||
h_src + offset + c*stride, sizeof(real_t));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Copy tags
|
||||
for (int t = 0; t < pset.GetNTags(); t++)
|
||||
{
|
||||
Array<int> &tag_arr = pset.Tag(t);
|
||||
std::memcpy(pdata.data.data() + counter, &tag_arr[send_idxs[i]],
|
||||
sizeof(int));
|
||||
counter += sizeof(int);
|
||||
}
|
||||
counter += vdim*sizeof(real_t);
|
||||
}
|
||||
|
||||
int nparticles = pset.GetNParticles();
|
||||
int nsend = send_idxs.Size();
|
||||
// Pack tags after all real_t data. Each tag uses the same selective
|
||||
// device gather path when its Array is device-resident.
|
||||
for (int t = 0; t < pset.GetNTags(); t++)
|
||||
{
|
||||
const Array<int> &tag = pset.Tag(t);
|
||||
const size_t tag_counter = counter + t*sizeof(int);
|
||||
const bool use_dev = Device::IsEnabled() && tag.UseDevice();
|
||||
|
||||
if (use_dev)
|
||||
{
|
||||
GatherParticleTagsDevice(tag, send_idxs, send_tag, nsend);
|
||||
|
||||
const int *h_send_tag = send_tag.HostRead();
|
||||
for (int i = 0; i < nsend; i++)
|
||||
{
|
||||
std::memcpy(pdata_arr[i].data.data() + tag_counter,
|
||||
h_send_tag + i, sizeof(int));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
const int *h_tag = tag.HostRead();
|
||||
const int *h_send_idxs = send_idxs.HostRead();
|
||||
for (int i = 0; i < nsend; i++)
|
||||
{
|
||||
std::memcpy(pdata_arr[i].data.data() + tag_counter,
|
||||
h_tag + h_send_idxs[i], sizeof(int));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Transfer particles
|
||||
sarray_transfer_ext(parr_t, &gsl_arr, send_ranks.GetData(),
|
||||
@@ -304,11 +473,20 @@ void ParticleSet::TransferParticlesImpl(ParticleSet &pset,
|
||||
|
||||
// Make sure we have enough space for received particles
|
||||
int nrecv = (int) gsl_arr.n;
|
||||
|
||||
Vector recv_data;
|
||||
Array<int> recv_tag;
|
||||
if (Device::IsEnabled())
|
||||
{
|
||||
recv_data.SetSize(nrecv * max_vdim);
|
||||
recv_tag.SetSize(nrecv);
|
||||
}
|
||||
|
||||
int ndelete = nsend - nrecv;
|
||||
if (ndelete > 0)
|
||||
{
|
||||
// Remove unneeded particles
|
||||
auto datap = const_cast<int*>(send_idxs.GetData());
|
||||
auto datap = const_cast<int*>(send_idxs.HostRead());
|
||||
Array<int> delete_idxs(datap + nrecv, ndelete);
|
||||
pset.RemoveParticles(delete_idxs);
|
||||
}
|
||||
@@ -319,47 +497,133 @@ void ParticleSet::TransferParticlesImpl(ParticleSet &pset,
|
||||
|
||||
pdata_arr = (parr_t*) gsl_arr.ptr;
|
||||
|
||||
// Add newly-recvd data directly to active state
|
||||
// Make a list of new IDs to add
|
||||
int num_new = nrecv > nsend ? nrecv - nsend : 0;
|
||||
Array<IDType> new_ids(num_new);
|
||||
for (int i = 0; i < num_new; i++)
|
||||
{
|
||||
new_ids[i] = pdata_arr[nsend + i].id;
|
||||
}
|
||||
|
||||
// Add particles in batch
|
||||
Array<int> new_indices;
|
||||
if (num_new > 0)
|
||||
{
|
||||
pset.AddParticles(new_ids, &new_indices);
|
||||
}
|
||||
|
||||
// Map each received packet to the local particle slot it updates.
|
||||
Array<int> recv_locs(nrecv);
|
||||
int *h_recv_locs = recv_locs.HostWrite();
|
||||
const int *h_send_idxs_recv = send_idxs.HostRead();
|
||||
for (int i = 0; i < nrecv; i++)
|
||||
{
|
||||
parr_t &pdata = pdata_arr[i];
|
||||
IDType id = pdata.id;
|
||||
|
||||
int new_loc_idx;
|
||||
if (i < nsend) // update existing particle
|
||||
{
|
||||
new_loc_idx = send_idxs[i];
|
||||
pset.UpdateID(new_loc_idx, id);
|
||||
h_recv_locs[i] = h_send_idxs_recv[i];
|
||||
pset.UpdateID(h_recv_locs[i], pdata.id);
|
||||
}
|
||||
else
|
||||
{
|
||||
// add new particle
|
||||
Array<int> idx_temp;
|
||||
pset.AddParticles(Array<IDType>({id}), &idx_temp);
|
||||
new_loc_idx = idx_temp[0]; // Get index of newly-added particle
|
||||
h_recv_locs[i] = new_indices[i - nsend];
|
||||
}
|
||||
}
|
||||
|
||||
size_t counter = 0;
|
||||
for (int f = -1; f < pset.GetNFields(); f++)
|
||||
// Unpack coords and fields from GSLIB host packets. Device-resident
|
||||
// destinations use a compact host buffer followed by a device scatter.
|
||||
size_t recv_counter = 0;
|
||||
for (int f = -1; f < pset.GetNFields(); f++)
|
||||
{
|
||||
ParticleVector &pv = (f == -1 ? pset.Coords() : pset.Field(f));
|
||||
const int vdim = pv.GetVDim();
|
||||
const int ordering = pv.GetOrdering();
|
||||
const int num_particles = pv.GetNumParticles();
|
||||
const bool use_dev = Device::IsEnabled() && pv.UseDevice();
|
||||
|
||||
if (use_dev)
|
||||
{
|
||||
ParticleVector &pv = (f == -1 ? pset.Coords() : pset.Field(f));
|
||||
for (int c = 0; c < pv.GetVDim(); c++)
|
||||
recv_data.SetSize(nrecv*vdim);
|
||||
real_t *h_recv_data = recv_data.HostWrite();
|
||||
|
||||
for (int i = 0; i < nrecv; i++)
|
||||
{
|
||||
real_t& val = pv(new_loc_idx, c);
|
||||
std::memcpy(&val, pdata.data.data() + counter, sizeof(real_t));
|
||||
counter += sizeof(real_t);
|
||||
std::memcpy(h_recv_data + i*vdim,
|
||||
pdata_arr[i].data.data() + recv_counter,
|
||||
vdim*sizeof(real_t));
|
||||
}
|
||||
|
||||
ScatterParticleVectorDevice(pv, recv_data, recv_locs, nrecv);
|
||||
}
|
||||
else
|
||||
{
|
||||
real_t *h_dst = pv.HostReadWrite();
|
||||
const int *h_recv_locs_read = recv_locs.HostRead();
|
||||
for (int i = 0; i < nrecv; i++)
|
||||
{
|
||||
parr_t &pdata = pdata_arr[i];
|
||||
const int p = h_recv_locs_read[i];
|
||||
const int offset = (ordering == Ordering::byVDIM) ? p * vdim : p;
|
||||
const int stride = (ordering == Ordering::byVDIM) ? 1 :
|
||||
num_particles;
|
||||
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
std::memcpy(h_dst + offset + c*stride,
|
||||
pdata.data.data() + recv_counter + c*sizeof(real_t),
|
||||
sizeof(real_t));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int t = 0; t < pset.GetNTags(); t++)
|
||||
recv_counter += vdim*sizeof(real_t);
|
||||
}
|
||||
|
||||
// Unpack tags after all real_t data, using the same compact scatter path
|
||||
// for device-resident tag arrays.
|
||||
for (int t = 0; t < pset.GetNTags(); t++)
|
||||
{
|
||||
Array<int> &tag = pset.Tag(t);
|
||||
const size_t tag_counter = recv_counter + t*sizeof(int);
|
||||
const bool use_dev = Device::IsEnabled() && tag.UseDevice();
|
||||
|
||||
if (use_dev)
|
||||
{
|
||||
Array<int> &tag_arr = pset.Tag(t);
|
||||
std::memcpy(&tag_arr[new_loc_idx],
|
||||
pdata.data.data() + counter, sizeof(int));
|
||||
counter += sizeof(int);
|
||||
recv_tag.SetSize(nrecv);
|
||||
int *h_recv_tag = recv_tag.HostWrite();
|
||||
|
||||
for (int i = 0; i < nrecv; i++)
|
||||
{
|
||||
std::memcpy(h_recv_tag + i,
|
||||
pdata_arr[i].data.data() + tag_counter, sizeof(int));
|
||||
}
|
||||
|
||||
ScatterParticleTagsDevice(tag, recv_tag, recv_locs, nrecv);
|
||||
}
|
||||
else
|
||||
{
|
||||
int *h_tag = tag.HostReadWrite();
|
||||
const int *h_recv_locs_read = recv_locs.HostRead();
|
||||
for (int i = 0; i < nrecv; i++)
|
||||
{
|
||||
std::memcpy(h_tag + h_recv_locs_read[i],
|
||||
pdata_arr[i].data.data() + tag_counter, sizeof(int));
|
||||
}
|
||||
}
|
||||
}
|
||||
array_free(&gsl_arr);
|
||||
|
||||
// Restore Device validity if needed
|
||||
for (int f = -1; f < pset.GetNFields(); f++)
|
||||
{
|
||||
ParticleVector &pv = (f == -1 ? pset.Coords() : pset.Field(f));
|
||||
pv.ReadWrite(pv.UseDevice());
|
||||
}
|
||||
for (int t = 0; t < pset.GetNTags(); t++)
|
||||
{
|
||||
Array<int> &tag_arr = pset.Tag(t);
|
||||
if (tag_arr.UseDevice()) { tag_arr.ReadWrite(true); }
|
||||
}
|
||||
}
|
||||
|
||||
template<size_t NBytes>
|
||||
@@ -526,11 +790,14 @@ ParticleSet::ParticleSet(int id_stride_, IDType id_counter_, int num_particles,
|
||||
int dim, Ordering::Type coords_ordering, const Array<int> &field_vdims,
|
||||
const Array<Ordering::Type> &field_orderings,
|
||||
const Array<const char*> &field_names_, int num_tags,
|
||||
const Array<const char*> &tag_names_)
|
||||
const Array<const char*> &tag_names_,
|
||||
bool use_device)
|
||||
: id_stride(id_stride_),
|
||||
id_counter(id_counter_),
|
||||
coords(dim, coords_ordering)
|
||||
{
|
||||
if (use_device) { coords.UseDevice(true); }
|
||||
|
||||
// Initialize fields
|
||||
for (int f = 0; f < field_vdims.Size(); f++)
|
||||
{
|
||||
@@ -580,21 +847,22 @@ bool ParticleSet::IsValidParticle(const Particle &p) const
|
||||
}
|
||||
|
||||
ParticleSet::ParticleSet(int num_particles, int dim,
|
||||
Ordering::Type coords_ordering)
|
||||
Ordering::Type coords_ordering,
|
||||
bool use_device)
|
||||
: ParticleSet(1, 0, num_particles, dim, coords_ordering, Array<int>(),
|
||||
Array<Ordering::Type>(), Array<const char*>(), 0,
|
||||
Array<const char*>())
|
||||
Array<const char*>(), use_device)
|
||||
{
|
||||
|
||||
}
|
||||
|
||||
ParticleSet::ParticleSet(int num_particles, int dim,
|
||||
const Array<int> &field_vdims, int num_tags,
|
||||
Ordering::Type all_ordering)
|
||||
Ordering::Type all_ordering, bool use_device)
|
||||
: ParticleSet(1, 0, num_particles, dim, all_ordering, field_vdims,
|
||||
GetOrderingArray(all_ordering, field_vdims.Size()),
|
||||
GetEmptyNameArray(field_vdims.Size()), num_tags,
|
||||
GetEmptyNameArray(num_tags))
|
||||
GetEmptyNameArray(num_tags), use_device)
|
||||
{
|
||||
}
|
||||
|
||||
@@ -602,11 +870,11 @@ ParticleSet::ParticleSet(int num_particles, int dim,
|
||||
const Array<int> &field_vdims, const Array<const
|
||||
char*> &field_names_, int num_tags,
|
||||
const Array<const char*> &tag_names_,
|
||||
Ordering::Type all_ordering)
|
||||
Ordering::Type all_ordering, bool use_device)
|
||||
: ParticleSet(1, 0, num_particles, dim, all_ordering, field_vdims,
|
||||
GetOrderingArray(all_ordering, field_vdims.Size()),
|
||||
field_names_, num_tags,
|
||||
tag_names_)
|
||||
tag_names_, use_device)
|
||||
{
|
||||
|
||||
}
|
||||
@@ -616,9 +884,9 @@ ParticleSet::ParticleSet(int num_particles, int dim,
|
||||
const Array<int> &field_vdims,
|
||||
const Array<Ordering::Type> &field_orderings,
|
||||
const Array<const char*> &field_names_, int num_tags,
|
||||
const Array<const char*> &tag_names_)
|
||||
const Array<const char*> &tag_names_, bool use_device)
|
||||
: ParticleSet(1, 0, num_particles, dim, coords_ordering, field_vdims,
|
||||
field_orderings, field_names_, num_tags, tag_names_)
|
||||
field_orderings, field_names_, num_tags, tag_names_, use_device)
|
||||
{
|
||||
|
||||
}
|
||||
@@ -627,21 +895,21 @@ ParticleSet::ParticleSet(int num_particles, int dim,
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
|
||||
Ordering::Type coords_ordering)
|
||||
Ordering::Type coords_ordering, bool use_device)
|
||||
: ParticleSet(comm_, rank_num_particles, dim, coords_ordering, Array<int>(),
|
||||
Array<Ordering::Type>(), Array<const char*>(), 0,
|
||||
Array<const char*>())
|
||||
Array<const char*>(), use_device)
|
||||
{
|
||||
|
||||
};
|
||||
|
||||
ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
|
||||
const Array<int> &field_vdims, int num_tags,
|
||||
Ordering::Type all_ordering)
|
||||
Ordering::Type all_ordering, bool use_device)
|
||||
: ParticleSet(comm_, rank_num_particles, dim, all_ordering, field_vdims,
|
||||
GetOrderingArray(all_ordering, field_vdims.Size()),
|
||||
GetEmptyNameArray(field_vdims.Size()), num_tags,
|
||||
GetEmptyNameArray(num_tags))
|
||||
GetEmptyNameArray(num_tags), use_device)
|
||||
{
|
||||
|
||||
}
|
||||
@@ -650,11 +918,11 @@ ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
|
||||
const Array<int> &field_vdims, const Array<const
|
||||
char*> &field_names_,
|
||||
int num_tags, const Array<const char*> &tag_names_,
|
||||
Ordering::Type all_ordering)
|
||||
Ordering::Type all_ordering, bool use_device)
|
||||
: ParticleSet(comm_, rank_num_particles, dim, all_ordering, field_vdims,
|
||||
GetOrderingArray(all_ordering, field_vdims.Size()),
|
||||
field_names_, num_tags,
|
||||
tag_names_)
|
||||
tag_names_, use_device)
|
||||
{
|
||||
|
||||
}
|
||||
@@ -664,7 +932,7 @@ ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
|
||||
const Array<int> &field_vdims,
|
||||
const Array<Ordering::Type> &field_orderings,
|
||||
const Array<const char*> &field_names_, int num_tags,
|
||||
const Array<const char*> &tag_names_)
|
||||
const Array<const char*> &tag_names_, bool use_device)
|
||||
: ParticleSet(GetSize(comm_), (IDType)GetRank(comm_),
|
||||
rank_num_particles,
|
||||
dim,
|
||||
@@ -673,7 +941,7 @@ ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
|
||||
field_orderings,
|
||||
field_names_,
|
||||
num_tags,
|
||||
tag_names_)
|
||||
tag_names_, use_device)
|
||||
{
|
||||
comm = comm_;
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
@@ -705,6 +973,7 @@ int ParticleSet::AddField(int vdim, Ordering::Type field_ordering,
|
||||
}
|
||||
fields.emplace_back(std::make_unique<ParticleVector>(vdim, field_ordering,
|
||||
GetNParticles()));
|
||||
if (coords.UseDevice()) { fields.back()->UseDevice(true); }
|
||||
field_names.emplace_back(field_name_str);
|
||||
|
||||
return GetNFields() - 1;
|
||||
@@ -718,6 +987,7 @@ int ParticleSet::AddTag(const char* tag_name)
|
||||
tag_name_str = GetDefaultTagName(tag_names.size());
|
||||
}
|
||||
tags.emplace_back(std::make_unique<Array<int>>(GetNParticles()));
|
||||
if (coords.UseDevice()) { tags.back()->GetMemory().UseDevice(true); }
|
||||
tag_names.emplace_back(tag_name_str);
|
||||
|
||||
return GetNTags() - 1;
|
||||
@@ -782,7 +1052,7 @@ Particle ParticleSet::GetParticle(int i) const
|
||||
|
||||
for (int t = 0; t < GetNTags(); t++)
|
||||
{
|
||||
p.Tag(t) = Tag(t)[i];
|
||||
p.Tag(t) = Tag(t).HostRead()[i];
|
||||
}
|
||||
|
||||
return p;
|
||||
@@ -790,13 +1060,21 @@ Particle ParticleSet::GetParticle(int i) const
|
||||
|
||||
bool ParticleSet::IsParticleRefValid() const
|
||||
{
|
||||
if (coords.GetOrdering() == Ordering::byNODES)
|
||||
if (coords.GetOrdering() == Ordering::byNODES || coords.UseDevice())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
for (int f = 0; f < GetNFields(); f++)
|
||||
{
|
||||
if (fields[f]->GetOrdering() == Ordering::byNODES)
|
||||
if (fields[f]->GetOrdering() == Ordering::byNODES ||
|
||||
fields[f]->UseDevice())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
for (int t = 0; t < GetNTags(); t++)
|
||||
{
|
||||
if (tags[t]->UseDevice())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -806,6 +1084,10 @@ bool ParticleSet::IsParticleRefValid() const
|
||||
|
||||
Particle ParticleSet::GetParticleRef(int i)
|
||||
{
|
||||
MFEM_ASSERT(IsParticleRefValid(),
|
||||
"GetParticleRef is only valid when coordinates and fields are "
|
||||
"ordered byVDIM and particle data is host-resident.");
|
||||
|
||||
Particle p = CreateParticle();
|
||||
|
||||
Coords().GetValuesRef(i, p.Coords());
|
||||
@@ -839,7 +1121,7 @@ void ParticleSet::SetParticle(int i, const Particle &p)
|
||||
|
||||
for (int t = 0; t < GetNTags(); t++)
|
||||
{
|
||||
Tag(t)[i] = p.Tag(t);
|
||||
Tag(t).HostReadWrite()[i] = p.Tag(t);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -900,6 +1182,15 @@ void ParticleSet::PrintCSV(const char *fname, const Array<int> &field_idxs,
|
||||
#ifdef MFEM_USE_MPI
|
||||
int rank = GetRank(comm);
|
||||
#endif // MFEM_USE_MPI
|
||||
// make sure we can read tag data on host. fields and coords will be read as
|
||||
// needed in the loop below, so we don't need to pre-read them here.
|
||||
for (int i = 0; i < GetNTags(); i++)
|
||||
{
|
||||
tags[i]->HostRead();
|
||||
}
|
||||
ids.HostRead();
|
||||
|
||||
// Write particle data
|
||||
for (int i = 0; i < GetNParticles(); i++)
|
||||
{
|
||||
ss_data << ids[i];
|
||||
|
||||
+49
-12
@@ -211,6 +211,12 @@ public:
|
||||
* byVDIM). The unique_ptrs to all the ParticleVectors are stored in the
|
||||
* std::vector \ref fields.
|
||||
*
|
||||
* @par Device Behavior:
|
||||
* When a ParticleSet is constructed with \p use_device=true, \ref coords and
|
||||
* all ParticleVector fields are marked to use device memory. Fields added
|
||||
* later through \ref AddField inherit the current device mode (through
|
||||
* \ref coords).
|
||||
*
|
||||
* @par Tags:
|
||||
* Tags represent integers associated with each particle. For a given tag,
|
||||
* all particle data are stored in a single Array<int>. The unique_ptrs to all
|
||||
@@ -369,7 +375,10 @@ protected:
|
||||
* ID of a particle.
|
||||
*/
|
||||
void UpdateID(int local_idx, IDType new_global_id)
|
||||
{ ids[local_idx] = new_global_id; }
|
||||
{
|
||||
ids.HostReadWrite();
|
||||
ids[local_idx] = new_global_id;
|
||||
}
|
||||
|
||||
/** @brief Create a Particle object with the same spatial dimension,
|
||||
* number of fields and field vdims, and number of tags as this ParticleSet.
|
||||
@@ -399,12 +408,14 @@ protected:
|
||||
* @param[in] field_names_ Array of field names.
|
||||
* @param[in] num_tags Number of tags to register.
|
||||
* @param[in] tag_names_ Array of tag names.
|
||||
* @param[in] use_device Use device memory for particle fields.
|
||||
*/
|
||||
ParticleSet(int id_stride_, IDType id_counter_, int num_particles, int dim,
|
||||
Ordering::Type coords_ordering, const Array<int> &field_vdims,
|
||||
const Array<Ordering::Type> &field_orderings,
|
||||
const Array<const char*> &field_names_, int num_tags,
|
||||
const Array<const char*> &tag_names_);
|
||||
const Array<const char*> &tag_names_,
|
||||
bool use_device);
|
||||
|
||||
public:
|
||||
|
||||
@@ -413,9 +424,12 @@ public:
|
||||
* @param[in] num_particles Number of particles to initialize.
|
||||
* @param[in] dim Particle spatial dimension.
|
||||
* @param[in] coords_ordering Ordering of coordinates.
|
||||
* @param[in] use_device (Optional) Use device memory for particle
|
||||
* fields.
|
||||
*/
|
||||
ParticleSet(int num_particles, int dim,
|
||||
Ordering::Type coords_ordering=Ordering::byVDIM);
|
||||
Ordering::Type coords_ordering=Ordering::byVDIM,
|
||||
bool use_device=false);
|
||||
|
||||
/** @brief Construct a serial ParticleSet with specified fields and tags at
|
||||
* construction.
|
||||
@@ -426,9 +440,12 @@ public:
|
||||
* @param[in] num_tags Number of tags to register.
|
||||
* @param[in] all_ordering (Optional) Ordering of coordinates and
|
||||
* field ParticleVector.
|
||||
* @param[in] use_device (Optional) Use device memory for particle
|
||||
* fields.
|
||||
*/
|
||||
ParticleSet(int num_particles, int dim, const Array<int> &field_vdims,
|
||||
int num_tags, Ordering::Type all_ordering=Ordering::byVDIM);
|
||||
int num_tags, Ordering::Type all_ordering=Ordering::byVDIM,
|
||||
bool use_device=false);
|
||||
|
||||
/** @brief Construct a serial ParticleSet with specified fields and tags at
|
||||
* construction, with names.
|
||||
@@ -441,11 +458,14 @@ public:
|
||||
* @param[in] tag_names_ Array of tag names.
|
||||
* @param[in] all_ordering (Optional) Ordering of coordinates and
|
||||
* field ParticleVector.
|
||||
* @param[in] use_device (Optional) Use device memory for particle
|
||||
* fields.
|
||||
*/
|
||||
ParticleSet(int num_particles, int dim, const Array<int> &field_vdims,
|
||||
const Array<const char*> &field_names_, int num_tags,
|
||||
const Array<const char*> &tag_names_,
|
||||
Ordering::Type all_ordering=Ordering::byVDIM);
|
||||
Ordering::Type all_ordering=Ordering::byVDIM,
|
||||
bool use_device=false);
|
||||
|
||||
/** @brief Comprehensive serial constructor of ParticleSet.
|
||||
*
|
||||
@@ -457,12 +477,15 @@ public:
|
||||
* @param[in] field_names_ Array of field names.
|
||||
* @param[in] num_tags Number of tags to register.
|
||||
* @param[in] tag_names_ Array of tag names.
|
||||
* @param[in] use_device (Optional) Use device memory for particle
|
||||
* fields.
|
||||
*/
|
||||
ParticleSet(int num_particles, int dim, Ordering::Type coords_ordering,
|
||||
const Array<int> &field_vdims,
|
||||
const Array<Ordering::Type> &field_orderings,
|
||||
const Array<const char*> &field_names_, int num_tags,
|
||||
const Array<const char*> &tag_names_);
|
||||
const Array<const char*> &tag_names_,
|
||||
bool use_device=false);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/** @brief Construct a parallel ParticleSet.
|
||||
@@ -471,9 +494,12 @@ public:
|
||||
* @param[in] rank_num_particles Number of particles to initialize.
|
||||
* @param[in] dim Particle spatial dimension.
|
||||
* @param[in] coords_ordering (Optional) Ordering of coordinates.
|
||||
* @param[in] use_device (Optional) Use device memory for particle
|
||||
* fields.
|
||||
*/
|
||||
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
|
||||
Ordering::Type coords_ordering=Ordering::byVDIM);
|
||||
Ordering::Type coords_ordering=Ordering::byVDIM,
|
||||
bool use_device=false);
|
||||
|
||||
/** @brief Construct a parallel ParticleSet with specified fields and tags
|
||||
* at construction.
|
||||
@@ -485,10 +511,13 @@ public:
|
||||
* @param[in] num_tags Number of tags to register.
|
||||
* @param[in] all_ordering (Optional) Ordering of coordinates and
|
||||
* field ParticleVector.
|
||||
* @param[in] use_device (Optional) Use device memory for particle
|
||||
* fields.
|
||||
*/
|
||||
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
|
||||
const Array<int> &field_vdims, int num_tags,
|
||||
Ordering::Type all_ordering=Ordering::byVDIM);
|
||||
Ordering::Type all_ordering=Ordering::byVDIM,
|
||||
bool use_device=false);
|
||||
|
||||
/** @brief Construct a parallel ParticleSet with specified fields and tags
|
||||
* at construction, with names (for PrintCSV()).
|
||||
@@ -502,12 +531,15 @@ public:
|
||||
* @param[in] tag_names_ Array of tag names.
|
||||
* @param[in] all_ordering (Optional) Ordering of coordinates and
|
||||
* field ParticleVector.
|
||||
* @param[in] use_device (Optional) Use device memory for particle
|
||||
* fields.
|
||||
*/
|
||||
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
|
||||
const Array<int> &field_vdims,
|
||||
const Array<const char*> &field_names_,
|
||||
int num_tags, const Array<const char*> &tag_names_,
|
||||
Ordering::Type all_ordering=Ordering::byVDIM);
|
||||
Ordering::Type all_ordering=Ordering::byVDIM,
|
||||
bool use_device=false);
|
||||
|
||||
/** @brief Comprehensive parallel constructor of ParticleSet.
|
||||
*
|
||||
@@ -520,12 +552,15 @@ public:
|
||||
* @param[in] field_names_ Array of field names.
|
||||
* @param[in] num_tags Number of tags to register.
|
||||
* @param[in] tag_names_ Array of tag names.
|
||||
* @param[in] use_device (Optional) Use device memory for particle
|
||||
* fields.
|
||||
*/
|
||||
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
|
||||
Ordering::Type coords_ordering, const Array<int> &field_vdims,
|
||||
const Array<Ordering::Type> &field_orderings,
|
||||
const Array<const char*> &field_names_, int num_tags,
|
||||
const Array<const char*> &tag_names_);
|
||||
const Array<const char*> &tag_names_,
|
||||
bool use_device=false);
|
||||
|
||||
/// Get the MPI communicator for this ParticleSet.
|
||||
MPI_Comm GetComm() const { return comm; };
|
||||
@@ -545,6 +580,8 @@ public:
|
||||
* @param[in] field_ordering (Optional) Ordering::Type of the field.
|
||||
* @param[in] field_name (Optional) Name of the field.
|
||||
*
|
||||
* @note New fields inherit the current device mode of \ref coords.
|
||||
*
|
||||
* @return Index of the newly-added field.
|
||||
*/
|
||||
int AddField(int vdim, Ordering::Type field_ordering=Ordering::byVDIM,
|
||||
@@ -637,8 +674,8 @@ public:
|
||||
|
||||
/** @brief Determine if GetParticleRef is valid.
|
||||
*
|
||||
* If coordinates and all fields are ordered byVDIM, then returns true.
|
||||
* Otherwise, false.
|
||||
* Returns true when coordinates and all fields are ordered byVDIM and
|
||||
* particle data is host-resident. Otherwise, false.
|
||||
*/
|
||||
bool IsParticleRefValid() const;
|
||||
|
||||
|
||||
@@ -38,15 +38,6 @@
|
||||
#define CUB_IGNORE_DEPRECATED_CPP_DIALECT
|
||||
#define THRUST_IGNORE_DEPRECATED_CPP_DIALECT
|
||||
|
||||
// MFEM only supports using RAJA/CAMP backends in default stream mode because
|
||||
// memory calls are performed outside of the RAJA ecosystem
|
||||
#ifndef CAMP_USE_PLATFORM_DEFAULT_STREAM
|
||||
#define CAMP_USE_PLATFORM_DEFAULT_STREAM 1
|
||||
#else
|
||||
#if !CAMP_USE_PLATFORM_DEFAULT_STREAM
|
||||
#error "MFEM only supports RAJA/CAMP with the default platform stream."
|
||||
#endif
|
||||
#endif
|
||||
#include "RAJA/RAJA.hpp"
|
||||
#if defined(RAJA_ENABLE_CUDA) && !defined(MFEM_USE_CUDA)
|
||||
#error When RAJA is built with CUDA, MFEM_USE_CUDA=YES is required
|
||||
|
||||
+3
-1
@@ -581,7 +581,9 @@ void Device::Setup(const std::string &device_option, const int device_id)
|
||||
if (Allows(Backend::CUDA)) { CudaDeviceSetup(dev, ngpu); }
|
||||
if (Allows(Backend::HIP)) { HipDeviceSetup(dev, ngpu); }
|
||||
if (Allows(Backend::RAJA_CUDA) || Allows(Backend::RAJA_HIP))
|
||||
{ RajaDeviceSetup(dev, ngpu); }
|
||||
{
|
||||
RajaDeviceSetup(dev, ngpu);
|
||||
}
|
||||
// The check for MFEM_USE_OCCA is in the function OccaDeviceSetup().
|
||||
if (Allows(Backend::OCCA_MASK)) { OccaDeviceSetup(dev); }
|
||||
if (Allows(Backend::CEED_MASK))
|
||||
|
||||
@@ -16,6 +16,11 @@
|
||||
#include "globals.hpp"
|
||||
#include "mem_manager.hpp"
|
||||
|
||||
#ifdef MFEM_USE_RAJA
|
||||
#include "RAJA/RAJA.hpp"
|
||||
#endif
|
||||
|
||||
#include <memory>
|
||||
#include <string>
|
||||
|
||||
namespace mfem
|
||||
@@ -266,6 +271,18 @@ public:
|
||||
static inline bool Allows(unsigned long b_mask)
|
||||
{ return Get().backends & b_mask; }
|
||||
|
||||
#if defined(MFEM_USE_RAJA) && \
|
||||
(defined(RAJA_ENABLE_CUDA) || defined(RAJA_ENABLE_HIP))
|
||||
static inline auto GetRajaResource()
|
||||
{
|
||||
#if defined(RAJA_ENABLE_CUDA)
|
||||
return RAJA::resources::Cuda::CudaFromStream(0, Get().GetId());
|
||||
#elif defined(RAJA_ENABLE_HIP)
|
||||
return RAJA::resources::Hip::HipFromStream(0, Get().GetId());
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
/** @brief Get the current Host MemoryType. This is the MemoryType used by
|
||||
most MFEM classes when allocating memory used on the host.
|
||||
*/
|
||||
|
||||
+30
-38
@@ -317,8 +317,8 @@ template <typename DBODY>
|
||||
void RajaCuWrap1D(const int N, DBODY &&d_body)
|
||||
{
|
||||
//true denotes asynchronous kernel
|
||||
RAJA::forall<RAJA::cuda_exec<MFEM_CUDA_BLOCKS,true>>(RAJA::RangeSegment(0,N),
|
||||
d_body);
|
||||
RAJA::forall<RAJA::cuda_exec<MFEM_CUDA_BLOCKS, true> >(
|
||||
Device::GetRajaResource(), RAJA::RangeSegment(0, N), d_body);
|
||||
}
|
||||
|
||||
template <typename DBODY>
|
||||
@@ -331,9 +331,9 @@ void RajaCuWrap2D(const int N, DBODY &&d_body,
|
||||
using namespace RAJA;
|
||||
using RAJA::RangeSegment;
|
||||
|
||||
launch<cuda_launch_policy>
|
||||
(LaunchParams(Teams(G), Threads(X, Y, BZ)),
|
||||
[=] RAJA_DEVICE (LaunchContext ctx)
|
||||
launch<cuda_launch_policy>(Device::GetRajaResource(),
|
||||
LaunchParams(Teams(G), Threads(X, Y, BZ)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
{
|
||||
|
||||
loop<cuda_teams_x>(ctx, RangeSegment(0, G), [&] (const int n)
|
||||
@@ -349,7 +349,6 @@ void RajaCuWrap2D(const int N, DBODY &&d_body,
|
||||
});
|
||||
|
||||
});
|
||||
|
||||
});
|
||||
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
@@ -365,9 +364,9 @@ void RajaCuWrap2DLaunchBounds(const int N, DBODY &&d_body, const int X,
|
||||
using namespace RAJA;
|
||||
using RAJA::RangeSegment;
|
||||
|
||||
launch<cuda_launch_bounds_policy<LB> >
|
||||
(LaunchParams(Teams(G), Threads(X, Y, BZ)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
launch<cuda_launch_bounds_policy<LB> >(
|
||||
Device::GetRajaResource(), LaunchParams(Teams(G), Threads(X, Y, BZ)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
{
|
||||
loop<cuda_teams_x>(ctx, RangeSegment(0, G), [&] (const int n)
|
||||
{
|
||||
@@ -390,13 +389,12 @@ void RajaCuWrap3D(const int N, DBODY &&d_body,
|
||||
using namespace RAJA;
|
||||
using RAJA::RangeSegment;
|
||||
|
||||
launch<cuda_launch_policy>
|
||||
(LaunchParams(Teams(GRID), Threads(X, Y, Z)),
|
||||
[=] RAJA_DEVICE (LaunchContext ctx)
|
||||
launch<cuda_launch_policy>(Device::GetRajaResource(),
|
||||
LaunchParams(Teams(GRID), Threads(X, Y, Z)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
{
|
||||
|
||||
loop<cuda_teams_x>(ctx, RangeSegment(0, N), d_body);
|
||||
|
||||
});
|
||||
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
@@ -410,12 +408,10 @@ void RajaCuWrap3DLaunchBounds(const int N, DBODY &&d_body,
|
||||
using namespace RAJA;
|
||||
using RAJA::RangeSegment;
|
||||
|
||||
launch<cuda_launch_bounds_policy<LB> >
|
||||
(LaunchParams(Teams(GRID), Threads(X, Y, Z)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
{
|
||||
loop<cuda_teams_x>(ctx, RangeSegment(0, N), d_body);
|
||||
});
|
||||
launch<cuda_launch_bounds_policy<LB> >(
|
||||
Device::GetRajaResource(), LaunchParams(Teams(GRID), Threads(X, Y, Z)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
{ loop<cuda_teams_x>(ctx, RangeSegment(0, N), d_body); });
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
}
|
||||
|
||||
@@ -484,8 +480,8 @@ template <typename DBODY>
|
||||
void RajaHipWrap1D(const int N, DBODY &&d_body)
|
||||
{
|
||||
//true denotes asynchronous kernel
|
||||
RAJA::forall<RAJA::hip_exec<MFEM_HIP_BLOCKS,true>>(RAJA::RangeSegment(0,N),
|
||||
d_body);
|
||||
RAJA::forall<RAJA::hip_exec<MFEM_HIP_BLOCKS,true> >(RAJA::RangeSegment(0,N),
|
||||
d_body);
|
||||
}
|
||||
|
||||
template <typename DBODY>
|
||||
@@ -498,9 +494,9 @@ void RajaHipWrap2D(const int N, DBODY &&d_body,
|
||||
using namespace RAJA;
|
||||
using RAJA::RangeSegment;
|
||||
|
||||
launch<hip_launch_policy>
|
||||
(LaunchParams(Teams(G), Threads(X, Y, BZ)),
|
||||
[=] RAJA_DEVICE (LaunchContext ctx)
|
||||
launch<hip_launch_policy>(Device::GetRajaResource(),
|
||||
LaunchParams(Teams(G), Threads(X, Y, BZ)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
{
|
||||
|
||||
loop<hip_teams_x>(ctx, RangeSegment(0, G), [&] (const int n)
|
||||
@@ -516,7 +512,6 @@ void RajaHipWrap2D(const int N, DBODY &&d_body,
|
||||
});
|
||||
|
||||
});
|
||||
|
||||
});
|
||||
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
@@ -532,9 +527,9 @@ void RajaHipWrap2DLaunchBounds(const int N, DBODY &&d_body, const int X,
|
||||
using namespace RAJA;
|
||||
using RAJA::RangeSegment;
|
||||
|
||||
launch<hip_launch_bounds_policy<LB> >
|
||||
(LaunchParams(Teams(G), Threads(X, Y, BZ)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
launch<hip_launch_bounds_policy<LB> >(
|
||||
Device::GetRajaResource(), LaunchParams(Teams(G), Threads(X, Y, BZ)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
{
|
||||
loop<hip_teams_x>(ctx, RangeSegment(0, G), [&] (const int n)
|
||||
{
|
||||
@@ -557,13 +552,12 @@ void RajaHipWrap3D(const int N, DBODY &&d_body,
|
||||
using namespace RAJA;
|
||||
using RAJA::RangeSegment;
|
||||
|
||||
launch<hip_launch_policy>
|
||||
(LaunchParams(Teams(GRID), Threads(X, Y, Z)),
|
||||
[=] RAJA_DEVICE (LaunchContext ctx)
|
||||
launch<hip_launch_policy>(Device::GetRajaResource(),
|
||||
LaunchParams(Teams(GRID), Threads(X, Y, Z)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
{
|
||||
|
||||
loop<hip_teams_x>(ctx, RangeSegment(0, N), d_body);
|
||||
|
||||
});
|
||||
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
@@ -577,12 +571,10 @@ void RajaHipWrap3DLaunchBounds(const int N, DBODY &&d_body, const int X,
|
||||
using namespace RAJA;
|
||||
using RAJA::RangeSegment;
|
||||
|
||||
launch<hip_launch_bounds_policy<LB> >
|
||||
(LaunchParams(Teams(GRID), Threads(X, Y, Z)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
{
|
||||
loop<hip_teams_x>(ctx, RangeSegment(0, N), d_body);
|
||||
});
|
||||
launch<hip_launch_bounds_policy<LB> >(
|
||||
Device::GetRajaResource(), LaunchParams(Teams(GRID), Threads(X, Y, Z)),
|
||||
[=] RAJA_DEVICE(LaunchContext ctx)
|
||||
{ loop<hip_teams_x>(ctx, RangeSegment(0, N), d_body); });
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
}
|
||||
|
||||
|
||||
+40
-9
@@ -15,10 +15,20 @@
|
||||
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
|
||||
#if CUDSS_VERSION >= 800
|
||||
#ifdef MFEM_USE_SINGLE
|
||||
#define CUDA_REAL_T CUDA_R_32F
|
||||
#define CUDSS_REAL_T CUDSS_R_32F
|
||||
#else
|
||||
#define CUDA_REAL_T CUDA_R_64F
|
||||
#define CUDSS_REAL_T CUDSS_R_64F
|
||||
#endif
|
||||
#define CUDSS_INT_T CUDSS_R_32I
|
||||
#else
|
||||
#ifdef MFEM_USE_SINGLE
|
||||
#define CUDSS_REAL_T CUDA_R_32F
|
||||
#else
|
||||
#define CUDSS_REAL_T CUDA_R_64F
|
||||
#endif
|
||||
#define CUDSS_INT_T CUDA_R_32I
|
||||
#endif
|
||||
|
||||
// Define a cuDSS error check macro, MFEM_CUDSS_CHECK(x), where x returns/is of
|
||||
@@ -65,8 +75,13 @@ CuDSSSolver::CuDSSSolver(MPI_Comm comm_) : mpi_comm(comm_)
|
||||
#endif
|
||||
MFEM_CUDSS_CHECK(cudssSetCommLayer(handle, comm_lib));
|
||||
|
||||
#if CUDSS_VERSION >= 800
|
||||
MFEM_CUDSS_CHECK(cudssDataSet(handle, solverData, CUDSS_DATA_COMM_HOST,
|
||||
&mpi_comm, sizeof(MPI_Comm *)));
|
||||
#else
|
||||
MFEM_CUDSS_CHECK(cudssDataSet(handle, solverData, CUDSS_DATA_COMM,
|
||||
&mpi_comm, sizeof(MPI_Comm *)));
|
||||
#endif
|
||||
}
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
@@ -257,11 +272,19 @@ void CuDSSSolver::SetMatrixCuDSS(int *csr_offsets, int *csr_columns,
|
||||
CuMemcpyDtoD(csr_offsets_d, csr_offsets, (n_loc + 1) * sizeof(int));
|
||||
CuMemcpyDtoD(csr_columns_d, csr_columns, nnz * sizeof(int));
|
||||
|
||||
#if CUDSS_VERSION >= 800
|
||||
MFEM_CUDSS_CHECK(
|
||||
cudssMatrixCreateCsr(
|
||||
Ac.get(), n_global, n_global, nnz, csr_offsets_d, NULL,
|
||||
csr_columns_d, csr_values_d, CUDA_R_32I, CUDA_REAL_T, mat_type, mview,
|
||||
CUDSS_BASE_ZERO));
|
||||
csr_columns_d, csr_values_d, CUDSS_INT_T, CUDSS_INT_T, CUDSS_REAL_T,
|
||||
mat_type, mview, CUDSS_BASE_ZERO));
|
||||
#else
|
||||
MFEM_CUDSS_CHECK(
|
||||
cudssMatrixCreateCsr(
|
||||
Ac.get(), n_global, n_global, nnz, csr_offsets_d, NULL,
|
||||
csr_columns_d, csr_values_d, CUDSS_INT_T, CUDSS_REAL_T,
|
||||
mat_type, mview, CUDSS_BASE_ZERO));
|
||||
#endif
|
||||
}
|
||||
else // !reorder_reuse
|
||||
{
|
||||
@@ -269,11 +292,19 @@ void CuDSSSolver::SetMatrixCuDSS(int *csr_offsets, int *csr_columns,
|
||||
{
|
||||
MFEM_CUDSS_CHECK(cudssMatrixDestroy(*Ac));
|
||||
}
|
||||
#if CUDSS_VERSION >= 800
|
||||
MFEM_CUDSS_CHECK(
|
||||
cudssMatrixCreateCsr(
|
||||
Ac.get(), n_global, n_global, nnz, csr_offsets, NULL, csr_columns,
|
||||
csr_values_d, CUDA_R_32I, CUDA_REAL_T, mat_type, mview,
|
||||
CUDSS_BASE_ZERO));
|
||||
Ac.get(), n_global, n_global, nnz, csr_offsets, NULL,
|
||||
csr_columns, csr_values_d, CUDSS_INT_T, CUDSS_INT_T, CUDSS_REAL_T,
|
||||
mat_type, mview, CUDSS_BASE_ZERO));
|
||||
#else
|
||||
MFEM_CUDSS_CHECK(
|
||||
cudssMatrixCreateCsr(
|
||||
Ac.get(), n_global, n_global, nnz, csr_offsets, NULL,
|
||||
csr_columns, csr_values_d, CUDSS_INT_T, CUDSS_REAL_T,
|
||||
mat_type, mview, CUDSS_BASE_ZERO));
|
||||
#endif
|
||||
}
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (Mpi::IsInitialized())
|
||||
@@ -334,10 +365,10 @@ void CuDSSSolver::SetNumRHS(int nrhs_) const
|
||||
}
|
||||
// Create empty RHS and solution vectors
|
||||
MFEM_CUDSS_CHECK(cudssMatrixCreateDn(&xc, n_global, nrhs_, n_global, NULL,
|
||||
CUDA_REAL_T, CUDSS_LAYOUT_COL_MAJOR));
|
||||
CUDSS_REAL_T, CUDSS_LAYOUT_COL_MAJOR));
|
||||
|
||||
MFEM_CUDSS_CHECK(cudssMatrixCreateDn(&yc, n_global, nrhs_, n_global, NULL,
|
||||
CUDA_REAL_T, CUDSS_LAYOUT_COL_MAJOR));
|
||||
CUDSS_REAL_T, CUDSS_LAYOUT_COL_MAJOR));
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
MFEM_CUDSS_CHECK(cudssMatrixSetDistributionRow1d(xc, row_start, row_end));
|
||||
|
||||
@@ -39,6 +39,7 @@ namespace Ginkgo
|
||||
{
|
||||
|
||||
template <typename T> using gko_array = gko::array<T>;
|
||||
#if defined(MFEM_USE_MPI) && GINKGO_BUILD_MPI
|
||||
// for inter-operability with hypre integer types
|
||||
using gko_hypre_int =
|
||||
std::conditional_t<sizeof(HYPRE_Int) == sizeof(std::int32_t), std::int32_t,
|
||||
@@ -50,6 +51,7 @@ static_assert(!std::is_void_v<gko_hypre_int>,
|
||||
"HYPRE_Int type is incompatible with Ginkgo");
|
||||
static_assert(!std::is_void_v<gko_hypre_bigint>,
|
||||
"HYPRE_BigInt type is incompatible with Ginkgo");
|
||||
#endif
|
||||
|
||||
/**
|
||||
* Helper class for a case where a wrapped MFEM Vector
|
||||
|
||||
+29
-3
@@ -2872,8 +2872,8 @@ void HypreParMatrix::Destroy()
|
||||
if (HypreUsingGPU() && ParCSROwner && (diagOwner < 0 || offdOwner < 0))
|
||||
{
|
||||
// Put the "host" or "hypre" pointers in {i,j,data} of A->{diag,offd}, so
|
||||
// that they can be destroyed by hypre when hypre_ParCSRMatrixDestroy(A)
|
||||
// is called below.
|
||||
// that they can be destroyed by mfem_hypre_TFree_host() or hypre when
|
||||
// hypre_ParCSRMatrixDestroy(A) is called below, respectively.
|
||||
|
||||
// Check that if both diagOwner and offdOwner are negative then they have
|
||||
// the same value.
|
||||
@@ -2882,7 +2882,33 @@ void HypreParMatrix::Destroy()
|
||||
|
||||
MemoryClass mc = (diagOwner == -1 || offdOwner == -1) ?
|
||||
Device::GetHostMemoryClass() : GetHypreMemoryClass();
|
||||
Write(mc, diagOwner < 0, offdOwner <0);
|
||||
Write(mc, diagOwner < 0, offdOwner < 0);
|
||||
if (diagOwner == -1)
|
||||
{
|
||||
// Note: mfem_hypre_TFree_host() sets the pointer to NULL.
|
||||
mfem_hypre_TFree_host(hypre_CSRMatrixI(A->diag));
|
||||
if (hypre_CSRMatrixOwnsData(A->diag))
|
||||
{
|
||||
mfem_hypre_TFree_host(hypre_CSRMatrixJ(A->diag));
|
||||
mfem_hypre_TFree_host(hypre_CSRMatrixData(A->diag));
|
||||
}
|
||||
#if MFEM_HYPRE_VERSION >= 21800
|
||||
hypre_CSRMatrixMemoryLocation(A->diag) = GetHypreMemoryLocation();
|
||||
#endif
|
||||
}
|
||||
if (offdOwner == -1)
|
||||
{
|
||||
// Note: mfem_hypre_TFree_host() sets the pointer to NULL.
|
||||
mfem_hypre_TFree_host(hypre_CSRMatrixI(A->offd));
|
||||
if (hypre_CSRMatrixOwnsData(A->offd))
|
||||
{
|
||||
mfem_hypre_TFree_host(hypre_CSRMatrixJ(A->offd));
|
||||
mfem_hypre_TFree_host(hypre_CSRMatrixData(A->offd));
|
||||
}
|
||||
#if MFEM_HYPRE_VERSION >= 21800
|
||||
hypre_CSRMatrixMemoryLocation(A->offd) = GetHypreMemoryLocation();
|
||||
#endif
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
+5
-4
@@ -432,10 +432,11 @@ private:
|
||||
// and A->col_map_offd.
|
||||
// The possible values for diagOwner are:
|
||||
// -1: no special treatment of A->diag (default)
|
||||
// when hypre is built with CUDA support, A->diag owns the "host"
|
||||
// pointers (according to A->diag->owns_data)
|
||||
// -2: used when hypre is built with CUDA support, A->diag owns the "hypre"
|
||||
// pointers (according to A->diag->owns_data)
|
||||
// when hypre is using GPU, A->diag owns the "host" pointers (according
|
||||
// to A->diag->owns_data); these host pointers are freed by MFEM using
|
||||
// hypre's host deallocation macros
|
||||
// -2: used when hypre is using GPU, A->diag owns the "hypre" pointers
|
||||
// (according to A->diag->owns_data)
|
||||
// 0: prevent hypre from destroying A->diag->{i,j,data}
|
||||
// 1: same as 0, plus own the "host" A->diag->{i,j}
|
||||
// 2: same as 0, plus own the "host" A->diag->data
|
||||
|
||||
+112
-39
@@ -10,6 +10,7 @@
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "particlevector.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -46,20 +47,38 @@ void ParticleVector::GetValues(int i, Vector &nvals) const
|
||||
{
|
||||
nvals.SetSize(vdim);
|
||||
|
||||
if (ordering == Ordering::byNODES)
|
||||
const bool nvals_use_dev = nvals.UseDevice();
|
||||
// Use ParticleVector's device flag to minimize movement from large source
|
||||
const bool use_dev = UseDevice();
|
||||
const auto d_src = Read(use_dev);
|
||||
auto d_dest = nvals.Write(use_dev);
|
||||
|
||||
const int vdim_ = vdim;
|
||||
const int ordering_ = (int)ordering;
|
||||
const int nv = (ordering == Ordering::byNODES) ? size / vdim : 0;
|
||||
|
||||
mfem::forall_switch(use_dev, vdim_, [=] MFEM_HOST_DEVICE (int c)
|
||||
{
|
||||
int nv = GetNumParticles();
|
||||
for (int c = 0; c < vdim; c++)
|
||||
if (ordering_ == Ordering::byNODES)
|
||||
{
|
||||
nvals[c] = Vector::operator[](i+nv*c);
|
||||
d_dest[c] = d_src[i + nv*c];
|
||||
}
|
||||
else
|
||||
{
|
||||
d_dest[c] = d_src[c + vdim_*i];
|
||||
}
|
||||
});
|
||||
|
||||
// If nvals was not using device but ParticleVector is, copy back to host
|
||||
if (!nvals_use_dev && use_dev)
|
||||
{
|
||||
nvals.HostRead();
|
||||
nvals.UseDevice(false);
|
||||
}
|
||||
else
|
||||
// If nvals was using device but ParticleVector is not, copy back to device
|
||||
if (!use_dev && nvals_use_dev)
|
||||
{
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
nvals[c] = Vector::operator[](c+vdim*i);
|
||||
}
|
||||
nvals.Read();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -99,21 +118,27 @@ void ParticleVector::GetComponentsRef(int vd, Vector &nref)
|
||||
|
||||
void ParticleVector::SetValues(int i, const Vector &nvals)
|
||||
{
|
||||
if (ordering == Ordering::byNODES)
|
||||
const bool use_dev = UseDevice(); // use ParticleVector's device flag
|
||||
const auto mc = use_dev ? Device::GetDeviceMemoryClass()
|
||||
: Device::GetHostMemoryClass();
|
||||
auto d_dest = ReadWrite(use_dev);
|
||||
const auto d_src = nvals.GetMemory().Read(mc, nvals.Size());
|
||||
|
||||
const int vdim_ = vdim;
|
||||
const int ordering_ = (int)ordering;
|
||||
const int nv = (ordering == Ordering::byNODES) ? size / vdim : 0;
|
||||
|
||||
mfem::forall_switch(use_dev, vdim_, [=] MFEM_HOST_DEVICE (int c)
|
||||
{
|
||||
int nv = GetNumParticles();
|
||||
for (int c = 0; c < vdim; c++)
|
||||
if (ordering_ == Ordering::byNODES)
|
||||
{
|
||||
Vector::operator[](i + c*nv) = nvals[c];
|
||||
d_dest[i + c*nv] = d_src[c];
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int c = 0; c < vdim; c++)
|
||||
else
|
||||
{
|
||||
Vector::operator[](c + i*vdim) = nvals[c];
|
||||
d_dest[c + i*vdim_] = d_src[c];
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void ParticleVector::SetComponents(int vd, const Vector &comp)
|
||||
@@ -144,6 +169,9 @@ real_t& ParticleVector::operator()(int i, int comp)
|
||||
"Component index " << comp <<
|
||||
" is invalid for vector dimension " << vdim);
|
||||
|
||||
// non-const so we make host flag valid in case user modifies data
|
||||
HostReadWrite();
|
||||
|
||||
if (ordering == Ordering::byNODES)
|
||||
{
|
||||
return Vector::operator[](i + comp*GetNumParticles());
|
||||
@@ -163,6 +191,8 @@ const real_t& ParticleVector::operator()(int i, int comp) const
|
||||
"Component index " << comp <<
|
||||
" is invalid for vector dimension " << vdim);
|
||||
|
||||
HostRead();
|
||||
|
||||
if (ordering == Ordering::byNODES)
|
||||
{
|
||||
return Vector::operator[](i + comp*GetNumParticles());
|
||||
@@ -240,9 +270,37 @@ void ParticleVector::SetVDim(int vdim_, bool keep_data)
|
||||
|
||||
void ParticleVector::SetOrdering(Ordering::Type ordering_, bool keep_data)
|
||||
{
|
||||
if (keep_data)
|
||||
if (keep_data && ordering != ordering_)
|
||||
{
|
||||
Ordering::Reorder(*this, vdim, ordering, ordering_);
|
||||
int num_particles = GetNumParticles();
|
||||
// create deep copy of old data that will be copied
|
||||
Vector old_data(*this);
|
||||
|
||||
const bool use_dev = UseDevice();
|
||||
const auto d_src = old_data.Read(use_dev);
|
||||
auto d_dest = Write(use_dev);
|
||||
|
||||
const int vdim_ = vdim;
|
||||
const int size_ = size;
|
||||
|
||||
if (ordering_ == Ordering::byNODES) // byVDIM -> byNODES
|
||||
{
|
||||
mfem::forall_switch(use_dev, size_, [=] MFEM_HOST_DEVICE (int k)
|
||||
{
|
||||
int i = k / vdim_; // src particle index
|
||||
int d = k % vdim_; // src component index
|
||||
d_dest[i + d * num_particles] = d_src[k];
|
||||
});
|
||||
}
|
||||
else // byNODES -> byVDIM
|
||||
{
|
||||
mfem::forall_switch(use_dev, size_, [=] MFEM_HOST_DEVICE (int k)
|
||||
{
|
||||
int d = k / num_particles; // src component index
|
||||
int i = k % num_particles; // src particle index
|
||||
d_dest[d + i * vdim_] = d_src[k];
|
||||
});
|
||||
}
|
||||
}
|
||||
ordering = ordering_;
|
||||
}
|
||||
@@ -270,32 +328,47 @@ void ParticleVector::SetNumParticles(int num_vectors, bool keep_data)
|
||||
|
||||
if (!keep_data) { return; }
|
||||
|
||||
const bool use_dev = UseDevice();
|
||||
auto d_dest = this->ReadWrite(use_dev);
|
||||
|
||||
if (ordering == Ordering::byNODES)
|
||||
{
|
||||
// Shift entries for byNODES
|
||||
for (int c = vdim-1; c > 0; c--)
|
||||
{
|
||||
for (int i = old_nv-1; i >= 0; i--)
|
||||
{
|
||||
Vector::operator[](i+c*num_vectors) = Vector::operator[](i+c*old_nv);
|
||||
}
|
||||
}
|
||||
// create deep copy of old data that will be copied
|
||||
Vector old_slice;
|
||||
old_slice.MakeRef(*this, 0, old_nv * vdim);
|
||||
Vector old_copy(old_slice);
|
||||
|
||||
// Zero-out data now associated with new Vectors
|
||||
for (int c = 0; c < vdim; c++)
|
||||
const auto d_src = old_copy.Read(use_dev);
|
||||
const int vdim_ = vdim;
|
||||
|
||||
// Shift entries for byNODES
|
||||
mfem::forall_switch(use_dev, old_nv * vdim_,
|
||||
[=] MFEM_HOST_DEVICE (int k)
|
||||
{
|
||||
for (int i = old_nv; i < num_vectors; i++)
|
||||
{
|
||||
Vector::operator[](i+c*num_vectors) = 0.0;
|
||||
}
|
||||
}
|
||||
const int d = k / old_nv;
|
||||
const int i = k % old_nv;
|
||||
d_dest[i + d*num_vectors] = d_src[k];
|
||||
});
|
||||
|
||||
// Zero-out new data slots
|
||||
const int diff = num_vectors - old_nv;
|
||||
mfem::forall_switch(use_dev, diff * vdim,
|
||||
[=] MFEM_HOST_DEVICE (int k)
|
||||
{
|
||||
const int d = k / diff;
|
||||
const int i = k % diff;
|
||||
d_dest[d * num_vectors + old_nv + i] = 0.0;
|
||||
});
|
||||
}
|
||||
else // byVDIM
|
||||
{
|
||||
for (int i = old_nv*vdim; i < num_vectors*vdim; i++)
|
||||
const int start_idx = old_nv * vdim;
|
||||
const int end_idx = num_vectors * vdim;
|
||||
const int diff = end_idx - start_idx;
|
||||
mfem::forall_switch(use_dev, diff, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
data[i] = 0.0;
|
||||
}
|
||||
d_dest[start_idx + i] = 0.0;
|
||||
});
|
||||
}
|
||||
}
|
||||
else // Else just remove the trailing vector data
|
||||
|
||||
@@ -33,6 +33,7 @@ add_subdirectory(meshing)
|
||||
add_subdirectory(mtop)
|
||||
add_subdirectory(multidomain)
|
||||
add_subdirectory(nurbs)
|
||||
add_subdirectory(optprob)
|
||||
add_subdirectory(parelag)
|
||||
add_subdirectory(performance)
|
||||
add_subdirectory(plasma)
|
||||
|
||||
@@ -126,6 +126,7 @@ void VisualizeParticles(socketstream &sock, const char* vishost, int visport,
|
||||
{
|
||||
Vector pcoords;
|
||||
pset.Coords().GetValues(i, pcoords);
|
||||
pcoords.HostRead();
|
||||
if (dim == 2)
|
||||
{
|
||||
Add2DPoint(pcoords, particles_mesh, psize);
|
||||
@@ -139,6 +140,7 @@ void VisualizeParticles(socketstream &sock, const char* vishost, int visport,
|
||||
|
||||
FiniteElementSpace fes(&particles_mesh, &l2fec, 1);
|
||||
GridFunction gf(&fes);
|
||||
gf.HostWrite();
|
||||
|
||||
for (int i = 0; i < pset.GetNParticles(); i++)
|
||||
{
|
||||
@@ -193,6 +195,7 @@ void ParticleTrajectories::AddSegmentStart()
|
||||
{
|
||||
Vector pcoords;
|
||||
pset.Coords().GetValues(i, pcoords);
|
||||
pcoords.HostRead();
|
||||
segment_meshes.front().AddVertex(pcoords);
|
||||
}
|
||||
}
|
||||
@@ -213,6 +216,7 @@ void ParticleTrajectories::SetSegmentEnd()
|
||||
{
|
||||
Vector pcoords;
|
||||
pset.Coords().GetValues(pidx, pcoords);
|
||||
pcoords.HostRead();
|
||||
segment_meshes.front().AddVertex(pcoords);
|
||||
}
|
||||
else // Otherwise set its end vertex == start vertex
|
||||
|
||||
@@ -91,6 +91,7 @@ struct LorentzContext
|
||||
int nt = 1000; // number of timesteps
|
||||
int redist_interval = 5; // redistribution interval
|
||||
int redist_mesh = 0; // redistribution mesh: 0: E mesh, 1: B mesh
|
||||
std::string device_config = "cpu";
|
||||
} ctx;
|
||||
|
||||
/// This class implements the Boris algorithm as described in the article
|
||||
@@ -130,7 +131,7 @@ protected:
|
||||
public:
|
||||
|
||||
Boris(MPI_Comm comm, GridFunction *E_gf_, GridFunction *B_gf_,
|
||||
int nparticles, Ordering::Type pdata_ordering);
|
||||
int nparticles, Ordering::Type pdata_ordering, bool use_device);
|
||||
|
||||
/// Find Particles in mesh corresponding to E and B fields
|
||||
void FindParticles();
|
||||
@@ -139,9 +140,12 @@ public:
|
||||
/// right after FindParticles has been called.
|
||||
void EvaluateFieldsAtParticles();
|
||||
|
||||
/// Advance particles one time step using Boris algorithm
|
||||
/// Advance particles one time step using Boris algorithm. Host version.
|
||||
void Step(real_t &t, real_t &dt);
|
||||
|
||||
/// Advance particles one time step using Boris algorithm. Device version.
|
||||
void StepDevice(real_t &t, real_t &dt);
|
||||
|
||||
/// Remove lost particles and return their indices
|
||||
Array<int> RemoveLostParticles();
|
||||
|
||||
@@ -235,6 +239,8 @@ int main(int argc, char *argv[])
|
||||
args.AddOption(&vis_interval, "-vf", "--vis-interval",
|
||||
"GLVis visualization update after this many timesteps. "
|
||||
"0 means no visualization.");
|
||||
args.AddOption(&ctx.device_config, "-d", "--device",
|
||||
"Device configuration definition string.");
|
||||
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
@@ -251,6 +257,10 @@ int main(int argc, char *argv[])
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
Device device(ctx.device_config);
|
||||
if (Mpi::Root()) { device.Print(); }
|
||||
bool use_device = (ctx.device_config != "cpu") && Device::IsEnabled();
|
||||
|
||||
std::unique_ptr<VisItDataCollection> E_dc, B_dc;
|
||||
ParGridFunction *E_gf = nullptr, *B_gf = nullptr;
|
||||
Vector bb_xmin, bb_xmax;
|
||||
@@ -266,6 +276,7 @@ int main(int argc, char *argv[])
|
||||
return 1;
|
||||
}
|
||||
E_gf->ParFESpace()->GetParMesh()->GetBoundingBox(bb_xmin, bb_xmax, 2);
|
||||
E_gf->UseDevice(use_device);
|
||||
}
|
||||
|
||||
// Read B field if provided
|
||||
@@ -280,6 +291,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
Vector bb_xmint, bb_xmaxt;
|
||||
B_gf->ParFESpace()->GetParMesh()->GetBoundingBox(bb_xmint, bb_xmaxt, 2);
|
||||
B_gf->UseDevice(use_device);
|
||||
if (ctx.E.coll_name != "")
|
||||
{
|
||||
// compute intersection of bounding boxes
|
||||
@@ -302,10 +314,14 @@ int main(int argc, char *argv[])
|
||||
// Initialize particles
|
||||
int num_particles = ctx.npt/num_ranks +
|
||||
(rank < (ctx.npt % num_ranks) ? 1 : 0);
|
||||
Boris boris(MPI_COMM_WORLD, E_gf, B_gf, num_particles, ordering_type);
|
||||
Boris boris(MPI_COMM_WORLD, E_gf, B_gf, num_particles, ordering_type,
|
||||
use_device);
|
||||
InitializeChargedParticles(boris.GetParticles(), ctx.x_min, ctx.x_max,
|
||||
ctx.p_min, ctx.p_max, ctx.m, ctx.q);
|
||||
|
||||
Array<int> removed_idxs_dummy;
|
||||
boris.FindParticles();
|
||||
boris.Redistribute(ctx.redist_mesh, removed_idxs_dummy);
|
||||
boris.EvaluateFieldsAtParticles();
|
||||
|
||||
real_t t = 0.0;
|
||||
@@ -329,7 +345,14 @@ int main(int argc, char *argv[])
|
||||
for (int step = 1; step <= ctx.nt; step++)
|
||||
{
|
||||
// Step the Boris algorithm
|
||||
boris.Step(t, dt);
|
||||
if (use_device)
|
||||
{
|
||||
boris.StepDevice(t, dt);
|
||||
}
|
||||
else
|
||||
{
|
||||
boris.Step(t, dt);
|
||||
}
|
||||
if (Mpi::Root())
|
||||
{
|
||||
mfem::out << "Step: " << step << " | Time: " << t << endl;
|
||||
@@ -397,7 +420,7 @@ void Boris::ParticleStep(Particle &part, real_t &dt)
|
||||
}
|
||||
|
||||
Boris::Boris(MPI_Comm comm, GridFunction *E_gf_, GridFunction *B_gf_,
|
||||
int nparticles, Ordering::Type pdata_ordering)
|
||||
int nparticles, Ordering::Type pdata_ordering, bool use_device)
|
||||
: E_gf(E_gf_),
|
||||
B_gf(B_gf_),
|
||||
E_finder(comm),
|
||||
@@ -426,6 +449,7 @@ Boris::Boris(MPI_Comm comm, GridFunction *E_gf_, GridFunction *B_gf_,
|
||||
}
|
||||
|
||||
int dim = E_mesh ? E_mesh->SpaceDimension() : B_mesh->SpaceDimension();
|
||||
MFEM_VERIFY(dim == 3, "Only 3D meshes are currently supported.");
|
||||
|
||||
pxB_.SetSize(dim); pm_.SetSize(dim); pp_.SetSize(dim);
|
||||
|
||||
@@ -435,7 +459,8 @@ Boris::Boris(MPI_Comm comm, GridFunction *E_gf_, GridFunction *B_gf_,
|
||||
Array<int> field_vdims({1, 1, dim, dim, dim});
|
||||
|
||||
charged_particles = std::make_unique<ParticleSet>
|
||||
(comm, nparticles, dim, field_vdims, 0, pdata_ordering);
|
||||
(comm, nparticles, dim, field_vdims, 0, pdata_ordering,
|
||||
use_device);
|
||||
}
|
||||
|
||||
void Boris::FindParticles()
|
||||
@@ -481,7 +506,6 @@ void Boris::Step(real_t &t, real_t &dt)
|
||||
{
|
||||
// Interpolate E and B fields onto particles
|
||||
EvaluateFieldsAtParticles();
|
||||
|
||||
// Individually step each particle. If all ParticleSet fields are ordered
|
||||
// byVDIM, we can use GetParticleRef for better performance.
|
||||
if (charged_particles->IsParticleRefValid())
|
||||
@@ -509,6 +533,112 @@ void Boris::Step(real_t &t, real_t &dt)
|
||||
t += dt;
|
||||
}
|
||||
|
||||
void Boris::StepDevice(real_t &t, real_t &dt)
|
||||
{
|
||||
// Interpolate E and B fields onto particles
|
||||
EvaluateFieldsAtParticles();
|
||||
const int N = charged_particles->GetNParticles();
|
||||
auto &X = charged_particles->Coords();
|
||||
auto &M = charged_particles->Field(MASS);
|
||||
auto &Q = charged_particles->Field(CHARGE);
|
||||
auto &P = charged_particles->Field(MOM);
|
||||
auto &E = charged_particles->Field(EFIELD);
|
||||
auto &B = charged_particles->Field(BFIELD);
|
||||
|
||||
const int dim = X.GetVDim();
|
||||
|
||||
// Capture orderings for each field to ensure correct access
|
||||
const bool byVDIM_X = (X.GetOrdering() == Ordering::byVDIM);
|
||||
const bool byVDIM_P = (P.GetOrdering() == Ordering::byVDIM);
|
||||
const bool byVDIM_E = (E.GetOrdering() == Ordering::byVDIM);
|
||||
const bool byVDIM_B = (B.GetOrdering() == Ordering::byVDIM);
|
||||
|
||||
auto d_x = X.ReadWrite();
|
||||
auto d_m = M.Read();
|
||||
auto d_q = Q.Read();
|
||||
auto d_p = P.ReadWrite();
|
||||
auto d_e = E.Read();
|
||||
auto d_b = B.Read();
|
||||
|
||||
mfem::forall(N, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
const real_t m = d_m[i];
|
||||
const real_t q = d_q[i];
|
||||
|
||||
real_t x[3], p[3], e[3], b[3];
|
||||
// Load data
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
x[d] = d_x[byVDIM_X ? i * dim + d : i + d * N];
|
||||
p[d] = d_p[byVDIM_P ? i * dim + d : i + d * N];
|
||||
e[d] = d_e[byVDIM_E ? i * dim + d : i + d * N];
|
||||
b[d] = d_b[byVDIM_B ? i * dim + d : i + d * N];
|
||||
}
|
||||
|
||||
// Boris algorithm implementation
|
||||
real_t pm[3], pxB[3], pp[3];
|
||||
|
||||
// Compute half of the contribution from q E
|
||||
// pm = p + 0.5 * dt * q * e
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
pm[d] = p[d] + (0.5 * dt * q) * e[d];
|
||||
}
|
||||
|
||||
// Compute the contribution from q p x B
|
||||
real_t B2 = 0.0;
|
||||
for (int d = 0; d < dim; d++) { B2 += b[d] * b[d]; }
|
||||
|
||||
// ... along pm x B
|
||||
// pxB = pm x b
|
||||
pxB[0] = pm[1] * b[2] - pm[2] * b[1];
|
||||
pxB[1] = pm[2] * b[0] - pm[0] * b[2];
|
||||
pxB[2] = pm[0] * b[1] - pm[1] * b[0];
|
||||
|
||||
// pp = a1 * pxB
|
||||
const real_t a1 = 4.0 * dt * q * m;
|
||||
for (int d = 0; d < dim; d++) { pp[d] = a1 * pxB[d]; }
|
||||
|
||||
// ... along pm
|
||||
// pp += a2 * pm
|
||||
const real_t a2 = 4.0 * m * m - dt * dt * q * q * B2;
|
||||
for (int d = 0; d < dim; d++) { pp[d] += a2 * pm[d]; }
|
||||
|
||||
// ... along B
|
||||
real_t b_dot_pm = 0.0;
|
||||
for (int d = 0; d < dim; d++) { b_dot_pm += b[d] * pm[d]; }
|
||||
const real_t a3 = 2.0 * dt * dt * q * q * b_dot_pm;
|
||||
// pp += a3 * b
|
||||
for (int d = 0; d < dim; d++) { pp[d] += a3 * b[d]; }
|
||||
|
||||
// scale by common denominator
|
||||
const real_t a4 = 4.0 * m * m + dt * dt * q * q * B2;
|
||||
for (int d = 0; d < dim; d++) { pp[d] /= a4; }
|
||||
|
||||
// Update the momentum
|
||||
// p = pp + 0.5 * dt * q * e
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
p[d] = pp[d] + (0.5 * dt * q) * e[d];
|
||||
}
|
||||
|
||||
// Update the position
|
||||
// x += (dt / m) * p
|
||||
// Store back to global arrays
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
d_p[byVDIM_P ? i * dim + d : i + d * N] = p[d];
|
||||
d_x[byVDIM_X ? i * dim + d : i + d * N] = x[d] + (dt / m) * p[d];
|
||||
}
|
||||
});
|
||||
|
||||
// Find updated particle locations in E and B field meshes
|
||||
FindParticles();
|
||||
|
||||
// Update time
|
||||
t += dt;
|
||||
}
|
||||
|
||||
Array<int> Boris::RemoveLostParticles()
|
||||
{
|
||||
Array<int> lost_idxs;
|
||||
@@ -617,6 +747,11 @@ void InitializeChargedParticles(ParticleSet &charged_particles,
|
||||
ParticleVector &M = charged_particles.Field(Boris::MASS);
|
||||
ParticleVector &Q = charged_particles.Field(Boris::CHARGE);
|
||||
|
||||
X.HostWrite();
|
||||
P.HostWrite();
|
||||
M.HostWrite();
|
||||
Q.HostWrite();
|
||||
|
||||
for (int i = 0; i < charged_particles.GetNParticles(); i++)
|
||||
{
|
||||
for (int d = 0; d < dim; d++)
|
||||
@@ -643,4 +778,9 @@ void InitializeChargedParticles(ParticleSet &charged_particles,
|
||||
M(i) = m;
|
||||
Q(i) = q;
|
||||
}
|
||||
|
||||
X.Read();
|
||||
P.Read();
|
||||
M.Read();
|
||||
Q.Read();
|
||||
}
|
||||
|
||||
@@ -49,6 +49,9 @@
|
||||
// findpts -m ../../data/ref-square.mesh -o 2 -mo 1 -random 1 -surf
|
||||
// findpts -m ../../data/ref-cube.mesh -o 2 -mo 1 -random 1 -surf
|
||||
// findpts -m ../../data/square-disc-p2.mesh -o 4 -mo 2 -random 1 -surf
|
||||
// Surface meshes + bounding box size increase:
|
||||
// findpts -m ../../data/square-disc-p2.mesh -o 4 -mo 2 -random 1 -surf -sabs 0.1
|
||||
// findpts -m ../../data/tinyzoo-3d.mesh -o 4 -mo 2 -random 1 -surf -sabs 0.1
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "../common/mfem-common.hpp"
|
||||
@@ -109,6 +112,7 @@ int main (int argc, char *argv[])
|
||||
int randomization = 0;
|
||||
int npt = 100;
|
||||
bool surface = false;
|
||||
double surf_aabb_sz_inc = 0.0;
|
||||
|
||||
// Parse command-line options.
|
||||
OptionsParser args(argc, argv);
|
||||
@@ -150,6 +154,9 @@ int main (int argc, char *argv[])
|
||||
args.AddOption(&surface, "-surf", "--surface", "-no-surf",
|
||||
"--no-surface",
|
||||
"Extract surface mesh from volume mesh.");
|
||||
args.AddOption(&surf_aabb_sz_inc, "-sabs", "--surface-aabb-size-inc",
|
||||
"Absolute AABB expansion applied to surface-search "
|
||||
"axis-aligned bounding boxes in FindPointsGSLIB surface meshes.");
|
||||
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
@@ -384,8 +391,17 @@ int main (int argc, char *argv[])
|
||||
|
||||
// Find and Interpolate FE function values on the desired points.
|
||||
Vector interp_vals(pts_cnt*vec_dim);
|
||||
FindPointsGSLIB finder(*mesh);
|
||||
finder.SetDistanceToleranceForPointsFoundOnBoundary(10);
|
||||
FindPointsGSLIB finder;
|
||||
if (surface && surf_aabb_sz_inc > 0.0)
|
||||
{
|
||||
Vector bb_size({surf_aabb_sz_inc});
|
||||
finder.SetupSurfWithAABBExpansion(*mesh, bb_size);
|
||||
}
|
||||
else
|
||||
{
|
||||
finder.Setup(*mesh);
|
||||
// finder.SetDistanceToleranceForPointsFoundOnBoundary(10);
|
||||
}
|
||||
finder.SetL2AvgType(FindPointsGSLIB::NONE);
|
||||
finder.Interpolate(vxyz, field_vals, interp_vals, point_ordering);
|
||||
Array<unsigned int> code_out = finder.GetCode();
|
||||
@@ -424,7 +440,7 @@ int main (int argc, char *argv[])
|
||||
<< "Searched points: " << pts_cnt
|
||||
<< "\nFound points: " << found
|
||||
<< "\nMax interp error: " << max_err
|
||||
<< "\nMax dist (of found): " << max_dist
|
||||
<< "\nMax dist^2 (of found): " << max_dist
|
||||
<< "\nPoints not found: " << not_found;
|
||||
if (randomization == 1)
|
||||
{
|
||||
|
||||
@@ -48,11 +48,14 @@
|
||||
// Device runs:
|
||||
// mpirun -np 2 pfindpts -m ../../data/inline-quad.mesh -o 3 -mo 2 -random 1 -d debug
|
||||
// mpirun -np 2 pfindpts -m ../../data/amr-quad.mesh -rs 1 -o 4 -mo 2 -random 1 -npt 100 -d debug
|
||||
// mpirun -np 2 pfindpts -m ../../data/inline-hex.mesh -o 3 -mo 2 -random 1 -d debug
|
||||
// mpirun -np 2 pfindpts -m ../../data/inline-hex.mesh -o 3 -mo 2 -random 1 -d debug -ft 1
|
||||
// Surface meshes:
|
||||
// mpirun -np 4 pfindpts -m ../../data/square-disc-p2.mesh -o 4 -mo 2 -vis -random 1 -surf
|
||||
// mpirun -np 4 pfindpts -m ../../data/star-q3.mesh -o 6 -mo 3 -vis -random 1 -surf
|
||||
// mpirun -np 4 pfindpts -m ../../data/fichera-q2.mesh -o 6 -mo 3 -vis -random 1 -surf
|
||||
// Surface meshes + bounding box size increase:
|
||||
// mpirun -np 4 pfindpts -m ../../data/square-disc-p2.mesh -o 4 -mo 2 -vis -random 1 -surf -sabs 0.1
|
||||
// mpirun -np 4 pfindpts -m ../../data/tinyzoo-3d.mesh -o 4 -mo 2 -vis -random 1 -surf -sabs 0.1
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "../common/mfem-common.hpp"
|
||||
@@ -102,6 +105,7 @@ int main (int argc, char *argv[])
|
||||
int randomization = 0;
|
||||
int npt = 100; //points per proc
|
||||
bool surface = false;
|
||||
double surf_aabb_sz_inc = 0.0;
|
||||
|
||||
// Parse command-line options.
|
||||
OptionsParser args(argc, argv);
|
||||
@@ -145,7 +149,9 @@ int main (int argc, char *argv[])
|
||||
args.AddOption(&surface, "-surf", "--surface", "-no-surf",
|
||||
"--no-surface",
|
||||
"Extract surface mesh from volume mesh.");
|
||||
|
||||
args.AddOption(&surf_aabb_sz_inc, "-sabs", "--surface-aabb-size-inc",
|
||||
"Absolute AABB expansion applied to surface-search "
|
||||
"axis-aligned bounding boxes in FindPointsGSLIB surface meshes.");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
@@ -343,7 +349,7 @@ int main (int argc, char *argv[])
|
||||
Geometry::GetRandomPoint(geom, ip);
|
||||
if (j < npt_face_per_elem)
|
||||
{
|
||||
ip.x = 0.0; // force point to be on the face
|
||||
ip.x = 0.0; // force point to be on a face
|
||||
npt_total_face++;
|
||||
}
|
||||
Vector pos_i(sdim);
|
||||
@@ -373,8 +379,17 @@ int main (int argc, char *argv[])
|
||||
|
||||
// Find and Interpolate FE function values on the desired points.
|
||||
Vector interp_vals(pts_cnt*vec_dim);
|
||||
FindPointsGSLIB finder(pmesh);
|
||||
finder.SetDistanceToleranceForPointsFoundOnBoundary(10);
|
||||
FindPointsGSLIB finder;
|
||||
if (surface && surf_aabb_sz_inc > 0.0)
|
||||
{
|
||||
Vector bb_size({surf_aabb_sz_inc});
|
||||
finder.SetupSurfWithAABBExpansion(pmesh, bb_size);
|
||||
}
|
||||
else
|
||||
{
|
||||
finder.Setup(pmesh);
|
||||
}
|
||||
// finder.SetDistanceToleranceForPointsFoundOnBoundary(1e-10);
|
||||
// Enable GPU to CPU fallback for GPUData only if you are using an older
|
||||
// version of GSLIB.
|
||||
// finder.SetGPUtoCPUFallback(true);
|
||||
@@ -456,10 +471,11 @@ int main (int argc, char *argv[])
|
||||
<< "\nPoints on faces: " << face_pts << " out of "
|
||||
<< npt_total_face
|
||||
<< "\nMax interp error: " << max_error
|
||||
<< "\nMax dist (of found): " << max_dist
|
||||
<< "\nMax dist^2 (of found): " << max_dist
|
||||
<< endl;
|
||||
}
|
||||
|
||||
|
||||
delete fec;
|
||||
|
||||
if (randomization != 0)
|
||||
|
||||
@@ -0,0 +1,204 @@
|
||||
#include "mfem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class StackedOperator : public Operator
|
||||
{
|
||||
public:
|
||||
StackedOperator(int m=0): Operator(0, m), offset{0} {}
|
||||
|
||||
virtual int AddOperator(Operator &op)
|
||||
{
|
||||
MFEM_VERIFY(!finalized, "Operator is finalized");
|
||||
MFEM_VERIFY(op.Width() == width, "Operator width inconsistent");
|
||||
offset.Append(op.Height());
|
||||
ops.Append(&op);
|
||||
return ops.Size()-1;
|
||||
}
|
||||
|
||||
void Finalize()
|
||||
{
|
||||
MFEM_VERIFY(!finalized, "Operator already been finalized");
|
||||
offset.PartialSum();
|
||||
Array<int> col_offset({0, width});
|
||||
blk_op.reset(new BlockOperator(offset, col_offset));
|
||||
for (int i=0; i<ops.Size(); i++)
|
||||
{
|
||||
blk_op->SetBlock(i, 0, ops[i]);
|
||||
}
|
||||
}
|
||||
|
||||
bool IsFinalized() const { return finalized; }
|
||||
|
||||
BlockOperator &AsBlockOperator() const
|
||||
{
|
||||
MFEM_VERIFY(finalized, "Operator not finalized");
|
||||
return *blk_op;
|
||||
}
|
||||
|
||||
void Mult(const Vector &x, Vector &y) const override
|
||||
{
|
||||
MFEM_VERIFY(finalized, "Operator not finalized");
|
||||
blk_op->Mult(x, y);
|
||||
}
|
||||
|
||||
Operator &GetGradient(const Vector &x) const override
|
||||
{
|
||||
MFEM_VERIFY(finalized, "Operator not finalized");
|
||||
if (!grad_op) { grad_op.reset(new ProblemGradient(*this)); }
|
||||
grad_op->SetPoint(x);
|
||||
return *grad_op;
|
||||
}
|
||||
|
||||
Operator &GetGradient(const int i, const Vector &x) const
|
||||
{
|
||||
MFEM_VERIFY(finalized, "Operator not finalized");
|
||||
return ops[i]->GetGradient(x);
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
class ProblemGradient : public Operator
|
||||
{
|
||||
public:
|
||||
ProblemGradient(const StackedOperator &prob)
|
||||
: Operator(prob.Width(), prob.Width())
|
||||
, prob(prob)
|
||||
{}
|
||||
void SetPoint(const Vector &x) { x_ = x; }
|
||||
void Mult(const Vector &x, Vector &y) const override
|
||||
{
|
||||
//
|
||||
}
|
||||
private:
|
||||
const StackedOperator &prob;
|
||||
Vector x_;
|
||||
};
|
||||
|
||||
protected:
|
||||
bool finalized = false;
|
||||
Array<int> offset;
|
||||
Array<Operator *> ops;
|
||||
std::unique_ptr<BlockOperator> blk_op;
|
||||
mutable std::unique_ptr<ProblemGradient> grad_op;
|
||||
};
|
||||
|
||||
class OptimProblem : public StackedOperator
|
||||
{
|
||||
enum class ConstType
|
||||
{
|
||||
EQ, // equality constraint
|
||||
LE, // less than or equal constraint
|
||||
};
|
||||
|
||||
int AddOperator(Operator &op) override
|
||||
{
|
||||
MFEM_ABORT("Use SetObjective or AddConstraint to add operators to the optimization problem");
|
||||
return -1;
|
||||
}
|
||||
|
||||
int SetObjective(Operator &obj, int obj_idx=0)
|
||||
{
|
||||
MFEM_VERIFY(!finalized, "Operator is finalized");
|
||||
MFEM_VERIFY(obj_blk_idx == -1, "Objective already set");
|
||||
MFEM_VERIFY(obj_idx >= 0, "Objective index must be non-negative");
|
||||
MFEM_VERIFY(obj_idx < ops.Size(), "Objective index out of bounds");
|
||||
obj_loc_idx = obj_idx;
|
||||
obj_blk_idx = StackedOperator::AddOperator(obj);
|
||||
return obj_blk_idx;
|
||||
}
|
||||
|
||||
int AddConstraint(Operator &con, ConstType type, int con_idx=0)
|
||||
{
|
||||
MFEM_VERIFY(!finalized, "Operator is finalized");
|
||||
MFEM_VERIFY(con_idx >= 0, "Constraint index must be non-negative");
|
||||
MFEM_VERIFY(con_idx < ops.Size(), "Constraint index out of bounds");
|
||||
constraint_types.Append(type);
|
||||
return StackedOperator::AddOperator(con);
|
||||
}
|
||||
|
||||
void UpdateObjectiveIndex(int obj_block, int obj_loc_idx_=0)
|
||||
{
|
||||
MFEM_VERIFY(obj_block >= 0 && obj_block < ops.Size(),
|
||||
"Objective block index out of bounds");
|
||||
MFEM_VERIFY(obj_loc_idx_ >= 0, "Objective index must be non-negative");
|
||||
MFEM_VERIFY(obj_loc_idx_ < ops[obj_block]->Height(),
|
||||
"Objective index out of bounds");
|
||||
obj_blk_idx = obj_block;
|
||||
obj_loc_idx = obj_loc_idx_;
|
||||
}
|
||||
|
||||
real_t GetEnergy(const Vector &x) const
|
||||
{
|
||||
MFEM_VERIFY(finalized, "Operator not finalized");
|
||||
aux_y.SetSize(ops[obj_blk_idx]->Height());
|
||||
ops[obj_blk_idx]->Mult(x, aux_y);
|
||||
return aux_y(obj_loc_idx);
|
||||
}
|
||||
|
||||
real_t Objective(const Vector &x) const
|
||||
{
|
||||
return GetEnergy(x);
|
||||
}
|
||||
|
||||
void Mult(const Vector &x, Vector &y) const override
|
||||
{
|
||||
MFEM_VERIFY(finalized, "Operator not finalized");
|
||||
blk_op->Mult(x, y);
|
||||
}
|
||||
|
||||
ConstType GetConstraintType(int con_block) const
|
||||
{
|
||||
MFEM_VERIFY(con_block >= 0 && con_block < constraint_types.Size(),
|
||||
"Constraint block index out of bounds");
|
||||
return constraint_types[con_block];
|
||||
}
|
||||
|
||||
// @brief Set the lower bound for the optimization variables (dof)
|
||||
// @param lb The lower bound vector (will be copied)
|
||||
void SetDofLowerBound(const Vector &lb)
|
||||
{
|
||||
MFEM_VERIFY(lb.Size() == width, "Lower bound size mismatch");
|
||||
dof_lb.UseDevice(true);
|
||||
dof_lb.SetSize(width);
|
||||
dof_lb = lb;
|
||||
}
|
||||
|
||||
// @brief Set the upper bound for the optimization variables (dof)
|
||||
// @param ub The upper bound vector (will be copied)
|
||||
void SetDofUpperBound(const Vector &ub)
|
||||
{
|
||||
MFEM_VERIFY(ub.Size() == width, "Upper bound size mismatch");
|
||||
dof_ub.SetSize(width);
|
||||
dof_ub = ub;
|
||||
}
|
||||
|
||||
// @brief Set the upper and lower bounds for the optimization variables (dof)
|
||||
// @param lb The lower bound vector (will be copied)
|
||||
// @param ub The upper bound vector (will be copied)
|
||||
void SetDofBounds(const Vector &lb, const Vector &ub)
|
||||
{
|
||||
MFEM_VERIFY(lb.Size() == width, "Lower bound size mismatch");
|
||||
MFEM_VERIFY(ub.Size() == width, "Upper bound size mismatch");
|
||||
dof_lb.SetSize(width);
|
||||
dof_lb = lb;
|
||||
dof_ub.SetSize(width);
|
||||
dof_ub = ub;
|
||||
}
|
||||
bool HasDofLowerBound() const { return dof_lb.Size() > 0; }
|
||||
bool HasDofUpperBound() const { return dof_ub.Size() > 0; }
|
||||
bool HasDofBounds() const { return dof_lb.Size() > 0 && dof_ub.Size() > 0; }
|
||||
|
||||
private:
|
||||
Array<ConstType> constraint_types;
|
||||
int obj_blk_idx = -1;
|
||||
int obj_loc_idx = -1;
|
||||
mutable Vector aux_y;
|
||||
|
||||
Vector dof_lb;
|
||||
Vector dof_ub;
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
@@ -83,7 +83,7 @@ real_t IntegrateBC(const ParGridFunction &x, const Array<int> &bdr,
|
||||
/// where A is
|
||||
/// A = div ( Theta(x) grad + Id ) u(x)
|
||||
/// and alpha is given as
|
||||
/// alpha = (2 nu + dim) / 2.
|
||||
/// alpha = (2 nu + dim) / 4.
|
||||
/// Theta (anisotropy tensor) and nu (smoothness) can be specified in the
|
||||
/// constructor. Traditionally, the SPDE method requires the specification of
|
||||
/// a white noise right hands side. SPDESolver accepts arbitrary right hand
|
||||
|
||||
@@ -12,6 +12,8 @@
|
||||
#include "unit_tests.hpp"
|
||||
#include "mfem.hpp"
|
||||
|
||||
#include <random>
|
||||
|
||||
using namespace mfem;
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
namespace gslib_test
|
||||
@@ -36,6 +38,165 @@ void F_exact(const Vector &p, Vector &F)
|
||||
|
||||
enum class Space { H1, L2 };
|
||||
|
||||
enum class SurfaceMeshType { Segment2D, Segment3D, Quad3D, Tri3D };
|
||||
|
||||
const char *SurfaceMeshName(const SurfaceMeshType type)
|
||||
{
|
||||
switch (type)
|
||||
{
|
||||
case SurfaceMeshType::Segment2D: return "segment-2d";
|
||||
case SurfaceMeshType::Segment3D: return "segment-3d";
|
||||
case SurfaceMeshType::Quad3D: return "quad-3d";
|
||||
case SurfaceMeshType::Tri3D: return "tri-3d";
|
||||
}
|
||||
return "unknown";
|
||||
}
|
||||
|
||||
int SurfaceSpaceDim(const SurfaceMeshType type)
|
||||
{
|
||||
switch (type)
|
||||
{
|
||||
case SurfaceMeshType::Segment2D: return 2;
|
||||
case SurfaceMeshType::Segment3D: return 3;
|
||||
case SurfaceMeshType::Quad3D: return 3;
|
||||
case SurfaceMeshType::Tri3D: return 3;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
Mesh MakeSurfaceMesh(const SurfaceMeshType type, const int ne)
|
||||
{
|
||||
switch (type)
|
||||
{
|
||||
case SurfaceMeshType::Segment2D:
|
||||
return Mesh::MakeCartesian1D(ne);
|
||||
case SurfaceMeshType::Segment3D:
|
||||
return Mesh::MakeCartesian1D(ne);
|
||||
case SurfaceMeshType::Quad3D:
|
||||
return Mesh::MakeCartesian2D(ne, ne, Element::QUADRILATERAL);
|
||||
case SurfaceMeshType::Tri3D:
|
||||
return Mesh::MakeCartesian2D(ne, ne, Element::TRIANGLE);
|
||||
}
|
||||
MFEM_ABORT("Unknown surface mesh type.");
|
||||
return Mesh();
|
||||
}
|
||||
|
||||
void GetSurfaceInteriorPoints(Mesh &mesh, const int npt_per_el,
|
||||
const int ordering, Vector &xyz,
|
||||
const int p0 = 0)
|
||||
{
|
||||
MFEM_VERIFY(mesh.GetNodes() != nullptr, "Mesh nodes are required.");
|
||||
|
||||
const int sdim = mesh.SpaceDimension();
|
||||
const int npt = xyz.Size()/sdim;
|
||||
MFEM_VERIFY((p0 + mesh.GetNE()*npt_per_el)*sdim <= xyz.Size(),
|
||||
"Output vector is too small.");
|
||||
Vector point(sdim);
|
||||
std::mt19937 gen(123);
|
||||
std::uniform_real_distribution<double> uni(0.01, 0.99);
|
||||
int p = p0;
|
||||
|
||||
for (int e = 0; e < mesh.GetNE(); e++)
|
||||
{
|
||||
ElementTransformation *T = mesh.GetElementTransformation(e);
|
||||
const Geometry::Type geom = mesh.GetElementBaseGeometry(e);
|
||||
for (int j = 0; j < npt_per_el; j++)
|
||||
{
|
||||
IntegrationPoint ip;
|
||||
real_t xv = uni(gen);
|
||||
if (geom == Geometry::SEGMENT)
|
||||
{
|
||||
ip.x = xv;
|
||||
}
|
||||
else if (geom == Geometry::SQUARE)
|
||||
{
|
||||
ip.Set2(xv, uni(gen));
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_VERIFY(geom == Geometry::TRIANGLE,
|
||||
"Unsupported surface element geometry.");
|
||||
ip.Set2(xv, uni(gen)*(1.0 - xv));
|
||||
}
|
||||
T->Transform(ip, point);
|
||||
for (int d = 0; d < sdim; d++)
|
||||
{
|
||||
const int idx = (ordering == Ordering::byNODES) ?
|
||||
d*npt + p :
|
||||
p*sdim + d;
|
||||
xyz(idx) = point(d);
|
||||
}
|
||||
p++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void GetSurfaceBoundaryPoints(Mesh &mesh, const int npt_per_el,
|
||||
const int ordering, Vector &xyz,
|
||||
const int p0 = 0)
|
||||
{
|
||||
MFEM_VERIFY(mesh.GetNodes() != nullptr, "Mesh nodes are required.");
|
||||
|
||||
const int sdim = mesh.SpaceDimension();
|
||||
const int npt = xyz.Size()/sdim;
|
||||
MFEM_VERIFY((p0 + mesh.GetNE()*npt_per_el)*sdim <= xyz.Size(),
|
||||
"Output vector is too small.");
|
||||
Vector point(sdim);
|
||||
std::mt19937 gen(246);
|
||||
std::uniform_real_distribution<double> uni(0.01, 0.99);
|
||||
int p = p0;
|
||||
|
||||
for (int e = 0; e < mesh.GetNE(); e++)
|
||||
{
|
||||
ElementTransformation *T = mesh.GetElementTransformation(e);
|
||||
const Geometry::Type geom = mesh.GetElementBaseGeometry(e);
|
||||
for (int j = 0; j < npt_per_el; j++)
|
||||
{
|
||||
IntegrationPoint ip;
|
||||
if (geom == Geometry::SEGMENT)
|
||||
{
|
||||
MFEM_VERIFY(npt_per_el == 2,
|
||||
"Segment boundary sampling requires npt_per_el = 2.");
|
||||
ip.x = (j == 0) ? 0.0 : 1.0;
|
||||
}
|
||||
else
|
||||
{
|
||||
const double t = uni(gen);
|
||||
if (geom == Geometry::SQUARE)
|
||||
{
|
||||
switch (j % 4)
|
||||
{
|
||||
case 0: ip.Set2(t, 0.0); break;
|
||||
case 1: ip.Set2(1.0, t); break;
|
||||
case 2: ip.Set2(t, 1.0); break;
|
||||
case 3: ip.Set2(0.0, t); break;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_VERIFY(geom == Geometry::TRIANGLE,
|
||||
"Unsupported surface element geometry.");
|
||||
switch (j % 3)
|
||||
{
|
||||
case 0: ip.Set2(t, 0.0); break;
|
||||
case 1: ip.Set2(t, 1.0 - t); break;
|
||||
case 2: ip.Set2(0.0, t); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
T->Transform(ip, point);
|
||||
for (int d = 0; d < sdim; d++)
|
||||
{
|
||||
const int idx = (ordering == Ordering::byNODES) ?
|
||||
d*npt + p :
|
||||
p*sdim + d;
|
||||
xyz(idx) = point(d);
|
||||
}
|
||||
p++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("GSLIBInterpolate", "[GSLIBInterpolate][GSLIB]")
|
||||
{
|
||||
auto space = GENERATE(Space::H1, Space::L2);
|
||||
@@ -190,6 +351,92 @@ TEST_CASE("GSLIBInterpolate", "[GSLIBInterpolate][GSLIB]")
|
||||
delete c_fec;
|
||||
}
|
||||
|
||||
TEST_CASE("GSLIBSurfInterpolate", "[GSLIBSurfInterpolate][GSLIB]")
|
||||
{
|
||||
auto surface_mesh_type = GENERATE(SurfaceMeshType::Segment2D,
|
||||
SurfaceMeshType::Segment3D,
|
||||
SurfaceMeshType::Quad3D,
|
||||
SurfaceMeshType::Tri3D);
|
||||
func_order = GENERATE(1, 2);
|
||||
int mesh_order = GENERATE(1, 2);
|
||||
int mesh_node_ordering = GENERATE(0, 1);
|
||||
int point_ordering = GENERATE(0, 1);
|
||||
int ncomp = GENERATE(1, 2);
|
||||
int gf_ordering = GENERATE(0, 1);
|
||||
int func_out_ordering = GENERATE(0, 1);
|
||||
|
||||
const char *mesh_name = SurfaceMeshName(surface_mesh_type);
|
||||
CAPTURE(mesh_name, func_order, mesh_order, mesh_node_ordering,
|
||||
point_ordering, ncomp, gf_ordering, func_out_ordering);
|
||||
|
||||
if (ncomp == 1 && gf_ordering == 1)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
Mesh mesh = MakeSurfaceMesh(surface_mesh_type, 4);
|
||||
const int sdim = SurfaceSpaceDim(surface_mesh_type);
|
||||
mesh.SetCurvature(mesh_order, false, sdim, mesh_node_ordering);
|
||||
|
||||
H1_FECollection c_fec(func_order, mesh.Dimension());
|
||||
FiniteElementSpace c_fespace(&mesh, &c_fec, ncomp, gf_ordering);
|
||||
GridFunction field_vals(&c_fespace);
|
||||
|
||||
VectorFunctionCoefficient F(ncomp, F_exact);
|
||||
field_vals.ProjectCoefficient(F);
|
||||
|
||||
const int npt_per_el = 8;
|
||||
const int pts_cnt = mesh.GetNE()*npt_per_el;
|
||||
Vector vxyz(pts_cnt*sdim);
|
||||
GetSurfaceInteriorPoints(mesh, npt_per_el, point_ordering, vxyz);
|
||||
|
||||
Vector interp_vals(pts_cnt*ncomp);
|
||||
FindPointsGSLIB finder;
|
||||
finder.SetupSurf(mesh);
|
||||
finder.SetL2AvgType(FindPointsGSLIB::NONE);
|
||||
finder.Interpolate(vxyz, field_vals, interp_vals, point_ordering,
|
||||
func_out_ordering);
|
||||
|
||||
Array<unsigned int> code_out = finder.GetCode();
|
||||
Vector dist_p_out = finder.GetDist();
|
||||
|
||||
int not_found = 0;
|
||||
double err = 0.0, max_err = 0.0, max_dist = 0.0;
|
||||
Vector pos(sdim);
|
||||
Vector exact_val(ncomp);
|
||||
|
||||
for (int i = 0; i < pts_cnt; i++)
|
||||
{
|
||||
max_dist = std::max(max_dist, dist_p_out(i));
|
||||
for (int d = 0; d < sdim; d++)
|
||||
{
|
||||
const int idx = (point_ordering == Ordering::byNODES) ?
|
||||
d*pts_cnt + i :
|
||||
i*sdim + d;
|
||||
pos(d) = vxyz(idx);
|
||||
}
|
||||
F_exact(pos, exact_val);
|
||||
for (int j = 0; j < ncomp; j++)
|
||||
{
|
||||
if (code_out[i] < 2)
|
||||
{
|
||||
err = func_out_ordering == Ordering::byNODES ?
|
||||
fabs(exact_val(j) - interp_vals[i + j*pts_cnt]) :
|
||||
fabs(exact_val(j) - interp_vals[i*ncomp + j]);
|
||||
max_err = std::max(max_err, err);
|
||||
}
|
||||
else if (j == 0)
|
||||
{
|
||||
not_found++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
REQUIRE(max_err < 1e-12);
|
||||
REQUIRE(max_dist < 1e-10);
|
||||
REQUIRE(not_found == 0);
|
||||
}
|
||||
|
||||
// Generates meshes with different element types, followed by points at
|
||||
// element faces and interior, and finally checks to see if these points are
|
||||
// correctly detected at element boundary or not.
|
||||
@@ -257,9 +504,8 @@ TEST_CASE("GSLIBFindAtElementBoundary",
|
||||
int nptface = xyz.Size()/dim;
|
||||
|
||||
// Generate points inside each element
|
||||
FiniteElementCollection *l2_fec = new L2_FECollection(l2_order, dim);
|
||||
FiniteElementSpace l2_fespace =
|
||||
FiniteElementSpace(&mesh, l2_fec, 1);
|
||||
L2_FECollection l2_fec(l2_order, dim);
|
||||
FiniteElementSpace l2_fespace(&mesh, &l2_fec, 1);
|
||||
DenseMatrix vals;
|
||||
DenseMatrix tr;
|
||||
for (int e = 0; e < mesh.GetNE(); e++)
|
||||
@@ -295,7 +541,92 @@ TEST_CASE("GSLIBFindAtElementBoundary",
|
||||
cmax = std::max(code_out[i], cmax);
|
||||
}
|
||||
REQUIRE((cmin == 0 && cmax == 0)); // should be found inside element
|
||||
delete l2_fec;
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("GSLIBSurfFindAtElementBoundary",
|
||||
"[GSLIBSurfFindAtElementBoundary][GSLIB]")
|
||||
{
|
||||
auto surface_mesh_type = GENERATE(SurfaceMeshType::Segment2D,
|
||||
SurfaceMeshType::Segment3D,
|
||||
SurfaceMeshType::Quad3D,
|
||||
SurfaceMeshType::Tri3D);
|
||||
const char *mesh_name = SurfaceMeshName(surface_mesh_type);
|
||||
CAPTURE(mesh_name);
|
||||
|
||||
Mesh mesh = MakeSurfaceMesh(surface_mesh_type, 4);
|
||||
const int sdim = SurfaceSpaceDim(surface_mesh_type);
|
||||
mesh.SetCurvature(2, false, sdim);
|
||||
|
||||
const int nptface_per_el = (mesh.Dimension() == 1) ? 2 : 8;
|
||||
const int nptint_per_el = 8;
|
||||
const int nptface = mesh.GetNE()*nptface_per_el;
|
||||
const int nptint = mesh.GetNE()*nptint_per_el;
|
||||
Vector xyz((nptface + nptint)*sdim);
|
||||
GetSurfaceBoundaryPoints(mesh, nptface_per_el, Ordering::byVDIM, xyz, 0);
|
||||
GetSurfaceInteriorPoints(mesh, nptint_per_el, Ordering::byVDIM, xyz,
|
||||
nptface);
|
||||
|
||||
FindPointsGSLIB finder;
|
||||
finder.SetupSurf(mesh);
|
||||
finder.FindPoints(xyz, Ordering::byVDIM);
|
||||
Array<unsigned int> code_out = finder.GetCode();
|
||||
|
||||
for (int i = 0; i < nptface; i++)
|
||||
{
|
||||
REQUIRE(code_out[i] == 1);
|
||||
}
|
||||
for (int i = nptface; i < nptface + nptint; i++)
|
||||
{
|
||||
REQUIRE(code_out[i] == 0);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("GSLIBSurfAABBExpansion", "[GSLIBSurfAABBExpansion][GSLIB]")
|
||||
{
|
||||
auto surface_mesh_type = GENERATE(SurfaceMeshType::Segment2D,
|
||||
SurfaceMeshType::Segment3D,
|
||||
SurfaceMeshType::Quad3D,
|
||||
SurfaceMeshType::Tri3D);
|
||||
const char *mesh_name = SurfaceMeshName(surface_mesh_type);
|
||||
CAPTURE(mesh_name);
|
||||
|
||||
constexpr double offset = 1.0e-3;
|
||||
const int npt_per_el = 8;
|
||||
|
||||
Mesh mesh = MakeSurfaceMesh(surface_mesh_type, 4);
|
||||
const int sdim = SurfaceSpaceDim(surface_mesh_type);
|
||||
mesh.SetCurvature(2, false, sdim);
|
||||
|
||||
const int npt = mesh.GetNE()*npt_per_el;
|
||||
Vector xyz(npt*sdim);
|
||||
GetSurfaceInteriorPoints(mesh, npt_per_el, Ordering::byVDIM, xyz);
|
||||
// offset them to move away from the surface
|
||||
const int off_d = (surface_mesh_type == SurfaceMeshType::Segment2D) ? 1 : 2;
|
||||
for (int i = 0; i < npt; i++)
|
||||
{
|
||||
xyz(i*sdim + off_d) += offset;
|
||||
}
|
||||
|
||||
FindPointsGSLIB finder;
|
||||
finder.SetupSurf(mesh, 0.0);
|
||||
finder.FindPoints(xyz, Ordering::byVDIM);
|
||||
Array<unsigned int> code_no_pad = finder.GetCode();
|
||||
|
||||
for (int i = 0; i < code_no_pad.Size(); i++)
|
||||
{
|
||||
REQUIRE(code_no_pad[i] == 2);
|
||||
}
|
||||
|
||||
// make aabb at least big enough to include the offset points
|
||||
Vector aabb_sz_inc({2.1*offset});
|
||||
finder.SetupSurfWithAABBExpansion(mesh, aabb_sz_inc);
|
||||
finder.FindPoints(xyz, Ordering::byVDIM);
|
||||
Array<unsigned int> code_with_pad = finder.GetCode();
|
||||
|
||||
for (int i = 0; i < npt; i++)
|
||||
{
|
||||
REQUIRE(code_with_pad[i] == 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -322,9 +653,8 @@ TEST_CASE("GSLIBInterpolateL2ElementBoundary",
|
||||
mesh.SetCurvature(mesh_order);
|
||||
|
||||
// Set GridFunction to be interpolated
|
||||
FiniteElementCollection *c_fec = new L2_FECollection(3, dim);
|
||||
FiniteElementSpace c_fespace =
|
||||
FiniteElementSpace(&mesh, c_fec, 1);
|
||||
L2_FECollection c_fec(3, dim);
|
||||
FiniteElementSpace c_fespace(&mesh, &c_fec, 1);
|
||||
GridFunction field_vals(&c_fespace);
|
||||
Array<int> dofs;
|
||||
double leftval = 1.0;
|
||||
@@ -366,7 +696,6 @@ TEST_CASE("GSLIBInterpolateL2ElementBoundary",
|
||||
REQUIRE(interp_vals(0) == MFEM_Approx(0.5*(leftval+rightval)));
|
||||
|
||||
finder.FreeData();
|
||||
delete c_fec;
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
@@ -328,7 +328,7 @@ TEST_CASE("Linear Form Extension", "[LinearFormExtension], [GPU]")
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("H(div) Linear Form Extension", "[LinearFormExtension], [GPU]")
|
||||
TEST_CASE("Vector FE Linear Form Extension", "[LinearFormExtension], [GPU]")
|
||||
{
|
||||
const bool all = launch_all_non_regression_tests;
|
||||
|
||||
@@ -341,26 +341,44 @@ TEST_CASE("H(div) Linear Form Extension", "[LinearFormExtension], [GPU]")
|
||||
Mesh mesh(mesh_file);
|
||||
const int dim = mesh.Dimension();
|
||||
|
||||
CAPTURE(mesh_file, dim, p);
|
||||
{
|
||||
const auto space_type =
|
||||
dim == 3 ? GENERATE(FiniteElement::DIV, FiniteElement::CURL)
|
||||
: FiniteElement::DIV;
|
||||
|
||||
RT_FECollection fec(p, dim);
|
||||
FiniteElementSpace fes(&mesh, &fec);
|
||||
CAPTURE(mesh_file, dim, p, space_type);
|
||||
|
||||
VectorFunctionCoefficient coeff(dim, fvec_dim);
|
||||
std::unique_ptr<FiniteElementCollection> fec;
|
||||
|
||||
LinearForm d1(&fes);
|
||||
d1.AddDomainIntegrator(new VectorFEDomainLFIntegrator(coeff));
|
||||
d1.UseFastAssembly(true);
|
||||
d1.Assemble();
|
||||
switch (space_type)
|
||||
{
|
||||
case FiniteElement::DIV:
|
||||
fec.reset(new RT_FECollection(p, dim));
|
||||
break;
|
||||
case FiniteElement::CURL:
|
||||
fec.reset(new ND_FECollection(p, dim));
|
||||
break;
|
||||
default:
|
||||
MFEM_ABORT("unsupported space type");
|
||||
}
|
||||
FiniteElementSpace fes(&mesh, fec.get());
|
||||
|
||||
LinearForm d2(&fes);
|
||||
d2.AddDomainIntegrator(new VectorFEDomainLFIntegrator(coeff));
|
||||
d2.UseFastAssembly(false);
|
||||
d2.Assemble();
|
||||
VectorFunctionCoefficient coeff(dim, fvec_dim);
|
||||
|
||||
CAPTURE(d1.Norml2(), d2.Norml2());
|
||||
d1 -= d2;
|
||||
REQUIRE(d1.Norml2() == MFEM_Approx(0.0));
|
||||
LinearForm d1(&fes);
|
||||
d1.AddDomainIntegrator(new VectorFEDomainLFIntegrator(coeff));
|
||||
d1.UseFastAssembly(true);
|
||||
d1.Assemble();
|
||||
|
||||
LinearForm d2(&fes);
|
||||
d2.AddDomainIntegrator(new VectorFEDomainLFIntegrator(coeff));
|
||||
d2.UseFastAssembly(false);
|
||||
d2.Assemble();
|
||||
|
||||
CAPTURE(d1.Norml2(), d2.Norml2());
|
||||
d1 -= d2;
|
||||
REQUIRE(d1.Norml2() == MFEM_Approx(0.0));
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
@@ -200,10 +200,71 @@ int CheckArrayEquality(const Array<T> &arr1, const Array<T> &arr2)
|
||||
return wrong_ct;
|
||||
}
|
||||
|
||||
// Apply a deterministic perturbation to particle data on host.
|
||||
void PerturbParticleDataOnHost(std::vector<Particle> &particles)
|
||||
{
|
||||
for (auto &p : particles)
|
||||
{
|
||||
for (int f = -1; f < p.GetNFields(); f++)
|
||||
{
|
||||
Vector &field = f == -1 ? p.Coords() : p.Field(f);
|
||||
field.HostReadWrite();
|
||||
const real_t scale = (f == -1) ? 0.001 : 1.0;
|
||||
for (int c = 0; c < field.Size(); c++)
|
||||
{
|
||||
field(c) += scale * (f + c + 2);
|
||||
}
|
||||
}
|
||||
|
||||
for (int t = 0; t < p.GetNTags(); t++)
|
||||
{
|
||||
p.Tag(t) += t + 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Apply a deterministic perturbation to particle data on device.
|
||||
void PerturbParticleDataOnDevice(ParticleSet &pset)
|
||||
{
|
||||
const int np = pset.GetNParticles();
|
||||
|
||||
// Shift coordinates and fields using the same per-component formula while
|
||||
// honoring the ParticleVector ordering selected by the test.
|
||||
for (int f = -1; f < pset.GetNFields(); f++)
|
||||
{
|
||||
ParticleVector &field = f == -1 ? pset.Coords() : pset.Field(f);
|
||||
const int vdim = field.GetVDim();
|
||||
const bool by_vdim = (field.GetOrdering() == Ordering::byVDIM);
|
||||
const real_t scale = (f == -1) ? 0.001 : 1.0;
|
||||
auto d_field = field.ReadWrite();
|
||||
|
||||
mfem::forall(np, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
const int idx = by_vdim ? i * vdim + c : i + c * np;
|
||||
d_field[idx] += scale * (f + c + 2);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
for (int t = 0; t < pset.GetNTags(); t++)
|
||||
{
|
||||
Array<int> &tag = pset.Tag(t);
|
||||
auto d_tag = tag.ReadWrite();
|
||||
|
||||
mfem::forall(np, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_tag[i] += t + 1;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
void TestRedistribute(Ordering::Type ordering)
|
||||
{
|
||||
int size = Mpi::WorldSize();
|
||||
int rank = Mpi::WorldRank();
|
||||
const bool use_device = Device::IsEnabled();
|
||||
|
||||
// Create a 3D hex mesh
|
||||
Mesh m = Mesh::MakeCartesian3D(N_e, N_e, N_e, Element::Type::HEXAHEDRON);
|
||||
@@ -252,15 +313,22 @@ void TestRedistribute(Ordering::Type ordering)
|
||||
SECTION(std::string("Ordering: ") +
|
||||
(ordering == Ordering::byNODES ? "byNODES" : "byVDIM"))
|
||||
{
|
||||
// Add the particles uniquely to each rank particleset
|
||||
ParticleSet pset(MPI_COMM_WORLD, 0, SpaceDim, FieldVDims,
|
||||
NumTags, ordering);
|
||||
NumTags, ordering, use_device);
|
||||
CHECK(pset.IsParticleRefValid() ==
|
||||
(!use_device && ordering == Ordering::byVDIM));
|
||||
|
||||
for (int i = 0; i < N_rank; i++)
|
||||
{
|
||||
pset.AddParticle(all_particles[i*size+rank]);
|
||||
}
|
||||
|
||||
if (use_device)
|
||||
{
|
||||
PerturbParticleDataOnDevice(pset);
|
||||
PerturbParticleDataOnHost(all_particles);
|
||||
}
|
||||
|
||||
// Find points
|
||||
FindPointsGSLIB finder(MPI_COMM_WORLD);
|
||||
finder.Setup(pmesh);
|
||||
@@ -270,6 +338,7 @@ void TestRedistribute(Ordering::Type ordering)
|
||||
int code_1_count = 0;
|
||||
int code_2_count = 0;
|
||||
const Array<unsigned int> &code = finder.GetCode();
|
||||
code.HostRead();
|
||||
for (int i = 0; i < code.Size(); i++)
|
||||
{
|
||||
if (code[i] == 1)
|
||||
@@ -292,6 +361,7 @@ void TestRedistribute(Ordering::Type ordering)
|
||||
finder.FindPoints(pset.Coords(), ordering);
|
||||
|
||||
const Array<unsigned int> &procs = finder.GetProc();
|
||||
procs.HostRead();
|
||||
|
||||
int wrong_proc_count = 0;
|
||||
for (int i = 0; i < procs.Size(); i++)
|
||||
@@ -307,6 +377,11 @@ void TestRedistribute(Ordering::Type ordering)
|
||||
|
||||
// Check that coordinates + fields + tags are all still correct
|
||||
int wrong_particle_count = 0;
|
||||
pset.GetIDs().HostRead();
|
||||
for (int t = 0; t < pset.GetNTags(); t++)
|
||||
{
|
||||
pset.Tag(t).HostRead();
|
||||
}
|
||||
for (int i = 0; i < pset.GetNParticles(); i++)
|
||||
{
|
||||
Particle &actual_p = all_particles[pset.GetIDs()[i]];
|
||||
@@ -317,13 +392,13 @@ void TestRedistribute(Ordering::Type ordering)
|
||||
wrong_particle_count++;
|
||||
}
|
||||
}
|
||||
MPI_Allreduce(MPI_IN_PLACE, &wrong_proc_count, 1, MPI_INT, MPI_SUM,
|
||||
MPI_Allreduce(MPI_IN_PLACE, &wrong_particle_count, 1, MPI_INT, MPI_SUM,
|
||||
MPI_COMM_WORLD);
|
||||
CHECK(wrong_particle_count == 0);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("Particle Redistribution", "[ParticleSet][Parallel]")
|
||||
TEST_CASE("Particle Redistribution", "[ParticleSet][Parallel][GPU]")
|
||||
{
|
||||
TestRedistribute(Ordering::byNODES);
|
||||
TestRedistribute(Ordering::byVDIM);
|
||||
|
||||
@@ -113,3 +113,65 @@ TEST_CASE("ComplexOperator Quaternion Tests", "[ComplexOperator]")
|
||||
REQUIRE(qikx.Normlinf() < tol);
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
TEST_CASE("ComplexHypreParMatrix GetSystemMatrix",
|
||||
"[ComplexOperator][Parallel][GPU]")
|
||||
{
|
||||
// This test reproduces the issue described in PR #5200 on GitHub. See also
|
||||
// the follow up PR #5346.
|
||||
|
||||
// 1. Construct ComplexHypreParMatrix similar to ex25p.
|
||||
const char mesh_file[] = "../../data/inline-quad.mesh";
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
int ref_levels = 1;
|
||||
for (int l = 0; l < ref_levels; l++)
|
||||
{
|
||||
mesh->UniformRefinement();
|
||||
}
|
||||
ParMesh pmesh(MPI_COMM_WORLD, *mesh);
|
||||
delete mesh;
|
||||
int par_ref_levels = 1;
|
||||
for (int l = 0; l < par_ref_levels; l++)
|
||||
{
|
||||
pmesh.UniformRefinement();
|
||||
}
|
||||
int order = 1;
|
||||
ND_FECollection fec(order, dim);
|
||||
ParFiniteElementSpace fespace(&pmesh, &fec);
|
||||
Array<int> ess_tdof_list;
|
||||
Array<int> ess_bdr;
|
||||
if (pmesh.bdr_attributes.Size())
|
||||
{
|
||||
ess_bdr.SetSize(pmesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
}
|
||||
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
ComplexOperator::Convention conv = ComplexOperator::HERMITIAN;
|
||||
VectorConstantCoefficient f(Vector{1_r, 2_r});
|
||||
ParComplexLinearForm b(&fespace, conv);
|
||||
b.AddDomainIntegrator(NULL, new VectorFEDomainLFIntegrator(f));
|
||||
b = 0.0;
|
||||
b.Assemble();
|
||||
ParComplexGridFunction x(&fespace);
|
||||
x = 0.0;
|
||||
ConstantCoefficient one(1_r);
|
||||
ParSesquilinearForm a(&fespace, conv);
|
||||
a.AddDomainIntegrator(new CurlCurlIntegrator(one),
|
||||
new CurlCurlIntegrator(one));
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(one),
|
||||
new VectorFEMassIntegrator(one));
|
||||
a.Assemble();
|
||||
OperatorPtr Ah;
|
||||
Vector B, X;
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
|
||||
|
||||
// 2. Test the call to ComplexHypreParMatrix::GetSystemMatrix and destroying
|
||||
// the returned matrix.
|
||||
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
delete A;
|
||||
}
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
Reference in New Issue
Block a user