Compare commits
398
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ae695de95a | ||
|
|
f9be12c539 | ||
|
|
519e86f134 | ||
|
|
f2578fd479 | ||
|
|
563cd25971 | ||
|
|
875c62ca1f | ||
|
|
731b0b7049 | ||
|
|
3c77ad25e9 | ||
|
|
09e73b5cee | ||
|
|
464b441409 | ||
|
|
8ae9b217f9 | ||
|
|
3f0a5ad501 | ||
|
|
dfb697b9ae | ||
|
|
cf01c6cc8b | ||
|
|
e4dc9a3941 | ||
|
|
6a104e94e1 | ||
|
|
bf1b25d82e | ||
|
|
9a8f0ad0ef | ||
|
|
ee2c069531 | ||
|
|
27a60a4ca3 | ||
|
|
b1d5104fed | ||
|
|
26eecf7575 | ||
|
|
9c26ef5401 | ||
|
|
26f8f0e955 | ||
|
|
dfc2dfeb52 | ||
|
|
7173477670 | ||
|
|
30249a4857 | ||
|
|
8a9bdf863a | ||
|
|
b72368c698 | ||
|
|
e6224e00aa | ||
|
|
7b82f7b8e0 | ||
|
|
582ffe8b66 | ||
|
|
aa5b2d4b95 | ||
|
|
23616a21f0 | ||
|
|
df9cc72e58 | ||
|
|
15740fcbd3 | ||
|
|
862c539276 | ||
|
|
7ff759a7ee | ||
|
|
d220a2975c | ||
|
|
cdde0046ef | ||
|
|
9f36ae566c | ||
|
|
4465fcbd46 | ||
|
|
807b13b236 | ||
|
|
70f4468f0b | ||
|
|
5db2e7b347 | ||
|
|
c1df332094 | ||
|
|
75b115cf93 | ||
|
|
59579f2cdb | ||
|
|
c0d2821d3e | ||
|
|
082e25ffe8 | ||
|
|
7d343e56a5 | ||
|
|
1eb3d185d5 | ||
|
|
a2fd029daa | ||
|
|
b48b990ac8 | ||
|
|
80c69e9856 | ||
|
|
dbdee73113 | ||
|
|
7a8a1976a2 | ||
|
|
6911ed0a8a | ||
|
|
986e2c04d2 | ||
|
|
9db5388362 | ||
|
|
dbf987fb73 | ||
|
|
c4a6d855eb | ||
|
|
80c2ee6fed | ||
|
|
68aa9af7c2 | ||
|
|
19fa34eabe | ||
|
|
359ba5983b | ||
|
|
947f2f215a | ||
|
|
52ced84b8c | ||
|
|
6430c45f52 | ||
|
|
6d3c7dfc88 | ||
|
|
0e2d49799c | ||
|
|
562b980e7d | ||
|
|
388b7b61b2 | ||
|
|
2adb4576d3 | ||
|
|
9f2490cd95 | ||
|
|
4126d32772 | ||
|
|
080780f047 | ||
|
|
ca2ee9ab05 | ||
|
|
0d87692863 | ||
|
|
450628554c | ||
|
|
7bc231eb85 | ||
|
|
6de39a2637 | ||
|
|
c4a19c8df0 | ||
|
|
280e98b510 | ||
|
|
9794243c8b | ||
|
|
333726ce51 | ||
|
|
d14d286842 | ||
|
|
ea3bed0a91 | ||
|
|
877937b8ae | ||
|
|
1e2b33e21c | ||
|
|
f65311b8a0 | ||
|
|
1254e8782e | ||
|
|
b50d546dac | ||
|
|
fd01ef188f | ||
|
|
f2a29326c0 | ||
|
|
1f4a40d9a0 | ||
|
|
db003bdf86 | ||
|
|
8f8e5712ad | ||
|
|
e671fa5737 | ||
|
|
ae33f831eb | ||
|
|
c0a133fe2c | ||
|
|
1ffb08db61 | ||
|
|
90e1f08bc7 | ||
|
|
37d004206a | ||
|
|
e1d9b7cfff | ||
|
|
d1c44a7369 | ||
|
|
2c39547bdd | ||
|
|
db66a4423f | ||
|
|
8425377bd1 | ||
|
|
ea76f8d477 | ||
|
|
992c5dfc4c | ||
|
|
f0027c2ccd | ||
|
|
142b1ec60a | ||
|
|
77f1682547 | ||
|
|
6ccb7dea8a | ||
|
|
4f35f73c69 | ||
|
|
97ac0c3a3e | ||
|
|
6cdc6cd026 | ||
|
|
13dc17123d | ||
|
|
ec94332c2a | ||
|
|
c6329b292e | ||
|
|
5e75bd0837 | ||
|
|
5f889293c2 | ||
|
|
afc5aa96dc | ||
|
|
9b1d29a762 | ||
|
|
b4120d81e4 | ||
|
|
9125bf8291 | ||
|
|
9d285278f9 | ||
|
|
b435fbeae6 | ||
|
|
5c356a3e2b | ||
|
|
9e764dc809 | ||
|
|
b7bddf91bc | ||
|
|
bdbd234c56 | ||
|
|
a26ef22a49 | ||
|
|
1f89bfa040 | ||
|
|
3dc4463c56 | ||
|
|
0b79aa138c | ||
|
|
c76acb2a7b | ||
|
|
8fb8b77b3d | ||
|
|
8f6626aede | ||
|
|
c3dcf554df | ||
|
|
8f52ddcf6f | ||
|
|
696ce1e8ed | ||
|
|
87c6d3a370 | ||
|
|
8f2b4bb539 | ||
|
|
6c032cb7cf | ||
|
|
8104dd3f04 | ||
|
|
bdd8696d22 | ||
|
|
032fb568be | ||
|
|
3d6ee379e3 | ||
|
|
2361e45683 | ||
|
|
e0c1018c9c | ||
|
|
7f263221dd | ||
|
|
f82e8c04f9 | ||
|
|
a328693444 | ||
|
|
3889095a00 | ||
|
|
826b089e8f | ||
|
|
1bee2fc017 | ||
|
|
c0a66cee74 | ||
|
|
9b94a6c225 | ||
|
|
f0c4c45e02 | ||
|
|
8f6912dc2a | ||
|
|
ea6053ef9d | ||
|
|
7fc1fb501a | ||
|
|
f6afbb8cee | ||
|
|
05d71b6093 | ||
|
|
c731dee195 | ||
|
|
64c78ff17e | ||
|
|
01444aa93c | ||
|
|
d6cb23ab2f | ||
|
|
2a65d49db3 | ||
|
|
712bee9cce | ||
|
|
34043e730b | ||
|
|
850c1813c3 | ||
|
|
84c08a69ee | ||
|
|
5496699870 | ||
|
|
6915a8c6f3 | ||
|
|
0950325080 | ||
|
|
03a6f1b190 | ||
|
|
07c0f31e37 | ||
|
|
4caf208e36 | ||
|
|
b050e307db | ||
|
|
b70589bcac | ||
|
|
17bbaf500c | ||
|
|
a1179d6489 | ||
|
|
5932fcd331 | ||
|
|
5cf82dc903 | ||
|
|
f19ec00b0a | ||
|
|
1339b6b99a | ||
|
|
9480e5c5bb | ||
|
|
2cfb86aade | ||
|
|
0735280f9d | ||
|
|
e7c041247f | ||
|
|
af00755d68 | ||
|
|
cc21011998 | ||
|
|
09aa9374a9 | ||
|
|
78e8cdd7e8 | ||
|
|
86e75206b1 | ||
|
|
04a3fd9bb2 | ||
|
|
72f0b668c8 | ||
|
|
aefeb46c48 | ||
|
|
6e111c3ada | ||
|
|
30a3d7c0d4 | ||
|
|
ea449e1c41 | ||
|
|
c9115e74fb | ||
|
|
35040b0336 | ||
|
|
4ce11b4a12 | ||
|
|
cb580207c8 | ||
|
|
cd59e68993 | ||
|
|
eb61aa4244 | ||
|
|
3659acd79d | ||
|
|
9d3047b3a7 | ||
|
|
f2ac9b5653 | ||
|
|
f17b1c4e4f | ||
|
|
7cfd3f5c7e | ||
|
|
63ea83c455 | ||
|
|
38ee46c40f | ||
|
|
c271283490 | ||
|
|
808f5c94db | ||
|
|
287631bee4 | ||
|
|
4da6e52698 | ||
|
|
cc12d37693 | ||
|
|
35fe98417c | ||
|
|
936f1fc848 | ||
|
|
2cadeaad4c | ||
|
|
f7fa092013 | ||
|
|
9103db88b6 | ||
|
|
0303f12887 | ||
|
|
6650a1dffe | ||
|
|
9a22d4533f | ||
|
|
412715c2e4 | ||
|
|
9803cbb671 | ||
|
|
6805b8c7f6 | ||
|
|
7aeba0ff83 | ||
|
|
dbd55a8fb4 | ||
|
|
277199c3a5 | ||
|
|
e7e00e6e39 | ||
|
|
e78c1b8b4c | ||
|
|
dcec51b98a | ||
|
|
2e9f545a4e | ||
|
|
e49854f3ba | ||
|
|
0bb71aa1fa | ||
|
|
07dc0f6cfa | ||
|
|
847183e668 | ||
|
|
8a1a264eaa | ||
|
|
606a597303 | ||
|
|
e194835abd | ||
|
|
208f9b81b3 | ||
|
|
e50ebb573e | ||
|
|
c315298a86 | ||
|
|
12bafa69e8 | ||
|
|
1729961a89 | ||
|
|
6dcc19abab | ||
|
|
a5fccc7514 | ||
|
|
cfe25c432c | ||
|
|
b8fe4cbf97 | ||
|
|
ea0abf46fe | ||
|
|
556b43f900 | ||
|
|
aa567465ac | ||
|
|
986cee600f | ||
|
|
7634a4b663 | ||
|
|
1368cfb8cb | ||
|
|
53c561cbf0 | ||
|
|
3c2e847e0a | ||
|
|
6daba6f9fa | ||
|
|
f65f60dbb5 | ||
|
|
969ba74440 | ||
|
|
68ae30ad6f | ||
|
|
e49a62ab1b | ||
|
|
e3b732761f | ||
|
|
44f8eb9990 | ||
|
|
18a256d17b | ||
|
|
a4d8700473 | ||
|
|
47b73713a6 | ||
|
|
43f4d9d6e3 | ||
|
|
081bbdd1c2 | ||
|
|
80b04009af | ||
|
|
7983d25a14 | ||
|
|
bf1a63c920 | ||
|
|
b64bb19cf8 | ||
|
|
321bcceb1c | ||
|
|
c3656a5573 | ||
|
|
f84d5d1bfa | ||
|
|
6ce9d7b3be | ||
|
|
02dd999886 | ||
|
|
5132b9191c | ||
|
|
e16cf9a89d | ||
|
|
dd3e76db6c | ||
|
|
8578bf4918 | ||
|
|
97e5cf0dc4 | ||
|
|
40083984c1 | ||
|
|
96cc518acf | ||
|
|
acb1bb4dc2 | ||
|
|
8f391e9854 | ||
|
|
dfb73f248f | ||
|
|
c55f3c13af | ||
|
|
8d40910fd8 | ||
|
|
b2e61e7e1c | ||
|
|
7c80866b49 | ||
|
|
7ec6704a51 | ||
|
|
62dcb4cf1a | ||
|
|
0f0d70dad1 | ||
|
|
3c1606903c | ||
|
|
80274347ae | ||
|
|
622985d8da | ||
|
|
754235e932 | ||
|
|
0cc04c8bd9 | ||
|
|
5f82901f73 | ||
|
|
41a262563b | ||
|
|
b081aa7eea | ||
|
|
3c7a63f742 | ||
|
|
ffe14214ae | ||
|
|
c515b75b53 | ||
|
|
ecab08e1b6 | ||
|
|
5602c66e1a | ||
|
|
900085cb9d | ||
|
|
0376760aa1 | ||
|
|
2ab9ae818e | ||
|
|
1035382fad | ||
|
|
90f33b1a86 | ||
|
|
c354e4cd27 | ||
|
|
01d92c7133 | ||
|
|
136ae2d98f | ||
|
|
12896cd9ed | ||
|
|
15ecbb5e6e | ||
|
|
b5ed99e1cd | ||
|
|
449a57d9ad | ||
|
|
17e48c4d46 | ||
|
|
7ac5b61955 | ||
|
|
c98f117689 | ||
|
|
337a0298bf | ||
|
|
7930b9b3ca | ||
|
|
7c0bc9c338 | ||
|
|
b62aa1491f | ||
|
|
7bff34ba6c | ||
|
|
c0d8ba243d | ||
|
|
c946067b9b | ||
|
|
db803eb74a | ||
|
|
c35e5ad7fe | ||
|
|
059f0acee6 | ||
|
|
a1ea837c1d | ||
|
|
186ab1409e | ||
|
|
1c6d968ed7 | ||
|
|
e18d795334 | ||
|
|
377c5d16f5 | ||
|
|
674379e6c9 | ||
|
|
d0f0e2c392 | ||
|
|
bf6d19e152 | ||
|
|
243b234033 | ||
|
|
56c5f17e01 | ||
|
|
5afe2a09a3 | ||
|
|
64f8ab42c1 | ||
|
|
5fb9fc8ec5 | ||
|
|
5a5dcd44df | ||
|
|
d4ff68d2bd | ||
|
|
ac3417555c | ||
|
|
e8bd89a672 | ||
|
|
fc3c897fa6 | ||
|
|
58bc2b070e | ||
|
|
991284d3b6 | ||
|
|
587d437f32 | ||
|
|
c0ae0f0a4b | ||
|
|
615448bbc3 | ||
|
|
68cfee09e0 | ||
|
|
a0656da6ef | ||
|
|
1395d44724 | ||
|
|
a42a394111 | ||
|
|
0fa8b85391 | ||
|
|
d4db7ef8cd | ||
|
|
3d9af90191 | ||
|
|
1ab9f62208 | ||
|
|
179e6ec141 | ||
|
|
2434d4ac71 | ||
|
|
a9e5c58897 | ||
|
|
35d9fa1f6c | ||
|
|
fb9117e9fb | ||
|
|
4af5e6a758 | ||
|
|
d4d149a5ff | ||
|
|
b0dd0109bb | ||
|
|
ac9f49a137 | ||
|
|
178b9b8170 | ||
|
|
886579fb48 | ||
|
|
7367c5a42e | ||
|
|
b47ec8d14d | ||
|
|
a31f758d55 | ||
|
|
181247ffcf | ||
|
|
cf5e341f49 | ||
|
|
15b20e62ac | ||
|
|
b8ad8fb003 | ||
|
|
7fe11de76a | ||
|
|
d12164148a | ||
|
|
f480dd23a4 | ||
|
|
fd57eb3076 | ||
|
|
4e6ac3490e | ||
|
|
177e16ca9d | ||
|
|
29e3f9b7a6 | ||
|
|
5436debf31 | ||
|
|
2c875db251 |
+2
-2
@@ -12,7 +12,7 @@ coverage:
|
||||
threshold: 0%
|
||||
base: auto
|
||||
branches:
|
||||
- master
|
||||
- main
|
||||
if_ci_failed: error
|
||||
informational: true
|
||||
only_pulls: true
|
||||
@@ -22,7 +22,7 @@ coverage:
|
||||
threshold: 1% # allows variations around the target
|
||||
base: auto
|
||||
branches:
|
||||
- master
|
||||
- main
|
||||
if_ci_failed: error
|
||||
only_pulls: true
|
||||
|
||||
|
||||
@@ -29,3 +29,47 @@ jobs:
|
||||
operations-per-run: 500
|
||||
exempt-issue-labels: "bug,WIP,ready-for-review,in-review,in-next"
|
||||
exempt-pr-labels: "bug,WIP,ready-for-review,in-review,in-next"
|
||||
|
||||
# Stale action for PRs with "in-review" label.
|
||||
stale-in-review-pr:
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
actions: write
|
||||
|
||||
steps:
|
||||
- uses: actions/stale@v9
|
||||
with:
|
||||
repo-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
stale-pr-message: ':warning: This PR has been automatically marked as stale because it has not had any activity in the last 150 days. *If no activity occurs in the next 30 days, it will be automatically closed.* Thank you for your contributions.'
|
||||
only-pr-labels: "in-review"
|
||||
days-before-pr-stale: 150
|
||||
days-before-pr-close: 30
|
||||
days-before-issue-stale: -1
|
||||
days-before-issue-close: -1
|
||||
stale-pr-label: 'stale'
|
||||
operations-per-run: 500
|
||||
|
||||
# Stale action for PRs with "WIP" label.
|
||||
stale-wip-pr:
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
actions: write
|
||||
|
||||
steps:
|
||||
- uses: actions/stale@v9
|
||||
with:
|
||||
repo-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
stale-pr-message: ':warning: This PR has been automatically marked as stale because it has not had any activity in the last 300 days. *If no activity occurs in the next 30 days, it will be automatically closed.* Thank you for your contributions.'
|
||||
only-pr-labels: "WIP"
|
||||
days-before-pr-stale: 300
|
||||
days-before-pr-close: 30
|
||||
days-before-issue-stale: -1
|
||||
days-before-issue-close: -1
|
||||
stale-pr-label: 'stale'
|
||||
operations-per-run: 500
|
||||
|
||||
@@ -208,10 +208,13 @@ miniapps/electromagnetics/volta
|
||||
miniapps/electromagnetics/tesla
|
||||
miniapps/electromagnetics/maxwell
|
||||
miniapps/electromagnetics/joule
|
||||
miniapps/electromagnetics/lorentz
|
||||
miniapps/electromagnetics/Volta-AMR*
|
||||
miniapps/electromagnetics/Tesla-AMR*
|
||||
miniapps/electromagnetics/Maxwell-Parallel*
|
||||
miniapps/electromagnetics/Joule_[0-9]*
|
||||
miniapps/electromagnetics/Lorentz_[0-9]*
|
||||
miniapps/electromagnetics/Lorentz.dat
|
||||
|
||||
miniapps/gslib/field-diff
|
||||
miniapps/gslib/field-interp
|
||||
|
||||
@@ -40,9 +40,16 @@ Discretization improvements
|
||||
provided that neighboring hexahedra are not refined in conflicting directions.
|
||||
A new ParMesh method is added to check for such conflicts, before refinement.
|
||||
|
||||
- Renamed the default GitHub branch from "master" to "main".
|
||||
|
||||
Meshing improvements
|
||||
--------------------
|
||||
|
||||
- Introduced NC-patch NURBS meshes, which are conforming element-wise but allow
|
||||
for nonconforming patch topology. This new mesh format supports element
|
||||
spacing formulas for refinement, as well as local refinement factors for a
|
||||
subset of knot vectors.
|
||||
|
||||
- Added support for higher order meshes in Mesh::MakeSimplicial and
|
||||
ParMesh::MakeSimplicial.
|
||||
|
||||
@@ -66,6 +73,10 @@ GPU computing
|
||||
spaces, and is the default derefinement operator constructed by
|
||||
`FiniteElementSpace::Update` and `ParFiniteElementSpace::Update`.
|
||||
The operator requires `FiniteElementSpace::Nonconforming() == true`.
|
||||
- Added new method: GridFunction::GetGradients, with GPU support, for computing
|
||||
the gradients of a GridFunction on all elements.
|
||||
- Added GPU support in GradientGridFunctionCoefficient and
|
||||
InnerProductCoefficient by implementing their Project methods.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
@@ -82,6 +93,10 @@ New and updated examples and miniapps
|
||||
- Added a new miniapp (tools/gridfunction-bounds) to compute piecewise linear
|
||||
bounds on a given high-order grid function.
|
||||
|
||||
- Added a new miniapp (electromagnetics/lorentz) which computes the trajectory
|
||||
of a charged particle, subject to Lorentz forces, in electrostatic and/or
|
||||
magnetostatic fields as computed by the volta or tesla miniapps.
|
||||
|
||||
API changes:
|
||||
-----------
|
||||
- mfem::internal::tensor and mfem::internal::dual have been moved to
|
||||
|
||||
+15
-10
@@ -278,6 +278,11 @@ if (MFEM_USE_OPENMP OR MFEM_USE_LEGACY_OPENMP)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Umpire (must be included before hypre, so hypre can use it if needed)
|
||||
if (MFEM_USE_UMPIRE)
|
||||
find_package(UMPIRE REQUIRED)
|
||||
endif()
|
||||
|
||||
# MPI -> hypre; PETSc (optional)
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(MPI REQUIRED)
|
||||
@@ -495,14 +500,13 @@ endif()
|
||||
|
||||
# RAJA
|
||||
if (MFEM_USE_RAJA)
|
||||
# RAJA uses FindCUDA, which needs CMP0146=OLD in CMake >= 3.27
|
||||
if(CMAKE_VERSION VERSION_GREATER_EQUAL 3.27.0)
|
||||
cmake_policy(SET CMP0146 OLD)
|
||||
endif()
|
||||
find_package(RAJA REQUIRED)
|
||||
endif()
|
||||
|
||||
# UMPIRE
|
||||
if (MFEM_USE_UMPIRE)
|
||||
find_package(UMPIRE REQUIRED)
|
||||
endif()
|
||||
|
||||
# GOOGLE-BENCHMARK
|
||||
if (MFEM_USE_BENCHMARK)
|
||||
find_package(Benchmark REQUIRED)
|
||||
@@ -596,7 +600,7 @@ set(MFEM_TPLS OPENMP HYPRE LAPACK BLAS SuperLUDist STRUMPACK METIS SuiteSparse
|
||||
NETCDF MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE
|
||||
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CALIPER CODIPACK
|
||||
BENCHMARK PARELAG TRIBOL MPI_CXX HIP HIPBLAS HIPSPARSE MOONOLITH BLITZ
|
||||
ALGOIM ENZYME)
|
||||
ALGOIM ENZYME CUDA::cudart)
|
||||
|
||||
# Add all created targets and *_FOUND libraries in the variables TPL_TARGETS and
|
||||
# TPL_LIBRARIES, respectively.
|
||||
@@ -614,6 +618,7 @@ foreach(TPL IN LISTS MFEM_TPLS)
|
||||
endif()
|
||||
endif()
|
||||
endforeach(TPL)
|
||||
|
||||
list(REVERSE TPL_LIBRARIES)
|
||||
list(REMOVE_DUPLICATES TPL_LIBRARIES)
|
||||
list(REVERSE TPL_LIBRARIES)
|
||||
@@ -647,7 +652,7 @@ if (MFEM_USE_CUDA)
|
||||
endif()
|
||||
|
||||
add_subdirectory(config)
|
||||
set(MASTER_HEADERS
|
||||
set(MAIN_HEADERS
|
||||
${PROJECT_SOURCE_DIR}/mfem.hpp
|
||||
${PROJECT_SOURCE_DIR}/mfem-performance.hpp)
|
||||
|
||||
@@ -684,7 +689,7 @@ set(MFEM_SOURCE_DIR ${CMAKE_CURRENT_SOURCE_DIR})
|
||||
set(MFEM_INSTALL_DIR ${CMAKE_INSTALL_PREFIX})
|
||||
|
||||
# Declaring the library
|
||||
mfem_add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
|
||||
mfem_add_library(mfem ${SOURCES} ${HEADERS} ${MAIN_HEADERS})
|
||||
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
|
||||
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES} ${TPL_TARGETS})
|
||||
if (TPL_TARGETS)
|
||||
@@ -883,12 +888,12 @@ install(TARGETS ${PROJECT_NAME}
|
||||
LIBRARY DESTINATION ${INSTALL_LIB_DIR}
|
||||
ARCHIVE DESTINATION ${INSTALL_LIB_DIR})
|
||||
|
||||
# Install the master headers
|
||||
# Install the main headers
|
||||
foreach(Header mfem.hpp mfem-performance.hpp)
|
||||
install(FILES ${PROJECT_BINARY_DIR}/InstallHeaders/${Header}
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR})
|
||||
endforeach()
|
||||
install(FILES ${MASTER_HEADERS} DESTINATION ${INSTALL_INCLUDE_DIR}/mfem)
|
||||
install(FILES ${MAIN_HEADERS} DESTINATION ${INSTALL_INCLUDE_DIR}/mfem)
|
||||
|
||||
# Install the headers (except common miniapp which is installed from its subdir)
|
||||
install(DIRECTORY ${MFEM_SOURCE_DIRS}
|
||||
|
||||
+33
-33
@@ -3,10 +3,10 @@
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-brightgreen.svg"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Arepo-check+branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuild-analysis+branch%3Amaster"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuilds-and-tests+branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/blob/main/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-brightgreen.svg"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Arepo-check+branch%3Amain"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=main"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuild-analysis+branch%3Amain"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=main"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuilds-and-tests+branch%3Amain"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=main"></a>
|
||||
<a href="https://ci.appveyor.com/project/mfem/mfem"><img alt="Build Status" src="https://ci.appveyor.com/api/projects/status/19non9sqm6msi2wy?svg=true"></a>
|
||||
<a href="https://docs.mfem.org/html/index.html"><img alt="Doxygen" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
|
||||
</p>
|
||||
@@ -27,7 +27,7 @@ in the MFEM community, you agree to abide by its rules.
|
||||
If you plan on contributing to MFEM, consider reviewing the
|
||||
[issue tracker](https://github.com/mfem/mfem/issues) first to check if a thread
|
||||
already exists for your desired feature or the bug you ran into. Use a pull
|
||||
request (PR) toward the `mfem:master` branch to propose your contribution. If
|
||||
request (PR) toward the `mfem:main` branch to propose your contribution. If
|
||||
you are planning significant code changes or have questions, you may want to
|
||||
open an [issue](https://github.com/mfem/mfem/issues) before issuing a PR. In
|
||||
addition to technical contributions, we are also interested in your results and
|
||||
@@ -47,7 +47,7 @@ back to them before issuing pull requests:
|
||||
- [Pull Requests](#pull-requests)
|
||||
- [MFEM PR Rules](#mfem-pr-rules)
|
||||
- [Pull Request Checklist](#pull-request-checklist)
|
||||
- [Master/Next Workflow](#masternext-workflow)
|
||||
- [Main/Next Workflow](#mainnext-workflow)
|
||||
- [Releases](#releases)
|
||||
- [Release Checklist](#release-checklist)
|
||||
- [LLNL Workflow](#llnl-workflow)
|
||||
@@ -66,12 +66,12 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
|
||||
## Quick Summary
|
||||
|
||||
- We encourage you to [join the MFEM organization](#mfem-organization) and create
|
||||
development branches off `mfem:master`.
|
||||
development branches off `mfem:main`.
|
||||
- Please follow the [developer guidelines](#developer-guidelines), in particular
|
||||
with regards to documentation and code styling.
|
||||
- Please do not commit large/binary files to the central repository (use a fork
|
||||
instead).
|
||||
- Pull requests should be issued toward `mfem:master`. Make sure
|
||||
- Pull requests should be issued toward `mfem:main`. Make sure
|
||||
to check the items off the [Pull Request Checklist](#pull-request-checklist) and
|
||||
follow the [MFEM PR Rules](#mfem-pr-rules).
|
||||
- When your contribution is fully working and ready to be reviewed, add
|
||||
@@ -81,8 +81,8 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
|
||||
- The reviewers have 3 weeks to evaluate the PR and work with the author to
|
||||
fix issues and implement improvements.
|
||||
- During review there should be no force pushes/rewriting history in the branch.
|
||||
- After approval, MFEM developers merge the PR manually in the [mfem:next branch](#masternext-workflow).
|
||||
- After a week of testing in `mfem:next`, the original PR is merged in `mfem:master`.
|
||||
- After approval, MFEM developers merge the PR manually in the [mfem:next branch](#mainnext-workflow).
|
||||
- After a week of testing in `mfem:next`, the original PR is merged in `mfem:main`.
|
||||
- We use [milestones](https://github.com/mfem/mfem/milestones) to coordinate the
|
||||
work on different PRs toward a release.
|
||||
- Don't hesitate to [contact us](#contact-information) if you have any questions.
|
||||
@@ -289,7 +289,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
author, is willing to work on it and be its champion.
|
||||
|
||||
- The author creates a branch for the new feature (with suffix `-dev`), off
|
||||
the `master` branch, or another existing feature branch, for example:
|
||||
the `main` branch, or another existing feature branch, for example:
|
||||
|
||||
```
|
||||
# Clone assuming you have setup your ssh keys on GitHub:
|
||||
@@ -298,8 +298,8 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
# Alternatively, clone using the "https" protocol:
|
||||
git clone https://github.com/mfem/mfem.git
|
||||
|
||||
# Create a new feature branch starting from "master":
|
||||
git checkout master
|
||||
# Create a new feature branch starting from "main":
|
||||
git checkout main
|
||||
git pull
|
||||
git checkout -b feature-dev
|
||||
|
||||
@@ -375,7 +375,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
### Pull Requests
|
||||
|
||||
- When your branch is ready for other developers to review / comment on
|
||||
the code, create a pull request towards `mfem:master`.
|
||||
the code, create a pull request towards `mfem:main`.
|
||||
|
||||
- Pull request typically have titles like:
|
||||
|
||||
@@ -411,7 +411,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
assigned, the PR is considered under review. To help with the review process
|
||||
there should be no force pushes/rewriting history in the branch.
|
||||
|
||||
- After approval, the PR is [tested](#masternext-workflow) for a week with
|
||||
- After approval, the PR is [tested](#mainnext-workflow) for a week with
|
||||
other approved PRs in the `mfem:next` branch.
|
||||
|
||||
- Consider manually running the tests in `tests/scripts` before merging in
|
||||
@@ -447,7 +447,7 @@ The Pull Request (PR) approval process in MFEM is similar to the approval of pap
|
||||
|
||||
3. A PR can be (manually) merged in the *next* branch only if 2 of the assigned reviewers have approved it and it has passed internal testing. This merge can be performed by any of the assigned reviewers or by any of the editors.
|
||||
|
||||
4. A PR can be merged in the *master* branch only if it has been tested successfully for a week in *next* and an editor has (optionally) taken a final look. This merge can be performed only by one of the editors.
|
||||
4. A PR can be merged in the *main* branch only if it has been tested successfully for a week in *next* and an editor has (optionally) taken a final look. This merge can be performed only by one of the editors.
|
||||
|
||||
#### Responsibilities of Editors
|
||||
|
||||
@@ -470,7 +470,7 @@ The current list of MFEM editors is:
|
||||
|
||||
5. To remind the reviewers about timely completion of their review.
|
||||
|
||||
6. To take a final look and complete the PR merge in *master*. The final look step is optional and shouldn't take more than 3 days.
|
||||
6. To take a final look and complete the PR merge in *main*. The final look step is optional and shouldn't take more than 3 days.
|
||||
|
||||
7. The assignment of bugfixes should be expedited proportional to their importance, e.g. in some cases the editor can assign much shorter review window.
|
||||
|
||||
@@ -492,7 +492,7 @@ Everyone on the MFEM team can be asked to serve as a reviewer on a PR in their a
|
||||
|
||||
5. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
|
||||
|
||||
6. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
|
||||
6. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *main*.
|
||||
|
||||
7. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
|
||||
|
||||
@@ -582,11 +582,11 @@ Before a PR can be merged, it should satisfy the following:
|
||||
- [ ] Update internal tests to include the new features.
|
||||
|
||||
|
||||
### Master/Next Workflow
|
||||
### Main/Next Workflow
|
||||
|
||||
MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
MFEM uses a `main`/`next`-branch workflow as described below:
|
||||
|
||||
- The `master` branch should always be of release quality and changes should not
|
||||
- The `main` branch should always be of release quality and changes should not
|
||||
be merged until they have been fully tested. This branch is protected, and
|
||||
changes can only be made through pull requests.
|
||||
|
||||
@@ -613,20 +613,20 @@ MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
|
||||
- After a week of testing in `next` (excluding bugfixes), both on GitHub, as
|
||||
well as [internally](#tests-at-llnl) at LLNL, the original PR is merged into
|
||||
`master` (provided there are no issues).
|
||||
`main` (provided there are no issues).
|
||||
|
||||
- After the merge, the feature branch is deleted (unless it is a long-term
|
||||
project with periodic PRs).
|
||||
|
||||
- The `next` branch is used just for integrated testing of all PRs approved for
|
||||
merging into `master` to verify that each works individually and that all of
|
||||
merging into `main` to verify that each works individually and that all of
|
||||
them work as a group. This branch can be discarded at any time, though we
|
||||
typically do that only at the end of a [release cycle](#releases).
|
||||
|
||||
|
||||
### Releases
|
||||
|
||||
- Releases are just tags in the `master` branch, e.g. https://github.com/mfem/mfem/releases/tag/v3.3.2,
|
||||
- Releases are just tags in the `main` branch, e.g. https://github.com/mfem/mfem/releases/tag/v3.3.2,
|
||||
and have a version that ends in an even "patch" number, e.g. `v3.2.2` or
|
||||
`v3.4` (by convention `v3.4` is the same as `v3.4.0`.) Between releases, the
|
||||
version ends in an odd "patch" number, e.g. `v3.3.3`.
|
||||
@@ -679,19 +679,19 @@ MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
|
||||
### Mirroring on Bitbucket
|
||||
|
||||
- The GitHub `master` and `next` branches are mirrored to the LLNL institutional
|
||||
Bitbucket repository as `gh-master` and `gh-next`.
|
||||
- The GitHub `main` and `next` branches are mirrored to the LLNL institutional
|
||||
Bitbucket repository as `gh-main` and `gh-next`.
|
||||
|
||||
- `gh-master` is merged into LLNL's internal `master` through pull requests; write
|
||||
permissions to `master` are restricted to ensure this is the only way in which it
|
||||
- `gh-main` is merged into LLNL's internal `main` through pull requests; write
|
||||
permissions to `main` are restricted to ensure this is the only way in which it
|
||||
gets updated.
|
||||
|
||||
- We never push directly from LLNL to GitHub.
|
||||
|
||||
- Versions of the code on LLNL's internal server, from most to least stable:
|
||||
- MFEM official release on mfem.org -- Most stable, tested in many apps.
|
||||
- `mfem:master` -- Recent development version, guaranteed to work.
|
||||
- `mfem:gh-master` -- Stable development version, passed testing, you can use
|
||||
- `mfem:main` -- Recent development version, guaranteed to work.
|
||||
- `mfem:gh-main` -- Stable development version, passed testing, you can use
|
||||
it to build your code between releases.
|
||||
- `mfem:gh-next` -- Bleeding-edge development version, may be broken, use at
|
||||
your own risk.
|
||||
@@ -721,14 +721,14 @@ directory.
|
||||
|
||||
|
||||
### Linux and Mac smoke tests
|
||||
We use GitHub Actions to drive the default tests on the `master` and `next`
|
||||
We use GitHub Actions to drive the default tests on the `main` and `next`
|
||||
branches. See the `.github/workflows` files and the logs at
|
||||
[https://github.com/mfem/mfem/actions](https://github.com/mfem/mfem/actions).
|
||||
|
||||
Testing using GitHub Actions should be kept lightweight, as there is a time
|
||||
constraint on jobs. Two virtual machines are configured - Mac (OS X) and Linux.
|
||||
|
||||
- Tests on the `master` branch are triggered whenever a PR is issued on this branch.
|
||||
- Tests on the `main` branch are triggered whenever a PR is issued on this branch.
|
||||
- Tests on the `next` branch are currently scheduled to run each night.
|
||||
|
||||
|
||||
@@ -744,7 +744,7 @@ and debug build is performed with a simple run of `ex1` to verify the executable
|
||||
|
||||
### Tests at LLNL
|
||||
|
||||
- We mirror the `master` and `next` branches internally (to `gh-master` and
|
||||
- We mirror the `main` and `next` branches internally (to `gh-main` and
|
||||
`gh-next`) and run longer nightly tests via cron. On the weekends, a more
|
||||
extensive test is run which extracts and executes all the different sample
|
||||
runs from each example and most miniapps.
|
||||
|
||||
@@ -859,7 +859,7 @@ The specific libraries and their options are:
|
||||
URL: https://github.com/CEED/libCEED
|
||||
https://ceed.exascaleproject.org/libceed
|
||||
Options: CEED_DIR, CEED_OPT, CEED_LIB.
|
||||
Versions: libCEED >= 0.12.
|
||||
Versions: libCEED >= 0.12.0.
|
||||
|
||||
- RAJA (optional), used when MFEM_USE_RAJA = YES.
|
||||
Beginning with MFEM v4.5.1, only RAJA v2022.10.3+ is supported.
|
||||
|
||||
@@ -84,6 +84,31 @@ set_and_check(MFEM_LIBRARY_DIR "@PACKAGE_LIB_INSTALL_DIR@")
|
||||
|
||||
check_required_components(MFEM)
|
||||
|
||||
include(CMakeFindDependencyMacro)
|
||||
|
||||
if (MFEM_USE_CUDA)
|
||||
# required for projects linking to MFEM+CUDA, even if they don't use CUDA directly
|
||||
find_dependency(CUDAToolkit)
|
||||
endif (MFEM_USE_CUDA)
|
||||
|
||||
if (MFEM_USE_HIP)
|
||||
# hip/rocm uses the modern MFEM way of linking to targets, need to find dependencies
|
||||
find_dependency(HIP)
|
||||
find_dependency(HIPBLAS)
|
||||
find_dependency(HIPSPARSE)
|
||||
if (MFEM_USE_MPI)
|
||||
# assume HYPRE uses HIP
|
||||
# alternatively could check HYPRE_USING_HIP
|
||||
find_dependency(rocsparse)
|
||||
find_dependency(rocrand)
|
||||
find_dependency(rocsolver)
|
||||
endif (MFEM_USE_MPI)
|
||||
endif (MFEM_USE_HIP)
|
||||
|
||||
if (MFEM_USE_RAJA)
|
||||
find_dependency(RAJA)
|
||||
endif()
|
||||
|
||||
if (NOT TARGET mfem)
|
||||
include(${CMAKE_CURRENT_LIST_DIR}/MFEMTargets.cmake)
|
||||
endif (NOT TARGET mfem)
|
||||
|
||||
@@ -27,6 +27,7 @@ if (HYPRE_FOUND OR TARGET HYPRE)
|
||||
if (HYPRE_USING_HIP)
|
||||
find_package(rocsparse REQUIRED)
|
||||
find_package(rocrand REQUIRED)
|
||||
find_package(rocsolver REQUIRED)
|
||||
endif()
|
||||
if (HYPRE_LIBRARIES AND HYPRE_INCLUDE_DIRS AND HYPRE_VERSION)
|
||||
find_package_handle_standard_args(HYPRE
|
||||
@@ -37,51 +38,91 @@ if (HYPRE_FOUND OR TARGET HYPRE)
|
||||
endif()
|
||||
|
||||
if (HYPRE_FETCH OR FETCH_TPLS)
|
||||
# Collect all HYPRE_ENABLE variables and pass them to hypre, assuming they are BOOL.
|
||||
set(HYPRE_CMAKE_OPTIONS "")
|
||||
get_cmake_property(all_vars VARIABLES)
|
||||
foreach(var ${all_vars})
|
||||
if(var MATCHES "^HYPRE_ENABLE")
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS "-D${var}:BOOL=${${var}}")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
set(HYPRE_FETCH_VERSION 2.33.0)
|
||||
set(HYPRE_FETCH_TAG "v${HYPRE_FETCH_VERSION}" CACHE STRING "Tag, branch, or commit for HYPRE")
|
||||
add_library(HYPRE STATIC IMPORTED)
|
||||
# set options and associated dependencies
|
||||
set(CMAKE_OPTIONS)
|
||||
list(APPEND CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
|
||||
if (MFEM_USE_CUDA)
|
||||
list(APPEND CMAKE_OPTIONS -DHYPRE_WITH_CUDA:BOOL=ON)
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DHYPRE_ENABLE_CUDA:BOOL=ON -DCMAKE_CUDA_ARCHITECTURES:STRING=${CMAKE_CUDA_ARCHITECTURES})
|
||||
find_package(CUDAToolkit REQUIRED)
|
||||
target_link_libraries(HYPRE INTERFACE CUDA::cusparse CUDA::curand CUDA::cublas)
|
||||
elseif (MFEM_USE_HIP)
|
||||
list(APPEND CMAKE_OPTIONS -DHYPRE_WITH_HIP:BOOL=ON)
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DHYPRE_ENABLE_HIP:BOOL=ON)
|
||||
find_package(rocsparse REQUIRED)
|
||||
find_package(rocrand REQUIRED)
|
||||
target_link_libraries(HYPRE INTERFACE rocsparse rocrand)
|
||||
endif()
|
||||
if (MFEM_USE_CUDA OR MFEM_USE_HIP)
|
||||
if (MFEM_USE_UMPIRE)
|
||||
if (EXISTS ${umpire_DIR})
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DHYPRE_ENABLE_UMPIRE:BOOL=ON -Dumpire_DIR:PATH=${umpire_DIR})
|
||||
else()
|
||||
message(FATAL_ERROR "MFEM_USE_UMPIRE=ON, however umpire_DIR isn't visible to HYPRE")
|
||||
endif()
|
||||
else()
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DHYPRE_ENABLE_UMPIRE:BOOL=OFF)
|
||||
message(WARNING
|
||||
"================================================================================
|
||||
Umpire is disabled while building HYPRE with GPU support.
|
||||
This is not recommended for performance reasons!
|
||||
Consider enabling Umpire with -DMFEM_USE_UMPIRE=ON and providing -DUMPIRE_DIR.
|
||||
================================================================================")
|
||||
endif()
|
||||
endif()
|
||||
if (MFEM_USE_SINGLE)
|
||||
list(APPEND CMAKE_OPTIONS -DHYPRE_ENABLE_SINGLE:BOOL=ON)
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DHYPRE_ENABLE_SINGLE:BOOL=ON)
|
||||
endif()
|
||||
# define external project and create future include directory so it is present
|
||||
# to pass CMake checks at end of MFEM configuration step
|
||||
message(STATUS "Will fetch HYPRE ${HYPRE_FETCH_VERSION} to be built with ${CMAKE_OPTIONS}")
|
||||
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/hypre)
|
||||
message(STATUS "Will fetch HYPRE ${HYPRE_FETCH_TAG} to be built with ${HYPRE_CMAKE_OPTIONS}")
|
||||
set(HYPRE_INSTALL ${CMAKE_BINARY_DIR}/fetch/hypre)
|
||||
include(ExternalProject)
|
||||
ExternalProject_Add(hypre
|
||||
GIT_REPOSITORY https://github.com/hypre-space/hypre.git
|
||||
GIT_TAG v${HYPRE_FETCH_VERSION}
|
||||
GIT_TAG ${HYPRE_FETCH_TAG}
|
||||
GIT_SHALLOW TRUE
|
||||
GIT_PROGRESS TRUE
|
||||
UPDATE_DISCONNECTED TRUE
|
||||
SOURCE_SUBDIR src
|
||||
PREFIX ${PREFIX}
|
||||
CMAKE_CACHE_ARGS -DCMAKE_INSTALL_PREFIX:PATH=${PREFIX} -DCMAKE_INSTALL_LIBDIR:PATH=lib ${CMAKE_OPTIONS})
|
||||
file(MAKE_DIRECTORY ${PREFIX}/include)
|
||||
PREFIX ${HYPRE_INSTALL}
|
||||
BUILD_COMMAND ${CMAKE_COMMAND} --build . -- -j${CMAKE_BUILD_PARALLEL_LEVEL}
|
||||
CMAKE_CACHE_ARGS -DCMAKE_INSTALL_PREFIX:PATH=${HYPRE_INSTALL} -DCMAKE_INSTALL_LIBDIR:PATH=lib ${HYPRE_CMAKE_OPTIONS})
|
||||
file(MAKE_DIRECTORY ${HYPRE_INSTALL}/include)
|
||||
# set imported library target properties
|
||||
add_dependencies(HYPRE hypre)
|
||||
set_target_properties(HYPRE PROPERTIES
|
||||
IMPORTED_LOCATION ${PREFIX}/lib/libHYPRE.a
|
||||
INTERFACE_INCLUDE_DIRECTORIES ${PREFIX}/include)
|
||||
IMPORTED_LOCATION ${HYPRE_INSTALL}/lib/libHYPRE.a
|
||||
INTERFACE_INCLUDE_DIRECTORIES ${HYPRE_INSTALL}/include)
|
||||
# convert HYPRE version to integer
|
||||
string(REGEX MATCHALL "[0-9]+" HYPRE_SPLIT_VERSION ${HYPRE_FETCH_VERSION})
|
||||
list(GET HYPRE_SPLIT_VERSION 0 HYPRE_MAJOR_VERSION)
|
||||
list(GET HYPRE_SPLIT_VERSION 1 HYPRE_MINOR_VERSION)
|
||||
list(GET HYPRE_SPLIT_VERSION 2 HYPRE_PATCH_VERSION)
|
||||
math(EXPR HYPRE_VERSION "10000*${HYPRE_MAJOR_VERSION} + 100*${HYPRE_MINOR_VERSION} + ${HYPRE_PATCH_VERSION}")
|
||||
# set cache variables that would otherwise be set after mfem_find_package call
|
||||
set(HYPRE_VERSION ${HYPRE_VERSION} CACHE STRING "HYPRE version." FORCE)
|
||||
if (HYPRE_FETCH_TAG MATCHES "^v?([0-9]+)\\.([0-9]+)\\.([0-9]+)$")
|
||||
# Exact release tag X.Y.Z
|
||||
string(REGEX MATCHALL "[0-9]+" HYPRE_SPLIT_VERSION "${HYPRE_FETCH_TAG}")
|
||||
elseif (HYPRE_FETCH_VERSION MATCHES "([0-9]+)\\.([0-9]+)(\\.([0-9]+))?")
|
||||
string(REGEX MATCHALL "[0-9]+" HYPRE_SPLIT_VERSION "${HYPRE_FETCH_VERSION}")
|
||||
else (NOT DEFINED HYPRE_VERSION)
|
||||
message(FATAL_ERROR "Unable to find HYPRE release version. Please provide it via -DHYPRE_VERSION")
|
||||
endif()
|
||||
if (HYPRE_SPLIT_VERSION AND NOT DEFINED HYPRE_VERSION)
|
||||
list(GET HYPRE_SPLIT_VERSION 0 HYPRE_MAJOR_VERSION)
|
||||
list(GET HYPRE_SPLIT_VERSION 1 HYPRE_MINOR_VERSION)
|
||||
if (HYPRE_SPLIT_VERSION GREATER 2)
|
||||
list(GET HYPRE_SPLIT_VERSION 2 HYPRE_PATCH_VERSION)
|
||||
else()
|
||||
set(HYPRE_PATCH_VERSION 0)
|
||||
endif()
|
||||
math(EXPR HYPRE_VERSION "10000*${HYPRE_MAJOR_VERSION} + 100*${HYPRE_MINOR_VERSION} + ${HYPRE_PATCH_VERSION}")
|
||||
set(HYPRE_VERSION ${HYPRE_VERSION} CACHE STRING "HYPRE version." FORCE)
|
||||
endif()
|
||||
return()
|
||||
endif()
|
||||
|
||||
@@ -149,7 +190,8 @@ endif()
|
||||
if (HYPRE_FOUND AND HYPRE_USING_HIP)
|
||||
find_package(rocsparse REQUIRED)
|
||||
find_package(rocrand REQUIRED)
|
||||
list(APPEND HYPRE_LIBRARIES ${rocsparse_LIBRARIES} ${rocrand_LIBRARIES})
|
||||
find_package(rocsolver REQUIRED)
|
||||
list(APPEND HYPRE_LIBRARIES ${rocsparse_LIBRARIES} ${rocrand_LIBRARIES} roc::rocsolver roc::rocblas)
|
||||
set(HYPRE_LIBRARIES ${HYPRE_LIBRARIES} CACHE STRING
|
||||
"HYPRE libraries + dependencies." FORCE)
|
||||
message(STATUS "Updated HYPRE_LIBRARIES: ${HYPRE_LIBRARIES}")
|
||||
|
||||
@@ -32,6 +32,7 @@ if (METIS_FETCH OR FETCH_TPLS)
|
||||
UPDATE_DISCONNECTED TRUE
|
||||
PREFIX ${PREFIX}
|
||||
CONFIGURE_COMMAND tar -xzf ../metis/metis-${METIS_FETCH_VERSION}-mac.tgz --strip=1
|
||||
BUILD_COMMAND $(MAKE) COPTIONS=-Wno-incompatible-pointer-types
|
||||
INSTALL_COMMAND mkdir -p ${PREFIX}/lib && cp libmetis.a ${PREFIX}/lib/)
|
||||
# set imported library target properties
|
||||
add_dependencies(METIS metis)
|
||||
|
||||
@@ -718,7 +718,7 @@ function(mfem_get_target_options Target CompileOptsVar LinkOptsVar)
|
||||
get_target_property(IsImported ${tgt} IMPORTED)
|
||||
# message(STATUS "${tgt}[IMPORTED]: ${IsImported}")
|
||||
# Generally, the possible target types are: STATIC_LIBRARY, MODULE_LIBRARY,
|
||||
# SHARED_LIBRARY, INTERFACE_LIBRARY, EXECUTABLE.
|
||||
# SHARED_LIBRARY, INTERFACE_LIBRARY, UNKNOWN_LIBRARY, EXECUTABLE.
|
||||
get_target_property(type ${tgt} TYPE)
|
||||
# message(STATUS "${tgt}[TYPE]: ${type}")
|
||||
unset(ImportConfig)
|
||||
@@ -766,7 +766,7 @@ function(mfem_get_target_options Target CompileOptsVar LinkOptsVar)
|
||||
else()
|
||||
message(STATUS " *** Warning: [${tgt}] LOCATION not defined!")
|
||||
endif()
|
||||
elseif ("${type}" STREQUAL "SHARED_LIBRARY")
|
||||
elseif ("${type}" STREQUAL "SHARED_LIBRARY" OR "${type}" STREQUAL "UNKNOWN_LIBRARY")
|
||||
get_target_property(Location ${tgt} LOCATION)
|
||||
if (Location)
|
||||
get_filename_component(Dir ${Location} DIRECTORY)
|
||||
@@ -932,12 +932,14 @@ function(mfem_export_mk_files)
|
||||
endif()
|
||||
set(MFEM_BUILD_TAG "${CMAKE_SYSTEM}")
|
||||
set(MFEM_PREFIX "${CMAKE_INSTALL_PREFIX}")
|
||||
# For the next 4 variable, these are the values for the build-tree version of
|
||||
# For the next 4 variables, these are the values for the build-tree version of
|
||||
# 'config.mk'
|
||||
set(MFEM_INC_DIR "${PROJECT_BINARY_DIR}")
|
||||
set(MFEM_LIB_DIR "${PROJECT_BINARY_DIR}")
|
||||
set(MFEM_TEST_MK "${PROJECT_SOURCE_DIR}/config/test.mk")
|
||||
set(MFEM_CONFIG_EXTRA "MFEM_BUILD_DIR ?= ${PROJECT_BINARY_DIR}")
|
||||
# TODO: CUDA/HIP support:
|
||||
set(MFEM_XLINKER "${CMAKE_CXX_LINKER_WRAPPER_FLAG}")
|
||||
set(MFEM_MPIEXEC ${MPIEXEC})
|
||||
if (NOT MFEM_MPIEXEC)
|
||||
set(MFEM_MPIEXEC "mpirun")
|
||||
|
||||
@@ -88,6 +88,7 @@ MFEM_BUILD_TAG = @MFEM_BUILD_TAG@
|
||||
MFEM_PREFIX = @MFEM_PREFIX@
|
||||
MFEM_INC_DIR = @MFEM_INC_DIR@
|
||||
MFEM_LIB_DIR = @MFEM_LIB_DIR@
|
||||
MFEM_XLINKER = @MFEM_XLINKER@
|
||||
|
||||
# Location of test.mk
|
||||
MFEM_TEST_MK = @MFEM_TEST_MK@
|
||||
|
||||
+1
-1
@@ -57,7 +57,7 @@ CUDA_DIR = $(or $(CUDA_HOME),$(patsubst %/,%,$(dir \
|
||||
CLANG_CUDA_FLAGS = -xcuda --cuda-path=$(CUDA_DIR) --cuda-gpu-arch=$(CUDA_ARCH)
|
||||
# flags for nvcc
|
||||
NVCC_FLAGS = -x=cu --expt-extended-lambda --expt-relaxed-constexpr \
|
||||
-arch=$(CUDA_ARCH)
|
||||
-arch=$(CUDA_ARCH) -isystem "$(CUDA_DIR)/include"
|
||||
# Prefixes for passing flags to the host compiler and linker when using
|
||||
# CUDA_CXX=nvcc
|
||||
CUDA_XCOMPILER = -Xcompiler=
|
||||
|
||||
@@ -33,8 +33,8 @@ RUN mkdir -p /opt/mfem-env \
|
||||
RUN cd /opt/mfem-env && \
|
||||
. /opt/spack/share/spack/setup-env.sh && \
|
||||
spack env activate . && \
|
||||
spack develop --path /code mfem@master+examples+miniapps && \
|
||||
spack add mfem@master+examples+miniapps && \
|
||||
spack develop --path /code mfem@main+examples+miniapps && \
|
||||
spack add mfem@main+examples+miniapps # && \
|
||||
spack install
|
||||
|
||||
# ensure mfem always on various paths
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
spack:
|
||||
specs: [mfem@master+examples+miniapps]
|
||||
view:
|
||||
specs: [mfem@main+examples+miniapps]
|
||||
view:
|
||||
mfem:
|
||||
root: /opt/mfem-view
|
||||
root: /opt/mfem-view
|
||||
link_type: copy
|
||||
concretization: together
|
||||
develop:
|
||||
mfem:
|
||||
path: /code
|
||||
spec: mfem@master+examples+miniapps
|
||||
spec: mfem@main+examples+miniapps
|
||||
|
||||
@@ -115,7 +115,7 @@ fi
|
||||
|
||||
# branch-history
|
||||
if [[ "${option}" == "--history" || "${option}" == "" ]]; then
|
||||
git fetch origin master:master
|
||||
git fetch origin main:main
|
||||
cd tests/scripts
|
||||
if ! ./runtest branch-history; then code=1; fi
|
||||
cd -
|
||||
|
||||
+593
@@ -0,0 +1,593 @@
|
||||
MFEM mesh v1.0
|
||||
|
||||
# Created by: Pointwise
|
||||
|
||||
# MFEM Geometry Types:
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
|
||||
dimension
|
||||
2
|
||||
|
||||
elements
|
||||
160
|
||||
1 3 1 164 163 0
|
||||
1 3 164 165 162 163
|
||||
1 3 2 166 164 1
|
||||
1 3 166 132 165 164
|
||||
1 3 3 167 166 2
|
||||
1 3 167 131 132 166
|
||||
1 3 4 168 167 3
|
||||
1 3 168 130 131 167
|
||||
1 3 5 169 168 4
|
||||
1 3 169 129 130 168
|
||||
1 3 6 170 169 5
|
||||
1 3 170 128 129 169
|
||||
1 3 171 172 170 6
|
||||
1 3 172 127 128 170
|
||||
1 3 124 125 172 171
|
||||
1 3 125 126 127 172
|
||||
1 3 162 165 173 161
|
||||
1 3 165 132 133 173
|
||||
1 3 161 173 174 160
|
||||
1 3 173 133 134 174
|
||||
1 3 160 174 175 159
|
||||
1 3 174 134 135 175
|
||||
1 3 6 7 176 171
|
||||
1 3 7 8 177 176
|
||||
1 3 171 176 123 124
|
||||
1 3 176 177 122 123
|
||||
1 3 159 175 178 158
|
||||
1 3 175 135 136 178
|
||||
1 3 158 178 179 157
|
||||
1 3 178 136 137 179
|
||||
1 3 157 179 180 156
|
||||
1 3 179 137 138 180
|
||||
1 3 122 177 181 121
|
||||
1 3 177 8 182 181
|
||||
1 3 8 9 183 182
|
||||
1 3 9 10 184 183
|
||||
1 3 10 11 185 184
|
||||
1 3 11 12 186 185
|
||||
1 3 12 13 187 186
|
||||
1 3 13 14 15 187
|
||||
1 3 121 181 119 120
|
||||
1 3 181 182 118 119
|
||||
1 3 182 183 117 118
|
||||
1 3 183 184 188 117
|
||||
1 3 184 185 109 188
|
||||
1 3 185 186 108 109
|
||||
1 3 186 187 189 108
|
||||
1 3 187 15 16 189
|
||||
1 3 109 110 190 188
|
||||
1 3 110 111 191 190
|
||||
1 3 111 112 113 191
|
||||
1 3 188 190 116 117
|
||||
1 3 190 191 115 116
|
||||
1 3 191 113 114 115
|
||||
1 3 189 192 107 108
|
||||
1 3 192 193 106 107
|
||||
1 3 193 194 105 106
|
||||
1 3 194 195 104 105
|
||||
1 3 195 196 103 104
|
||||
1 3 16 17 192 189
|
||||
1 3 17 18 193 192
|
||||
1 3 18 19 194 193
|
||||
1 3 19 20 195 194
|
||||
1 3 20 21 196 195
|
||||
1 3 97 98 197 96
|
||||
1 3 98 99 198 197
|
||||
1 3 99 100 199 198
|
||||
1 3 100 101 200 199
|
||||
1 3 101 102 201 200
|
||||
1 3 102 103 202 201
|
||||
1 3 103 196 203 202
|
||||
1 3 196 21 22 203
|
||||
1 3 96 197 204 95
|
||||
1 3 197 198 39 204
|
||||
1 3 198 199 38 39
|
||||
1 3 199 200 205 38
|
||||
1 3 200 201 32 205
|
||||
1 3 201 202 31 32
|
||||
1 3 202 203 206 31
|
||||
1 3 203 22 23 206
|
||||
1 3 32 33 207 205
|
||||
1 3 33 34 35 207
|
||||
1 3 205 207 37 38
|
||||
1 3 207 35 36 37
|
||||
1 3 39 40 208 204
|
||||
1 3 40 41 209 208
|
||||
1 3 41 42 210 209
|
||||
1 3 42 43 211 210
|
||||
1 3 43 44 212 211
|
||||
1 3 204 208 94 95
|
||||
1 3 208 209 93 94
|
||||
1 3 209 210 92 93
|
||||
1 3 210 211 91 92
|
||||
1 3 211 212 90 91
|
||||
1 3 90 212 213 89
|
||||
1 3 212 44 214 213
|
||||
1 3 44 45 215 214
|
||||
1 3 45 46 216 215
|
||||
1 3 46 47 217 216
|
||||
1 3 47 48 218 217
|
||||
1 3 48 49 219 218
|
||||
1 3 49 50 51 219
|
||||
1 3 89 213 87 88
|
||||
1 3 213 214 86 87
|
||||
1 3 214 215 85 86
|
||||
1 3 215 216 84 85
|
||||
1 3 216 217 83 84
|
||||
1 3 217 218 82 83
|
||||
1 3 218 219 220 82
|
||||
1 3 219 51 52 220
|
||||
1 3 53 221 220 52
|
||||
1 3 221 81 82 220
|
||||
1 3 54 222 221 53
|
||||
1 3 222 80 81 221
|
||||
1 3 55 223 222 54
|
||||
1 3 223 79 80 222
|
||||
1 3 26 27 224 25
|
||||
1 3 27 28 29 224
|
||||
1 3 25 224 225 24
|
||||
1 3 224 29 30 225
|
||||
1 3 24 225 206 23
|
||||
1 3 225 30 31 206
|
||||
1 3 154 155 226 153
|
||||
1 3 155 156 180 226
|
||||
1 3 153 226 227 152
|
||||
1 3 226 180 138 227
|
||||
1 3 152 227 228 151
|
||||
1 3 227 138 139 228
|
||||
1 3 151 228 229 150
|
||||
1 3 228 139 140 229
|
||||
1 3 150 229 230 149
|
||||
1 3 229 140 141 230
|
||||
1 3 149 230 231 148
|
||||
1 3 230 141 142 231
|
||||
1 3 148 231 232 147
|
||||
1 3 231 142 143 232
|
||||
1 3 147 232 145 146
|
||||
1 3 232 143 144 145
|
||||
1 3 56 233 223 55
|
||||
1 3 233 78 79 223
|
||||
1 3 57 234 233 56
|
||||
1 3 234 77 78 233
|
||||
1 3 58 235 234 57
|
||||
1 3 235 76 77 234
|
||||
1 3 61 236 59 60
|
||||
1 3 236 235 58 59
|
||||
1 3 62 237 236 61
|
||||
1 3 237 76 235 236
|
||||
1 3 63 238 237 62
|
||||
1 3 238 75 76 237
|
||||
1 3 64 239 238 63
|
||||
1 3 239 74 75 238
|
||||
1 3 65 240 239 64
|
||||
1 3 240 73 74 239
|
||||
1 3 66 241 240 65
|
||||
1 3 241 72 73 240
|
||||
1 3 67 242 241 66
|
||||
1 3 242 71 72 241
|
||||
1 3 68 69 242 67
|
||||
1 3 69 70 71 242
|
||||
|
||||
boundary
|
||||
164
|
||||
3 1 0 1
|
||||
3 1 1 2
|
||||
3 1 2 3
|
||||
3 1 3 4
|
||||
3 1 4 5
|
||||
3 1 5 6
|
||||
3 1 6 7
|
||||
3 1 7 8
|
||||
3 1 8 9
|
||||
3 1 9 10
|
||||
3 1 10 11
|
||||
3 1 11 12
|
||||
3 1 12 13
|
||||
3 1 13 14
|
||||
3 1 16 17
|
||||
3 1 17 18
|
||||
3 1 18 19
|
||||
3 1 19 20
|
||||
3 1 20 21
|
||||
3 1 21 22
|
||||
3 1 22 23
|
||||
3 1 23 24
|
||||
3 1 24 25
|
||||
3 1 25 26
|
||||
3 1 26 27
|
||||
3 1 27 28
|
||||
3 1 28 29
|
||||
3 1 29 30
|
||||
3 1 30 31
|
||||
3 1 31 32
|
||||
3 1 32 33
|
||||
3 1 33 34
|
||||
3 1 34 35
|
||||
3 1 35 36
|
||||
3 1 36 37
|
||||
3 1 37 38
|
||||
3 1 38 39
|
||||
3 1 39 40
|
||||
3 1 40 41
|
||||
3 1 41 42
|
||||
3 1 42 43
|
||||
3 1 43 44
|
||||
3 1 49 50
|
||||
3 1 48 49
|
||||
3 1 47 48
|
||||
3 1 46 47
|
||||
3 1 45 46
|
||||
3 1 44 45
|
||||
3 1 52 53
|
||||
3 1 53 54
|
||||
3 1 54 55
|
||||
3 1 57 58
|
||||
3 1 56 57
|
||||
3 1 55 56
|
||||
3 1 60 61
|
||||
3 1 61 62
|
||||
3 1 62 63
|
||||
3 1 63 64
|
||||
3 1 64 65
|
||||
3 1 65 66
|
||||
3 1 66 67
|
||||
3 1 67 68
|
||||
3 1 75 76
|
||||
3 1 74 75
|
||||
3 1 73 74
|
||||
3 1 72 73
|
||||
3 1 71 72
|
||||
3 1 70 71
|
||||
3 1 76 77
|
||||
3 1 77 78
|
||||
3 1 78 79
|
||||
3 1 81 82
|
||||
3 1 80 81
|
||||
3 1 79 80
|
||||
3 1 82 83
|
||||
3 1 83 84
|
||||
3 1 84 85
|
||||
3 1 85 86
|
||||
3 1 86 87
|
||||
3 1 87 88
|
||||
3 1 94 95
|
||||
3 1 93 94
|
||||
3 1 92 93
|
||||
3 1 91 92
|
||||
3 1 90 91
|
||||
3 1 96 97
|
||||
3 1 95 96
|
||||
3 1 97 98
|
||||
3 1 98 99
|
||||
3 1 99 100
|
||||
3 1 100 101
|
||||
3 1 101 102
|
||||
3 1 102 103
|
||||
3 1 107 108
|
||||
3 1 106 107
|
||||
3 1 105 106
|
||||
3 1 104 105
|
||||
3 1 103 104
|
||||
3 1 108 109
|
||||
3 1 109 110
|
||||
3 1 110 111
|
||||
3 1 111 112
|
||||
3 1 112 113
|
||||
3 1 113 114
|
||||
3 1 114 115
|
||||
3 1 115 116
|
||||
3 1 116 117
|
||||
3 1 119 120
|
||||
3 1 118 119
|
||||
3 1 117 118
|
||||
3 1 131 132
|
||||
3 1 130 131
|
||||
3 1 129 130
|
||||
3 1 128 129
|
||||
3 1 127 128
|
||||
3 1 126 127
|
||||
3 1 132 133
|
||||
3 1 133 134
|
||||
3 1 134 135
|
||||
3 1 137 138
|
||||
3 1 136 137
|
||||
3 1 135 136
|
||||
3 1 138 139
|
||||
3 1 139 140
|
||||
3 1 140 141
|
||||
3 1 141 142
|
||||
3 1 142 143
|
||||
3 1 143 144
|
||||
3 1 147 148
|
||||
3 1 146 147
|
||||
3 1 153 154
|
||||
3 1 152 153
|
||||
3 1 151 152
|
||||
3 1 150 151
|
||||
3 1 149 150
|
||||
3 1 148 149
|
||||
3 1 156 157
|
||||
3 1 157 158
|
||||
3 1 158 159
|
||||
3 1 161 162
|
||||
3 1 160 161
|
||||
3 1 159 160
|
||||
2 1 69 70
|
||||
2 1 68 69
|
||||
3 1 88 89
|
||||
3 1 89 90
|
||||
3 1 121 122
|
||||
3 1 120 121
|
||||
3 1 123 124
|
||||
3 1 122 123
|
||||
3 1 125 126
|
||||
3 1 124 125
|
||||
1 1 144 145
|
||||
1 1 145 146
|
||||
3 1 15 16
|
||||
3 1 14 15
|
||||
3 1 50 51
|
||||
3 1 51 52
|
||||
3 1 59 60
|
||||
3 1 58 59
|
||||
3 1 154 155
|
||||
3 1 155 156
|
||||
3 1 163 0
|
||||
3 1 162 163
|
||||
|
||||
vertices
|
||||
243
|
||||
2
|
||||
4 4
|
||||
4 3.5
|
||||
4 3
|
||||
4 2.5
|
||||
4 2
|
||||
4 1.5
|
||||
4 1
|
||||
4.5 1
|
||||
5 1
|
||||
5 1.5
|
||||
5 2
|
||||
5 2.5
|
||||
5 3
|
||||
5 3.5
|
||||
5 4
|
||||
5.500 4
|
||||
6 4
|
||||
6.500 4
|
||||
7 4
|
||||
7.5 4
|
||||
8 4
|
||||
8.5 4
|
||||
9 4
|
||||
9.5 4
|
||||
10 4
|
||||
10.5 4
|
||||
11 4
|
||||
11 3.5
|
||||
11 3
|
||||
10.5 3
|
||||
10 3
|
||||
9.5 3
|
||||
9.5 2.5
|
||||
10 2.5
|
||||
10.5 2.5
|
||||
10.5 2
|
||||
10.5 1.5
|
||||
10 1.5
|
||||
9.5 1.5
|
||||
9.5 1
|
||||
10 1
|
||||
10.5 1
|
||||
11 1
|
||||
11.5 1
|
||||
12 1
|
||||
12 1.5
|
||||
12 2
|
||||
12 2.5
|
||||
12 3
|
||||
12 3.5
|
||||
12 4
|
||||
12.5 4
|
||||
13 4
|
||||
13.333 3.75
|
||||
13.666 3.5
|
||||
14.000 3.25
|
||||
14.333 3.5
|
||||
14.666 3.75
|
||||
15.000 4
|
||||
15.500 4
|
||||
16.000 4
|
||||
16.000 3.5
|
||||
16.000 3
|
||||
16.000 2.5
|
||||
16.000 2
|
||||
16.000 1.5
|
||||
16.000 1
|
||||
16.000 0.5
|
||||
16.000 0
|
||||
15.500 0
|
||||
15.000 0
|
||||
15.000 0.5000000000000002
|
||||
15.000 1
|
||||
15.000 1.5
|
||||
15.000 2
|
||||
15.000 2.5
|
||||
15.000 3
|
||||
14.666 2.75
|
||||
14.333 2.5
|
||||
14.000 2.25
|
||||
13.666 2.5
|
||||
13.333 2.75
|
||||
13 3
|
||||
13 2.5
|
||||
13 2
|
||||
13 1.5
|
||||
13 1
|
||||
13 0.500
|
||||
13 0
|
||||
12.5 0
|
||||
12 0
|
||||
11.5 0
|
||||
11 0
|
||||
10.5 0
|
||||
10 0
|
||||
9.5 0
|
||||
9 0
|
||||
8.5 0
|
||||
8.5 0.5
|
||||
8.5 1
|
||||
8.5 1.5
|
||||
8.5 2
|
||||
8.5 2.5
|
||||
8.5 3
|
||||
8 3
|
||||
7.5 3
|
||||
7 3
|
||||
6.500 3
|
||||
6 3
|
||||
6 2.5
|
||||
6.5 2.5
|
||||
7 2.5
|
||||
7.5 2.5
|
||||
7.5 2
|
||||
7.5 1.5
|
||||
7.000 1.5
|
||||
6.5 1.5
|
||||
6 1.5
|
||||
6 1
|
||||
6 0.5
|
||||
6 0
|
||||
5.5 0
|
||||
5 0
|
||||
4.5 0
|
||||
4 0
|
||||
3.5 0
|
||||
3 0
|
||||
3 0.500
|
||||
3 1
|
||||
3 1.5
|
||||
3 2
|
||||
3 2.5
|
||||
3 3
|
||||
2.666 2.75
|
||||
2.333 2.5
|
||||
2.000 2.25
|
||||
1.666 2.5
|
||||
1.333 2.75
|
||||
1.000 3
|
||||
1.000 2.5
|
||||
1.000 2
|
||||
1.000 1.5
|
||||
1.000 1
|
||||
1.000 0.5000
|
||||
1.000 0
|
||||
0.5000 0
|
||||
0.0000 0
|
||||
0.0000 0.5
|
||||
0.0000 1
|
||||
0.0000 1.5
|
||||
0.0000 2
|
||||
0.0000 2.5
|
||||
0.0000 3
|
||||
0.0000 3.5
|
||||
0.0000 4
|
||||
0.5000 4
|
||||
1.000 4
|
||||
1.333 3.75
|
||||
1.666 3.5
|
||||
2.000 3.25
|
||||
2.333 3.5
|
||||
2.666 3.75
|
||||
3 4
|
||||
3.5 4
|
||||
3.5 3.5
|
||||
3 3.5
|
||||
3.5 3
|
||||
3.5 2.5
|
||||
3.5 2
|
||||
3.5 1.5
|
||||
3.5 1
|
||||
4 0.5
|
||||
3.5 0.5
|
||||
2.666 3.25
|
||||
2.333 3
|
||||
2.000 2.75
|
||||
4.5 0.5
|
||||
5 0.5
|
||||
1.666 3
|
||||
1.333 3.25
|
||||
1.000 3.5
|
||||
5.5 0.5
|
||||
5.500 1
|
||||
5.500 1.5
|
||||
5.500 2
|
||||
5.500 2.5
|
||||
5.500 3
|
||||
5.500 3.5
|
||||
6 2
|
||||
6 3.5
|
||||
6.5 2
|
||||
7 2
|
||||
6.5 3.5
|
||||
7 3.5
|
||||
7.5 3.5
|
||||
8 3.5
|
||||
8.5 3.5
|
||||
9 0.5
|
||||
9 1
|
||||
9 1.5
|
||||
9 2
|
||||
9 2.5
|
||||
9 3
|
||||
9 3.5
|
||||
9.5 0.5
|
||||
9.5 2
|
||||
9.5 3.5
|
||||
10 2
|
||||
10 0.5
|
||||
10.5 0.5
|
||||
11 0.5
|
||||
11.5 0.5
|
||||
12 0.5
|
||||
12.5 0.500
|
||||
12.5 1
|
||||
12.5 1.5
|
||||
12.5 2
|
||||
12.5 2.5
|
||||
12.5 3
|
||||
12.5 3.5
|
||||
13 3.5
|
||||
13.333 3.250
|
||||
13.666 3
|
||||
14.000 2.75
|
||||
10.5 3.5
|
||||
10 3.5
|
||||
0.500 3.5
|
||||
0.500 3
|
||||
0.500 2.5
|
||||
0.500 2
|
||||
0.500 1.5
|
||||
0.500 1
|
||||
0.500 0.5
|
||||
14.333 3
|
||||
14.666 3.25
|
||||
15.000 3.5
|
||||
15.500 3.5
|
||||
15.500 3
|
||||
15.500 2.5
|
||||
15.500 2
|
||||
15.500 1.5
|
||||
15.500 1
|
||||
15.500 0.5
|
||||
@@ -0,0 +1,342 @@
|
||||
MFEM NURBS NC-patch mesh v1.0
|
||||
dimension
|
||||
3
|
||||
|
||||
elements
|
||||
13
|
||||
0 1 5 0 8 10 11 9 4 6 7 5
|
||||
0 1 5 0 18 8 24 32 30 23 36 38
|
||||
0 1 5 0 0 18 32 14 12 30 38 29
|
||||
0 1 5 0 32 24 10 20 38 36 26 35
|
||||
0 1 5 0 14 32 20 2 29 38 35 16
|
||||
0 1 5 0 30 23 36 38 31 22 37 39
|
||||
0 1 5 0 12 30 38 29 13 31 39 28
|
||||
0 1 5 0 38 36 26 35 39 37 27 34
|
||||
0 1 5 0 29 38 35 16 28 39 34 17
|
||||
0 1 5 0 31 22 37 39 19 9 25 33
|
||||
0 1 5 0 13 31 39 28 1 19 33 15
|
||||
0 1 5 0 39 37 27 34 33 25 11 21
|
||||
0 1 5 0 28 39 34 17 15 33 21 3
|
||||
|
||||
boundary
|
||||
31
|
||||
9999 3 8 10 6 4
|
||||
9999 3 10 11 7 6
|
||||
9999 3 11 9 5 7
|
||||
9999 3 9 8 4 5
|
||||
9999 3 4 6 7 5
|
||||
9999 3 32 24 8 18
|
||||
9999 3 18 8 23 30
|
||||
9999 3 14 32 18 0
|
||||
9999 3 0 18 30 12
|
||||
9999 3 14 0 12 29
|
||||
9999 3 20 10 24 32
|
||||
9999 3 10 20 35 26
|
||||
9999 3 2 20 32 14
|
||||
9999 3 20 2 16 35
|
||||
9999 3 2 14 29 16
|
||||
9999 3 30 23 22 31
|
||||
9999 3 12 30 31 13
|
||||
9999 3 29 12 13 28
|
||||
9999 3 26 35 34 27
|
||||
9999 3 35 16 17 34
|
||||
9999 3 16 29 28 17
|
||||
9999 3 31 22 9 19
|
||||
9999 3 19 9 25 33
|
||||
9999 3 13 31 19 1
|
||||
9999 3 28 13 1 15
|
||||
9999 3 1 19 33 15
|
||||
9999 3 27 34 21 11
|
||||
9999 3 33 25 11 21
|
||||
9999 3 34 17 3 21
|
||||
9999 3 17 28 15 3
|
||||
9999 3 15 33 21 3
|
||||
|
||||
vertex_to_knotspan
|
||||
8
|
||||
23 0 1 8 10 11 9
|
||||
22 0 2 8 10 11 9
|
||||
24 1 0 8 10 11 9
|
||||
36 1 1 8 10 11 9
|
||||
37 1 2 8 10 11 9
|
||||
25 1 3 8 10 11 9
|
||||
26 2 1 8 10 11 9
|
||||
27 2 2 8 10 11 9
|
||||
|
||||
coordinates
|
||||
40
|
||||
3
|
||||
0 0 0
|
||||
0 1 0
|
||||
4 0 0
|
||||
4 1 0
|
||||
0 0 4
|
||||
0 1 4
|
||||
4 0 4
|
||||
4 1 4
|
||||
0 0 2
|
||||
0 1 2
|
||||
4 0 2
|
||||
4 1 2
|
||||
0 0.333333333333333 0
|
||||
0 0.666666666666667 0
|
||||
2 0 0
|
||||
2 1 0
|
||||
4 0.333333333333334 0
|
||||
4 0.666666666666667 0
|
||||
0 0 1
|
||||
0 1 1
|
||||
4 0 1
|
||||
4 1 1
|
||||
0 0.666666666666667 2
|
||||
0 0.333333333333333 2
|
||||
2 0 2
|
||||
2 1 2
|
||||
4 0.333333333333333 2
|
||||
4 0.666666666666667 2
|
||||
2 0.666666666666667 0
|
||||
2 0.333333333333333 0
|
||||
0 0.333333333333333 1
|
||||
0 0.666666666666667 1
|
||||
1.81325211007895 0 1
|
||||
1.81325211007895 1 1
|
||||
4 0.666666666666667 1
|
||||
4 0.333333333333333 1
|
||||
2 0.333333333333333 2
|
||||
2 0.666666666666667 2
|
||||
1.81325211007895 0.333333333333333 1
|
||||
1.81325211007895 0.666666666666667 1
|
||||
|
||||
edges
|
||||
87
|
||||
0 8 10
|
||||
1 10 11
|
||||
0 9 11
|
||||
1 8 9
|
||||
0 4 6
|
||||
1 6 7
|
||||
0 5 7
|
||||
1 4 5
|
||||
2 4 8
|
||||
2 6 10
|
||||
2 7 11
|
||||
2 5 9
|
||||
9 18 8
|
||||
7 8 24
|
||||
9 32 24
|
||||
7 18 32
|
||||
9 30 23
|
||||
7 23 36
|
||||
9 38 36
|
||||
7 30 38
|
||||
3 18 30
|
||||
3 8 23
|
||||
3 24 36
|
||||
3 32 38
|
||||
8 0 18
|
||||
8 14 32
|
||||
7 0 14
|
||||
8 12 30
|
||||
8 29 38
|
||||
7 12 29
|
||||
3 0 12
|
||||
3 14 29
|
||||
6 24 10
|
||||
9 20 10
|
||||
6 32 20
|
||||
6 36 26
|
||||
9 35 26
|
||||
6 38 35
|
||||
3 10 26
|
||||
3 20 35
|
||||
8 2 20
|
||||
6 14 2
|
||||
8 16 35
|
||||
6 29 16
|
||||
3 2 16
|
||||
9 31 22
|
||||
7 22 37
|
||||
9 39 37
|
||||
7 31 39
|
||||
4 30 31
|
||||
4 23 22
|
||||
4 36 37
|
||||
4 38 39
|
||||
8 13 31
|
||||
8 28 39
|
||||
7 13 28
|
||||
4 12 13
|
||||
4 29 28
|
||||
6 37 27
|
||||
9 34 27
|
||||
6 39 34
|
||||
4 26 27
|
||||
4 35 34
|
||||
8 17 34
|
||||
6 28 17
|
||||
4 16 17
|
||||
9 19 9
|
||||
7 9 25
|
||||
9 33 25
|
||||
7 19 33
|
||||
5 31 19
|
||||
5 22 9
|
||||
5 37 25
|
||||
5 39 33
|
||||
8 1 19
|
||||
8 15 33
|
||||
7 1 15
|
||||
5 13 1
|
||||
5 28 15
|
||||
6 25 11
|
||||
9 21 11
|
||||
6 33 21
|
||||
5 27 11
|
||||
5 34 21
|
||||
8 3 21
|
||||
6 15 3
|
||||
5 17 3
|
||||
|
||||
knotvectors
|
||||
10
|
||||
1 3 0 0 0.5 1 1
|
||||
1 4 0 0 0.333333333333333 0.666666666666667 1 1
|
||||
1 3 0 0 0.5 1 1
|
||||
1 2 0 0 1 1
|
||||
1 2 0 0 1 1
|
||||
1 2 0 0 1 1
|
||||
1 2 0 0 1 1
|
||||
1 2 0 0 1 1
|
||||
1 2 0 0 1 1
|
||||
1 2 0 0 1 1
|
||||
|
||||
spacing
|
||||
0
|
||||
|
||||
weights
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
|
||||
FiniteElementSpace
|
||||
FiniteElementCollection: NURBS1
|
||||
VDim: 3
|
||||
Ordering: 1
|
||||
|
||||
0 0 0
|
||||
0 1 0
|
||||
4 0 0
|
||||
4 1 0
|
||||
0 0 4
|
||||
0 1 4
|
||||
4 0 4
|
||||
4 1 4
|
||||
0 0 2
|
||||
0 1 2
|
||||
4 0 2
|
||||
4 1 2
|
||||
0 0.333333333333333 0
|
||||
0 0.666666666666667 0
|
||||
2 0 0
|
||||
2 1 0
|
||||
4 0.333333333333334 0
|
||||
4 0.666666666666667 0
|
||||
0 0 1
|
||||
0 1 1
|
||||
4 0 1
|
||||
4 1 1
|
||||
0 0.666666666666667 2
|
||||
0 0.333333333333333 2
|
||||
2 0 2
|
||||
2 1 2
|
||||
4 0.333333333333333 2
|
||||
4 0.666666666666667 2
|
||||
2 0.666666666666667 0
|
||||
2 0.333333333333333 0
|
||||
0 0.333333333333333 1
|
||||
0 0.666666666666667 1
|
||||
1.81325211007895 0 1
|
||||
1.81325211007895 1 1
|
||||
4 0.666666666666667 1
|
||||
4 0.333333333333333 1
|
||||
2 0.333333333333333 2
|
||||
2 0.666666666666667 2
|
||||
1.81325211007895 0.333333333333333 1
|
||||
1.81325211007895 0.666666666666667 1
|
||||
2 0 4
|
||||
4 0.333333333333333 4
|
||||
4 0.666666666666667 4
|
||||
2 1 4
|
||||
0 0.333333333333333 4
|
||||
0 0.666666666666667 4
|
||||
0 0 3
|
||||
4 0 3
|
||||
4 1 3
|
||||
0 1 3
|
||||
2 0 3
|
||||
4 0.333333333333333 3
|
||||
4 0.666666666666667 3
|
||||
2 1 3
|
||||
0 0.666666666666667 3
|
||||
0 0.333333333333333 3
|
||||
2 0.333333333333333 4
|
||||
2 0.666666666666667 4
|
||||
2 0.333333333333333 3
|
||||
2 0.666666666666667 3
|
||||
@@ -0,0 +1,96 @@
|
||||
MFEM NURBS NC-patch mesh v1.0
|
||||
dimension
|
||||
2
|
||||
|
||||
# rank attr geom ref_type nodes/children
|
||||
elements
|
||||
3
|
||||
0 1 3 0 0 4 5 1
|
||||
0 1 3 0 6 7 4 2
|
||||
0 1 3 0 6 3 5 7
|
||||
|
||||
# attr geom nodes
|
||||
boundary
|
||||
7
|
||||
1 1 0 4
|
||||
1 1 5 1
|
||||
1 1 1 0
|
||||
1 1 2 6
|
||||
1 1 6 3
|
||||
1 1 4 2
|
||||
1 1 5 3
|
||||
|
||||
vertex_to_knotspan
|
||||
1
|
||||
7 1 4 5
|
||||
|
||||
# top-level node coordinates
|
||||
coordinates
|
||||
8
|
||||
2
|
||||
0 0
|
||||
0 1
|
||||
2 0
|
||||
2 1
|
||||
1 0
|
||||
1 1
|
||||
2 0.5
|
||||
1 0.5
|
||||
|
||||
edges
|
||||
11
|
||||
0 0 4
|
||||
1 4 5
|
||||
0 1 5
|
||||
1 0 1
|
||||
2 6 7
|
||||
4 7 4
|
||||
2 2 4
|
||||
4 6 2
|
||||
3 6 3
|
||||
2 3 5
|
||||
3 7 5
|
||||
|
||||
knotvectors
|
||||
5
|
||||
1 3 0 0 0.5 1 1
|
||||
1 3 0 0 0.5 1 1
|
||||
1 2 0 0 1 1
|
||||
1 2 0 0 1 1
|
||||
1 2 0 0 1 1
|
||||
|
||||
spacing
|
||||
0
|
||||
|
||||
weights
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
1.0
|
||||
|
||||
FiniteElementSpace
|
||||
FiniteElementCollection: NURBS1
|
||||
VDim: 2
|
||||
Ordering: 1
|
||||
|
||||
0 0
|
||||
0 1
|
||||
2 0
|
||||
2 1
|
||||
1 0
|
||||
1 1
|
||||
2 0.5
|
||||
1 0.5
|
||||
0.5 0
|
||||
0.5 1
|
||||
0 0.5
|
||||
0.5 0.5
|
||||
mfem_mesh_end
|
||||
@@ -202,6 +202,7 @@ namespace mfem {
|
||||
* - <a class="el" href="tesla_8cpp_source.html">Tesla</a>: simple magnetostatics simulation code
|
||||
* - <a class="el" href="maxwell_8cpp_source.html">Maxwell</a>: simple transient full-wave electromagnetics simulation code
|
||||
* - <a class="el" href="joule_8cpp_source.html">Joule</a>: transient magnetics and Joule heating miniapp
|
||||
* - <a class="el" href="lorentz_8cpp_source.html">Lorentz</a>: simple particle tracking code based on the Lorentz force
|
||||
* - <a class="el" href="classmfem_1_1navier_1_1NavierSolver.html">Navier</a>: solve the transient incompressible Navier-Stokes equations
|
||||
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
|
||||
* - <a class="el" href="klein-bottle_8cpp_source.html">Klein Bottle</a>: generate three types of Klein bottle surfaces
|
||||
|
||||
@@ -205,6 +205,15 @@ if (MFEM_ENABLE_TESTING)
|
||||
$<TARGET_FILE:ex25p> "-no-vis" "--mumps-solver"
|
||||
${MPIEXEC_POSTFLAGS})
|
||||
endif()
|
||||
|
||||
# Parallel libCEED example
|
||||
if (MFEM_USE_CEED AND MFEM_USE_MPI)
|
||||
add_test(NAME ex1p_ceed_np=${MFEM_MPI_NP}
|
||||
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} ${MFEM_MPI_NP}
|
||||
${MPIEXEC_PREFLAGS}
|
||||
$<TARGET_FILE:ex1p> "-no-vis" "-d ceed-cpu" "-pa" "-a"
|
||||
${MPIEXEC_POSTFLAGS})
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Include the examples/amgx directory if AmgX is enabled
|
||||
|
||||
@@ -27,6 +27,7 @@
|
||||
// ex1 -m ../data/fichera-amr.mesh
|
||||
// ex1 -m ../data/mobius-strip.mesh
|
||||
// ex1 -m ../data/mobius-strip.mesh -o -1 -sc
|
||||
// ex1 -m ../data/nc3-nurbs.mesh -o -1
|
||||
//
|
||||
// Device sample runs:
|
||||
// ex1 -pa -d cuda
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Jupyter Notebooks using xeus-cling
|
||||
|
||||
[](https://mybinder.org/v2/gh/mfem/mfem/master?filepath=examples%2Fjupyter%2Fex.ipynb)
|
||||
[](https://mybinder.org/v2/gh/mfem/mfem/main?filepath=examples%2Fjupyter%2Fex.ipynb)
|
||||
|
||||
[xeus-cling](https://github.com/jupyter-xeus/xeus-cling) is a C++ Jupyter Kernel based on [cling](https://github.com/root-project/cling),
|
||||
which can be used to create interactive C++ MFEM and GLVis notebooks.
|
||||
|
||||
@@ -173,6 +173,12 @@ ex11p-test-cpardiso: ex11p
|
||||
@$(call mfem-test,$<, $(RUN_MPI), MKL_CPARDISO example,--cpardiso)
|
||||
test-par-YES: ex11p-test-cpardiso
|
||||
endif
|
||||
ifeq ($(MFEM_USE_CEED),YES)
|
||||
ex1p-test-ceed: ex1p
|
||||
@$(call mfem-test,$<, $(RUN_MPI),\
|
||||
Parallel libCEED example,-d ceed-cpu -pa -a)
|
||||
test-par-YES: ex1p-test-ceed
|
||||
endif
|
||||
|
||||
# Testing: "test" target and mfem-test* variables are defined in config/test.mk
|
||||
|
||||
|
||||
@@ -171,6 +171,11 @@ set(HDRS
|
||||
bilinearform.hpp
|
||||
bilinearform_ext.hpp
|
||||
bilininteg.hpp
|
||||
integ/lininteg_domain_kernels.hpp
|
||||
integ/bilininteg_dgdiffusion_kernels.hpp
|
||||
integ/bilininteg_dgtrace_kernels.hpp
|
||||
integ/bilininteg_vecdiffusion_kernels.hpp
|
||||
integ/bilininteg_convection_kernels.hpp
|
||||
integ/bilininteg_diffusion_kernels.hpp
|
||||
integ/bilininteg_elasticity_kernels.hpp
|
||||
integ/bilininteg_hcurl_kernels.hpp
|
||||
@@ -241,8 +246,13 @@ set(HDRS
|
||||
lor/lor_ams.hpp
|
||||
lor/lor_batched.hpp
|
||||
lor/lor_h1.hpp
|
||||
lor/lor_dg.hpp
|
||||
lor/lor_nd.hpp
|
||||
lor/lor_rt.hpp
|
||||
lor/lor_h1_impl.hpp
|
||||
lor/lor_dg_impl.hpp
|
||||
lor/lor_nd_impl.hpp
|
||||
lor/lor_rt_impl.hpp
|
||||
lor/lor_util.hpp
|
||||
multigrid.hpp
|
||||
nonlinearform.hpp
|
||||
|
||||
+140
-40
@@ -2496,8 +2496,7 @@ private:
|
||||
#endif
|
||||
|
||||
public:
|
||||
ConvectionIntegrator(VectorCoefficient &q, real_t a = 1.0)
|
||||
: Q(&q) { alpha = a; }
|
||||
ConvectionIntegrator(VectorCoefficient &q, real_t a = 1.0);
|
||||
|
||||
void AssembleElementMatrix(const FiniteElement &,
|
||||
ElementTransformation &,
|
||||
@@ -2530,6 +2529,28 @@ public:
|
||||
|
||||
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
|
||||
|
||||
/// arguments: NE, B, G, Bt, Gt, pa_data, x, y, D1D, Q1D
|
||||
using ApplyKernelType = void (*)(const int, const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &, const Vector &,
|
||||
const Vector &, Vector &, const int,
|
||||
const int);
|
||||
|
||||
/// arguments: DIMS, D1D, Q1D
|
||||
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
|
||||
/// arguments: DIMS, D1D, Q1D
|
||||
MFEM_REGISTER_KERNELS(ApplyPATKernels, ApplyKernelType, (int, int, int));
|
||||
|
||||
template <int DIM, int D1D, int Q1D>
|
||||
static void AddSpecialization()
|
||||
{
|
||||
ApplyPAKernels::Specialization<DIM, D1D, Q1D>::Add();
|
||||
ApplyPATKernels::Specialization<DIM, D1D, Q1D>::Add();
|
||||
}
|
||||
|
||||
struct Kernels { Kernels(); };
|
||||
|
||||
protected:
|
||||
const IntegrationRule* GetDefaultIntegrationRule(
|
||||
const FiniteElement& trial_fe,
|
||||
@@ -2800,15 +2821,13 @@ protected:
|
||||
bool symmetric = true; ///< False if using a nonsymmetric matrix coefficient
|
||||
|
||||
public:
|
||||
CurlCurlIntegrator() { Q = NULL; DQ = NULL; MQ = NULL; }
|
||||
CurlCurlIntegrator();
|
||||
/// Construct a bilinear form integrator for Nedelec elements
|
||||
CurlCurlIntegrator(Coefficient &q, const IntegrationRule *ir = NULL) :
|
||||
BilinearFormIntegrator(ir), Q(&q), DQ(NULL), MQ(NULL) { }
|
||||
CurlCurlIntegrator(Coefficient &q, const IntegrationRule *ir = nullptr);
|
||||
CurlCurlIntegrator(DiagonalMatrixCoefficient &dq,
|
||||
const IntegrationRule *ir = NULL) :
|
||||
BilinearFormIntegrator(ir), Q(NULL), DQ(&dq), MQ(NULL) { }
|
||||
CurlCurlIntegrator(MatrixCoefficient &mq, const IntegrationRule *ir = NULL) :
|
||||
BilinearFormIntegrator(ir), Q(NULL), DQ(NULL), MQ(&mq) { }
|
||||
const IntegrationRule *ir = nullptr);
|
||||
CurlCurlIntegrator(MatrixCoefficient &mq,
|
||||
const IntegrationRule *ir = nullptr);
|
||||
|
||||
/* Given a particular Finite Element, compute the
|
||||
element curl-curl matrix elmat */
|
||||
@@ -2838,6 +2857,34 @@ public:
|
||||
void AssembleDiagonalPA(Vector& diag) override;
|
||||
|
||||
const Coefficient *GetCoefficient() const { return Q; }
|
||||
|
||||
/// arguments: d1d, q1d, symmetric, NE, bo, bc, bot, bct, gc, gct, pa_data,
|
||||
/// x, y, useAbs
|
||||
using ApplyKernelType = void (*)(
|
||||
const int, const int, const bool, const int, const Array<real_t> &,
|
||||
const Array<real_t> &, const Array<real_t> &, const Array<real_t> &,
|
||||
const Array<real_t> &, const Array<real_t> &, const Vector &,
|
||||
const Vector &, Vector &, const bool);
|
||||
|
||||
/// arguments: d1d, q1d, symmetric, ne, Bo, Bc, Go, Gc, pa_data, diag
|
||||
using DiagonalKernelType = void (*)(const int, const int, const bool,
|
||||
const int, const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &, const Vector &,
|
||||
Vector &);
|
||||
|
||||
/// parameters: dim, d1d, q1d
|
||||
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
|
||||
/// parameters: dim, d1d, q1d
|
||||
MFEM_REGISTER_KERNELS(DiagonalPAKernels, DiagonalKernelType, (int, int, int));
|
||||
struct Kernels { Kernels(); };
|
||||
|
||||
template <int DIM, int D1D, int Q1D> static void AddSpecialization()
|
||||
{
|
||||
ApplyPAKernels::Specialization<DIM, D1D, Q1D>::Add();
|
||||
DiagonalPAKernels::Specialization<DIM, D1D, Q1D>::Add();
|
||||
}
|
||||
};
|
||||
|
||||
/** Integrator for $(\mathrm{curl}(u), \mathrm{curl}(v))$ for FE spaces defined by 'dim' copies of a
|
||||
@@ -3091,21 +3138,18 @@ private:
|
||||
Vector vcoeff;
|
||||
|
||||
public:
|
||||
VectorDiffusionIntegrator() { }
|
||||
VectorDiffusionIntegrator(const IntegrationRule *ir = nullptr);
|
||||
|
||||
/** \brief Integrator with unit coefficient for caller-specified vector
|
||||
dimension.
|
||||
|
||||
If the vector dimension does not match the true dimension of the space,
|
||||
the resulting element matrix will be mathematically invalid. */
|
||||
VectorDiffusionIntegrator(int vector_dimension)
|
||||
: vdim(vector_dimension) { }
|
||||
VectorDiffusionIntegrator(int vector_dimension);
|
||||
|
||||
VectorDiffusionIntegrator(Coefficient &q)
|
||||
: Q(&q) { }
|
||||
VectorDiffusionIntegrator(Coefficient &q);
|
||||
|
||||
VectorDiffusionIntegrator(Coefficient &q, const IntegrationRule *ir)
|
||||
: BilinearFormIntegrator(ir), Q(&q) { }
|
||||
VectorDiffusionIntegrator(Coefficient &q, const IntegrationRule *ir);
|
||||
|
||||
/** \brief Integrator with scalar coefficient for caller-specified vector
|
||||
dimension.
|
||||
@@ -3115,8 +3159,7 @@ public:
|
||||
|
||||
If the vector dimension does not match the true dimension of the space,
|
||||
the resulting element matrix will be mathematically invalid. */
|
||||
VectorDiffusionIntegrator(Coefficient &q, int vector_dimension)
|
||||
: Q(&q), vdim(vector_dimension) { }
|
||||
VectorDiffusionIntegrator(Coefficient &q, int vector_dimension);
|
||||
|
||||
/** \brief Integrator with \c VectorCoefficient. The vector dimension of the
|
||||
\c FiniteElementSpace is assumed to be the same as the dimension of the
|
||||
@@ -3127,8 +3170,7 @@ public:
|
||||
|
||||
If the vector dimension does not match the true dimension of the space,
|
||||
the resulting element matrix will be mathematically invalid. */
|
||||
VectorDiffusionIntegrator(VectorCoefficient &vq)
|
||||
: VQ(&vq), vdim(vq.GetVDim()) { }
|
||||
VectorDiffusionIntegrator(VectorCoefficient &vq);
|
||||
|
||||
/** \brief Integrator with \c MatrixCoefficient. The vector dimension of the
|
||||
\c FiniteElementSpace is assumed to be the same as the dimension of the
|
||||
@@ -3139,8 +3181,7 @@ public:
|
||||
|
||||
If the vector dimension does not match the true dimension of the space,
|
||||
the resulting element matrix will be mathematically invalid. */
|
||||
VectorDiffusionIntegrator(MatrixCoefficient& mq)
|
||||
: MQ(&mq), vdim(mq.GetVDim()) { }
|
||||
VectorDiffusionIntegrator(MatrixCoefficient& mq);
|
||||
|
||||
void AssembleElementMatrix(const FiniteElement &el,
|
||||
ElementTransformation &Trans,
|
||||
@@ -3156,6 +3197,28 @@ public:
|
||||
void AddMultPA(const Vector &x, Vector &y) const override;
|
||||
void AddMultMF(const Vector &x, Vector &y) const override;
|
||||
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
|
||||
|
||||
/// arguments: ne, B, G, Bt, Gt, pa_data, x, y, d1d, q1d, vdim
|
||||
using ApplyKernelType = void (*)(const int, const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &, const Vector &,
|
||||
const Vector &, Vector &, const int,
|
||||
const int, const int);
|
||||
|
||||
/// arguments: dim, vdim, d1d, q1d
|
||||
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int, int));
|
||||
|
||||
template <int DIM, int VDIM, int D1D, int Q1D>
|
||||
static void AddSpecialization()
|
||||
{
|
||||
ApplyPAKernels::Specialization<DIM, VDIM, D1D, Q1D>::Add();
|
||||
}
|
||||
|
||||
struct Kernels
|
||||
{
|
||||
Kernels();
|
||||
};
|
||||
};
|
||||
|
||||
/** Integrator for the linear elasticity form:
|
||||
@@ -3309,8 +3372,8 @@ public:
|
||||
class DGTraceIntegrator : public BilinearFormIntegrator
|
||||
{
|
||||
protected:
|
||||
Coefficient *rho;
|
||||
VectorCoefficient *u;
|
||||
Coefficient *rho = nullptr;
|
||||
VectorCoefficient *u = nullptr;
|
||||
real_t alpha, beta;
|
||||
// PA extension
|
||||
Vector pa_data;
|
||||
@@ -3323,17 +3386,16 @@ private:
|
||||
Vector tr_shape1, te_shape1, tr_shape2, te_shape2;
|
||||
|
||||
public:
|
||||
DGTraceIntegrator(real_t a, real_t b);
|
||||
|
||||
/// Construct integrator with $\rho = 1$, $\beta = \alpha/2$.
|
||||
DGTraceIntegrator(VectorCoefficient &u_, real_t a)
|
||||
{ rho = NULL; u = &u_; alpha = a; beta = 0.5*a; }
|
||||
DGTraceIntegrator(VectorCoefficient &u_, real_t a);
|
||||
|
||||
/// Construct integrator with $\rho = 1$.
|
||||
DGTraceIntegrator(VectorCoefficient &u_, real_t a, real_t b)
|
||||
{ rho = NULL; u = &u_; alpha = a; beta = b; }
|
||||
DGTraceIntegrator(VectorCoefficient &u_, real_t a, real_t b);
|
||||
|
||||
DGTraceIntegrator(Coefficient &rho_, VectorCoefficient &u_,
|
||||
real_t a, real_t b)
|
||||
{ rho = &rho_; u = &u_; alpha = a; beta = b; }
|
||||
real_t a, real_t b);
|
||||
|
||||
using BilinearFormIntegrator::AssembleFaceMatrix;
|
||||
void AssembleFaceMatrix(const FiniteElement &el1,
|
||||
@@ -3372,6 +3434,26 @@ public:
|
||||
static const IntegrationRule &GetRule(Geometry::Type geom, int order,
|
||||
const ElementTransformation &T);
|
||||
|
||||
/// arguments: nf, B, Bt, pa_data, x, y, dofs1D, quad1D
|
||||
using ApplyKernelType = void (*)(const int, const Array<real_t> &,
|
||||
const Array<real_t> &, const Vector &,
|
||||
const Vector &, Vector &, const int,
|
||||
const int);
|
||||
|
||||
/// arguments: DIM, d1d, q1d
|
||||
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
|
||||
/// arguments: DIM, d1d, q1d
|
||||
MFEM_REGISTER_KERNELS(ApplyPATKernels, ApplyKernelType, (int, int, int));
|
||||
|
||||
template <int DIM, int D1D, int Q1D> static void AddSpecialization()
|
||||
{
|
||||
ApplyPAKernels::Specialization<DIM, D1D, Q1D>::Add();
|
||||
ApplyPATKernels::Specialization<DIM, D1D, Q1D>::Add();
|
||||
}
|
||||
|
||||
struct Kernels { Kernels(); };
|
||||
|
||||
|
||||
private:
|
||||
void SetupPA(const FiniteElementSpace &fes, FaceType type);
|
||||
};
|
||||
@@ -3418,8 +3500,8 @@ public:
|
||||
class DGDiffusionIntegrator : public BilinearFormIntegrator
|
||||
{
|
||||
protected:
|
||||
Coefficient *Q;
|
||||
MatrixCoefficient *MQ;
|
||||
Coefficient *Q = nullptr;
|
||||
MatrixCoefficient *MQ = nullptr;
|
||||
real_t sigma, kappa;
|
||||
|
||||
// these are not thread-safe!
|
||||
@@ -3434,15 +3516,11 @@ protected:
|
||||
IntegrationRules irs{0, Quadrature1D::GaussLobatto};
|
||||
|
||||
public:
|
||||
DGDiffusionIntegrator(const real_t s, const real_t k)
|
||||
: Q(NULL), MQ(NULL), sigma(s), kappa(k) { }
|
||||
DGDiffusionIntegrator(Coefficient &q, const real_t s, const real_t k)
|
||||
: Q(&q), MQ(NULL), sigma(s), kappa(k) { }
|
||||
DGDiffusionIntegrator(MatrixCoefficient &q, const real_t s, const real_t k)
|
||||
: Q(NULL), MQ(&q), sigma(s), kappa(k) { }
|
||||
DGDiffusionIntegrator(const real_t s, const real_t k);
|
||||
DGDiffusionIntegrator(Coefficient &q, const real_t s, const real_t k);
|
||||
DGDiffusionIntegrator(MatrixCoefficient &q, const real_t s, const real_t k);
|
||||
using BilinearFormIntegrator::AssembleFaceMatrix;
|
||||
void AssembleFaceMatrix(const FiniteElement &el1,
|
||||
const FiniteElement &el2,
|
||||
void AssembleFaceMatrix(const FiniteElement &el1, const FiniteElement &el2,
|
||||
FaceElementTransformations &Trans,
|
||||
DenseMatrix &elmat) override;
|
||||
|
||||
@@ -3461,6 +3539,28 @@ public:
|
||||
|
||||
const IntegrationRule &GetRule(int order, Geometry::Type geom);
|
||||
|
||||
real_t GetPenaltyParameter() const { return kappa; }
|
||||
|
||||
/// arguments: nf, B, Bt, G, Gt, sigma, pa_data, x, dxdn, y, dydn, dofs1D,
|
||||
/// quad1D
|
||||
using ApplyKernelType = void (*)(const int, const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &, const real_t,
|
||||
const Vector &, const Vector &_,
|
||||
const Vector &, Vector &, Vector &,
|
||||
const int, const int);
|
||||
|
||||
/// arguments: DIM, d1d, q1d
|
||||
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
|
||||
|
||||
template <int DIM, int D1D, int Q1D> static void AddSpecialization()
|
||||
{
|
||||
ApplyPAKernels::Specialization<DIM, D1D, Q1D>::Add();
|
||||
}
|
||||
|
||||
struct Kernels { Kernels(); };
|
||||
|
||||
private:
|
||||
void SetupPA(const FiniteElementSpace &fes, FaceType type);
|
||||
};
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#include <ceed/types.h>
|
||||
|
||||
/// A structure used to pass additional data to f_build_conv and f_apply_conv
|
||||
struct ConvectionContext {
|
||||
@@ -91,7 +92,7 @@ CEED_QFUNCTION(f_build_conv_const)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for building quadrature data for a convection operator
|
||||
@@ -167,7 +168,7 @@ CEED_QFUNCTION(f_build_conv_quad)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for applying a conv operator
|
||||
@@ -233,7 +234,7 @@ CEED_QFUNCTION(f_apply_conv)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for applying a conv operator
|
||||
@@ -381,7 +382,7 @@ CEED_QFUNCTION(f_apply_conv_mf_const)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
CEED_QFUNCTION(f_apply_conv_mf_quad)(void *ctx, CeedInt Q,
|
||||
@@ -525,5 +526,5 @@ CEED_QFUNCTION(f_apply_conv_mf_quad)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include <ceed/types.h>
|
||||
|
||||
/// A structure used to pass additional data to f_build_diff and f_apply_diff
|
||||
struct DiffusionContext { CeedInt dim, space_dim, vdim; CeedScalar coeff; };
|
||||
@@ -85,7 +85,7 @@ CEED_QFUNCTION(f_build_diff_const)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for building quadrature data for a diffusion operator
|
||||
@@ -161,7 +161,7 @@ CEED_QFUNCTION(f_build_diff_quad)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for applying a diff operator
|
||||
@@ -241,7 +241,7 @@ CEED_QFUNCTION(f_apply_diff)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for applying a diff operator
|
||||
@@ -394,7 +394,7 @@ CEED_QFUNCTION(f_apply_diff_mf_const)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
CEED_QFUNCTION(f_apply_diff_mf_quad)(void *ctx, CeedInt Q,
|
||||
@@ -549,5 +549,5 @@ CEED_QFUNCTION(f_apply_diff_mf_quad)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include <ceed/types.h>
|
||||
|
||||
/// A structure used to pass additional data to f_build_diff and f_apply_diff
|
||||
struct MassContext { CeedInt dim, space_dim, vdim; CeedScalar coeff; };
|
||||
@@ -53,7 +53,7 @@ CEED_QFUNCTION(f_build_mass_const)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for building quadrature data for a mass operator with a
|
||||
@@ -95,7 +95,7 @@ CEED_QFUNCTION(f_build_mass_quad)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for applying a mass operator
|
||||
@@ -135,7 +135,7 @@ CEED_QFUNCTION(f_apply_mass)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for applying a diff operator
|
||||
@@ -199,7 +199,7 @@ CEED_QFUNCTION(f_apply_mass_mf_const)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
CEED_QFUNCTION(f_apply_mass_mf_quad)(void *ctx, CeedInt Q,
|
||||
@@ -266,5 +266,5 @@ CEED_QFUNCTION(f_apply_mass_mf_quad)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#include <ceed/types.h>
|
||||
|
||||
/// A structure used to pass additional data to f_build_conv and f_apply_conv
|
||||
struct NLConvectionContext { CeedInt dim, space_dim, vdim; CeedScalar coeff; };
|
||||
@@ -87,7 +88,7 @@ CEED_QFUNCTION(f_build_conv_const)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for building quadrature data for a convection operator
|
||||
@@ -167,7 +168,7 @@ CEED_QFUNCTION(f_build_conv_quad)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for applying a conv operator
|
||||
@@ -247,7 +248,7 @@ CEED_QFUNCTION(f_apply_conv)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
/// libCEED Q-function for applying a conv operator
|
||||
@@ -362,7 +363,7 @@ CEED_QFUNCTION(f_apply_conv_mf_const)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
CEED_QFUNCTION(f_apply_conv_mf_quad)(void *ctx, CeedInt Q,
|
||||
@@ -475,5 +476,5 @@ CEED_QFUNCTION(f_apply_conv_mf_quad)(void *ctx, CeedInt Q,
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
return CEED_ERROR_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -18,10 +18,21 @@
|
||||
|
||||
#include <ceed.h>
|
||||
|
||||
#if !CEED_VERSION_GE(0,12,0)
|
||||
#if !CEED_VERSION_GE(0, 12, 0)
|
||||
#error MFEM requires a libCEED version >= 0.12.0
|
||||
#endif
|
||||
|
||||
#if !CEED_VERSION_GE(0, 13, 0)
|
||||
#define CeedOperatorCreateComposite(ceed, op) \
|
||||
CeedCompositeOperatorCreate((ceed), (op))
|
||||
#define CeedOperatorCompositeAddSub(op, sub) \
|
||||
CeedCompositeOperatorAddSub((op), (sub))
|
||||
#define CeedOperatorCompositeGetNumSub(op, num) \
|
||||
CeedCompositeOperatorGetNumSub((op), (num))
|
||||
#define CeedOperatorCompositeGetSubList(op, list) \
|
||||
CeedCompositeOperatorGetSubList((op), (list))
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
|
||||
@@ -83,7 +83,7 @@ public:
|
||||
}
|
||||
|
||||
// Create composite CeedOperator
|
||||
CeedCompositeOperatorCreate(internal::ceed, &oper);
|
||||
CeedOperatorCreateComposite(internal::ceed, &oper);
|
||||
|
||||
// Create each sub-CeedOperator
|
||||
sub_ops.reserve(element_indices.size());
|
||||
@@ -101,7 +101,7 @@ public:
|
||||
int nelem = *count[value.first];
|
||||
sub_op->Assemble(info, fes, ir, nelem, indices, Q);
|
||||
sub_ops.push_back(sub_op);
|
||||
CeedCompositeOperatorAddSub(oper, sub_op->GetCeedOperator());
|
||||
CeedOperatorCompositeAddSub(oper, sub_op->GetCeedOperator());
|
||||
}
|
||||
|
||||
const int ndofs = fes.GetVDim() * fes.GetNDofs();
|
||||
|
||||
@@ -140,11 +140,7 @@ int CeedOperatorGetActiveField(CeedOperator oper, CeedOperatorField *field)
|
||||
CeedOperator *subops;
|
||||
if (isComposite)
|
||||
{
|
||||
#if CEED_VERSION_GE(0, 10, 2)
|
||||
ierr = CeedCompositeOperatorGetSubList(oper, &subops); PCeedChk(ierr);
|
||||
#else
|
||||
ierr = CeedOperatorGetSubList(oper, &subops); PCeedChk(ierr);
|
||||
#endif
|
||||
ierr = CeedOperatorCompositeGetSubList(oper, &subops); PCeedChk(ierr);
|
||||
ierr = CeedOperatorGetQFunction(subops[0], &qf); PCeedChk(ierr);
|
||||
}
|
||||
else
|
||||
@@ -171,7 +167,11 @@ int CeedOperatorGetActiveField(CeedOperator oper, CeedOperatorField *field)
|
||||
for (int i = 0; i < numinputfields; ++i)
|
||||
{
|
||||
ierr = CeedOperatorFieldGetVector(inputfields[i], &if_vector); PCeedChk(ierr);
|
||||
if (if_vector == CEED_VECTOR_ACTIVE)
|
||||
bool is_active = if_vector == CEED_VECTOR_ACTIVE;
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedVectorDestroy(&if_vector); PCeedChk(ierr);
|
||||
#endif
|
||||
if (is_active)
|
||||
{
|
||||
if (found)
|
||||
{
|
||||
|
||||
@@ -228,7 +228,7 @@ void AddToCompositeOperator(BilinearFormIntegrator *integ, CeedOperator op)
|
||||
{
|
||||
if (integ->SupportsCeed())
|
||||
{
|
||||
CeedCompositeOperatorAddSub(op, integ->GetCeedOp().GetCeedOperator());
|
||||
CeedOperatorCompositeAddSub(op, integ->GetCeedOp().GetCeedOperator());
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -240,7 +240,7 @@ CeedOperator CreateCeedCompositeOperatorFromBilinearForm(BilinearForm &form)
|
||||
{
|
||||
int ierr;
|
||||
CeedOperator op;
|
||||
ierr = CeedCompositeOperatorCreate(internal::ceed, &op); PCeedChk(ierr);
|
||||
ierr = CeedOperatorCreateComposite(internal::ceed, &op); PCeedChk(ierr);
|
||||
|
||||
MFEM_VERIFY(form.GetBBFI()->Size() == 0,
|
||||
"Not implemented for this integrator!");
|
||||
@@ -271,18 +271,13 @@ CeedOperator CoarsenCeedCompositeOperator(
|
||||
MFEM_ASSERT(isComposite, "");
|
||||
|
||||
CeedOperator op_coarse;
|
||||
ierr = CeedCompositeOperatorCreate(internal::ceed,
|
||||
ierr = CeedOperatorCreateComposite(internal::ceed,
|
||||
&op_coarse); PCeedChk(ierr);
|
||||
|
||||
int nsub;
|
||||
CeedOperator *subops;
|
||||
#if CEED_VERSION_GE(0, 10, 2)
|
||||
ierr = CeedCompositeOperatorGetNumSub(op, &nsub); PCeedChk(ierr);
|
||||
ierr = CeedCompositeOperatorGetSubList(op, &subops); PCeedChk(ierr);
|
||||
#else
|
||||
ierr = CeedOperatorGetNumSub(op, &nsub); PCeedChk(ierr);
|
||||
ierr = CeedOperatorGetSubList(op, &subops); PCeedChk(ierr);
|
||||
#endif
|
||||
ierr = CeedOperatorCompositeGetNumSub(op, &nsub); PCeedChk(ierr);
|
||||
ierr = CeedOperatorCompositeGetSubList(op, &subops); PCeedChk(ierr);
|
||||
for (int isub=0; isub<nsub; ++isub)
|
||||
{
|
||||
CeedOperator subop = subops[isub];
|
||||
@@ -294,7 +289,7 @@ CeedOperator CoarsenCeedCompositeOperator(
|
||||
// refcounted by existing objects
|
||||
ierr = CeedBasisDestroy(&basis_coarse); PCeedChk(ierr);
|
||||
ierr = CeedBasisDestroy(&basis_c2f); PCeedChk(ierr);
|
||||
ierr = CeedCompositeOperatorAddSub(op_coarse, subop_coarse);
|
||||
ierr = CeedOperatorCompositeAddSub(op_coarse, subop_coarse);
|
||||
PCeedChk(ierr);
|
||||
ierr = CeedOperatorDestroy(&subop_coarse); PCeedChk(ierr);
|
||||
}
|
||||
|
||||
@@ -81,12 +81,27 @@ int CeedSingleOperatorFullAssemble(CeedOperator op, SparseMatrix *out)
|
||||
ierr = CeedOperatorFieldGetVector(input_fields[i], &vec); PCeedChk(ierr);
|
||||
if (vec == CEED_VECTOR_ACTIVE)
|
||||
{
|
||||
ierr = CeedOperatorFieldGetBasis(input_fields[i], &basisin);
|
||||
PCeedChk(ierr);
|
||||
CeedBasis basis;
|
||||
ierr = CeedOperatorFieldGetBasis(input_fields[i], &basis); PCeedChk(ierr);
|
||||
if (!basisin)
|
||||
{
|
||||
ierr = CeedBasisReferenceCopy(basis, &basisin); PCeedChk(ierr);
|
||||
}
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedBasisDestroy(&basis); PCeedChk(ierr);
|
||||
#endif
|
||||
ierr = CeedBasisGetNumComponents(basisin, &ncomp); PCeedChk(ierr);
|
||||
ierr = CeedBasisGetDimension(basisin, &dim); PCeedChk(ierr);
|
||||
ierr = CeedOperatorFieldGetElemRestriction(input_fields[i], &rstrin);
|
||||
CeedElemRestriction rstr;
|
||||
ierr = CeedOperatorFieldGetElemRestriction(input_fields[i], &rstr);
|
||||
PCeedChk(ierr);
|
||||
if (!rstrin)
|
||||
{
|
||||
ierr = CeedElemRestrictionReferenceCopy(rstr, &rstrin); PCeedChk(ierr);
|
||||
}
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedElemRestrictionDestroy(&rstr); PCeedChk(ierr);
|
||||
#endif
|
||||
CeedEvalMode emode;
|
||||
ierr = CeedQFunctionFieldGetEvalMode(qffields[i], &emode);
|
||||
PCeedChk(ierr);
|
||||
@@ -112,6 +127,9 @@ int CeedSingleOperatorFullAssemble(CeedOperator op, SparseMatrix *out)
|
||||
break; // Caught by QF Assembly
|
||||
}
|
||||
}
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedVectorDestroy(&vec); PCeedChk(ierr);
|
||||
#endif
|
||||
}
|
||||
|
||||
// Determine active output basis
|
||||
@@ -127,11 +145,25 @@ int CeedSingleOperatorFullAssemble(CeedOperator op, SparseMatrix *out)
|
||||
ierr = CeedOperatorFieldGetVector(output_fields[i], &vec); PCeedChk(ierr);
|
||||
if (vec == CEED_VECTOR_ACTIVE)
|
||||
{
|
||||
ierr = CeedOperatorFieldGetBasis(output_fields[i], &basisout);
|
||||
PCeedChk(ierr);
|
||||
ierr = CeedOperatorFieldGetElemRestriction(output_fields[i], &rstrout);
|
||||
PCeedChk(ierr);
|
||||
CeedBasis basis;
|
||||
ierr = CeedOperatorFieldGetBasis(output_fields[i], &basis); PCeedChk(ierr);
|
||||
if (!basisout)
|
||||
{
|
||||
ierr = CeedBasisReferenceCopy(basis, &basisout); PCeedChk(ierr);
|
||||
}
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedBasisDestroy(&basis); PCeedChk(ierr);
|
||||
#endif
|
||||
CeedElemRestriction rstr;
|
||||
ierr = CeedOperatorFieldGetElemRestriction(output_fields[i], &rstr);
|
||||
PCeedChk(ierr);
|
||||
if (!rstrout)
|
||||
{
|
||||
ierr = CeedElemRestrictionReferenceCopy(rstr, &rstrout); PCeedChk(ierr);
|
||||
}
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedElemRestrictionDestroy(&rstr); PCeedChk(ierr);
|
||||
#endif
|
||||
CeedEvalMode emode;
|
||||
ierr = CeedQFunctionFieldGetEvalMode(qffields[i], &emode);
|
||||
PCeedChk(ierr);
|
||||
@@ -157,6 +189,9 @@ int CeedSingleOperatorFullAssemble(CeedOperator op, SparseMatrix *out)
|
||||
break; // Caught by QF Assembly
|
||||
}
|
||||
}
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedVectorDestroy(&vec); PCeedChk(ierr);
|
||||
#endif
|
||||
}
|
||||
|
||||
CeedInt nelem, elemsize, nqpts;
|
||||
@@ -200,7 +235,11 @@ int CeedSingleOperatorFullAssemble(CeedOperator op, SparseMatrix *out)
|
||||
PCeedChk(ierr);
|
||||
|
||||
CeedInt layout[3];
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedElemRestrictionGetELayout(rstr_q, layout); PCeedChk(ierr);
|
||||
#else
|
||||
ierr = CeedElemRestrictionGetELayout(rstr_q, &layout); PCeedChk(ierr);
|
||||
#endif
|
||||
ierr = CeedElemRestrictionDestroy(&rstr_q); PCeedChk(ierr);
|
||||
|
||||
// enforce structurally symmetric for later elimination
|
||||
@@ -285,6 +324,10 @@ int CeedSingleOperatorFullAssemble(CeedOperator op, SparseMatrix *out)
|
||||
ierr = CeedVectorRestoreArrayRead(assembledqf, &assembledqfarray);
|
||||
PCeedChk(ierr);
|
||||
ierr = CeedVectorDestroy(&assembledqf); PCeedChk(ierr);
|
||||
ierr = CeedElemRestrictionDestroy(&rstrin); PCeedChk(ierr);
|
||||
ierr = CeedElemRestrictionDestroy(&rstrout); PCeedChk(ierr);
|
||||
ierr = CeedBasisDestroy(&basisin); PCeedChk(ierr);
|
||||
ierr = CeedBasisDestroy(&basisout); PCeedChk(ierr);
|
||||
ierr = CeedHackFree(&emodein); PCeedChk(ierr);
|
||||
ierr = CeedHackFree(&emodeout); PCeedChk(ierr);
|
||||
|
||||
@@ -310,13 +353,8 @@ int CeedOperatorFullAssemble(CeedOperator op, SparseMatrix **mat)
|
||||
{
|
||||
CeedInt numsub;
|
||||
CeedOperator *subops;
|
||||
#if CEED_VERSION_GE(0, 10, 2)
|
||||
CeedCompositeOperatorGetNumSub(op, &numsub);
|
||||
ierr = CeedCompositeOperatorGetSubList(op, &subops); PCeedChk(ierr);
|
||||
#else
|
||||
CeedOperatorGetNumSub(op, &numsub);
|
||||
ierr = CeedOperatorGetSubList(op, &subops); PCeedChk(ierr);
|
||||
#endif
|
||||
ierr = CeedOperatorCompositeGetNumSub(op, &numsub); PCeedChk(ierr);
|
||||
ierr = CeedOperatorCompositeGetSubList(op, &subops); PCeedChk(ierr);
|
||||
for (int i = 0; i < numsub; ++i)
|
||||
{
|
||||
ierr = CeedSingleOperatorFullAssemble(subops[i], out); PCeedChk(ierr);
|
||||
|
||||
@@ -120,7 +120,11 @@ int CeedATPMGElemRestriction(int order,
|
||||
}
|
||||
ierr = CeedVectorRestoreArray(in_lvec, &lvec_data); PCeedChk(ierr);
|
||||
CeedInt in_layout[3];
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedElemRestrictionGetELayout(er_in, in_layout); PCeedChk(ierr);
|
||||
#else
|
||||
ierr = CeedElemRestrictionGetELayout(er_in, &in_layout); PCeedChk(ierr);
|
||||
#endif
|
||||
if (in_layout[0] == 0 && in_layout[1] == 0 && in_layout[2] == 0)
|
||||
{
|
||||
return CeedError(ceed, 1, "Cannot interpret e-vector ordering of given"
|
||||
@@ -664,7 +668,11 @@ int CeedATPMGOperator(CeedOperator oper, int order_reduction,
|
||||
|
||||
for (int i = 0; i < numinputfields; ++i)
|
||||
{
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
const char * fieldname;
|
||||
#else
|
||||
char * fieldname;
|
||||
#endif
|
||||
ierr = CeedQFunctionFieldGetName(inputqfields[i], &fieldname); PCeedChk(ierr);
|
||||
if (if_vector[i] == CEED_VECTOR_ACTIVE)
|
||||
{
|
||||
@@ -676,10 +684,19 @@ int CeedATPMGOperator(CeedOperator oper, int order_reduction,
|
||||
ierr = CeedOperatorSetField(coper, fieldname, er_input[i], basis_input[i],
|
||||
if_vector[i]); PCeedChk(ierr);
|
||||
}
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedVectorDestroy(&if_vector[i]); PCeedChk(ierr);
|
||||
ierr = CeedElemRestrictionDestroy(&er_input[i]); PCeedChk(ierr);
|
||||
ierr = CeedBasisDestroy(&basis_input[i]); PCeedChk(ierr);
|
||||
#endif
|
||||
}
|
||||
for (int i = 0; i < numoutputfields; ++i)
|
||||
{
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
const char * fieldname;
|
||||
#else
|
||||
char * fieldname;
|
||||
#endif
|
||||
ierr = CeedQFunctionFieldGetName(outputqfields[i], &fieldname); PCeedChk(ierr);
|
||||
if (of_vector[i] == CEED_VECTOR_ACTIVE)
|
||||
{
|
||||
@@ -691,6 +708,11 @@ int CeedATPMGOperator(CeedOperator oper, int order_reduction,
|
||||
ierr = CeedOperatorSetField(coper, fieldname, er_output[i], basis_output[i],
|
||||
of_vector[i]); PCeedChk(ierr);
|
||||
}
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedVectorDestroy(&of_vector[i]); PCeedChk(ierr);
|
||||
ierr = CeedElemRestrictionDestroy(&er_output[i]); PCeedChk(ierr);
|
||||
ierr = CeedBasisDestroy(&basis_output[i]); PCeedChk(ierr);
|
||||
#endif
|
||||
}
|
||||
delete [] er_input;
|
||||
delete [] er_output;
|
||||
@@ -741,7 +763,9 @@ int CeedOperatorGetOrder(CeedOperator oper, CeedInt * order)
|
||||
int P1d;
|
||||
ierr = CeedBasisGetNumNodes1D(basis, &P1d); PCeedChk(ierr);
|
||||
*order = P1d - 1;
|
||||
|
||||
#if CEED_VERSION_GE(0, 13, 0)
|
||||
ierr = CeedBasisDestroy(&basis); PCeedChk(ierr);
|
||||
#endif
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
@@ -563,6 +563,26 @@ void GradientGridFunctionCoefficient::Eval(
|
||||
}
|
||||
}
|
||||
|
||||
void GradientGridFunctionCoefficient::Project(QuadratureFunction &qf)
|
||||
{
|
||||
const FiniteElementSpace &fes = *GridFunc->FESpace();
|
||||
const Mesh &mesh = *fes.GetMesh();
|
||||
const int sdim = mesh.SpaceDimension();
|
||||
const int gf_vdim = fes.GetVDim(); // assumed to be 1 in this class
|
||||
qf.SetVDim(sdim*gf_vdim);
|
||||
if (mesh.GetNE() == 0) { return; }
|
||||
// All mesh element must be the same type:
|
||||
MFEM_VERIFY(mesh.GetNumGeometries(mesh.Dimension()) == 1,
|
||||
"All mesh elements must be the same type!");
|
||||
const IntegrationRule &ir = qf.GetIntRule(0);
|
||||
// All elements must use the same quadrature rule:
|
||||
MFEM_VERIFY(qf.Size() == sdim*gf_vdim*ir.GetNPoints()*mesh.GetNE(),
|
||||
"All mesh elements must use the same quadrature rule!");
|
||||
// QuadratureFunction uses the layout qf_vdim x nq x ne, i.e.
|
||||
// gf_vdim x sdim x nq x nq, so we need to request QVectorLayout::byVDIM:
|
||||
GridFunc->GetGradients(ir, qf, QVectorLayout::byVDIM);
|
||||
}
|
||||
|
||||
CurlGridFunctionCoefficient::CurlGridFunctionCoefficient(
|
||||
const GridFunction *gf)
|
||||
: VectorCoefficient(0)
|
||||
@@ -1109,6 +1129,41 @@ real_t InnerProductCoefficient::Eval(ElementTransformation &T,
|
||||
return va * vb;
|
||||
}
|
||||
|
||||
void InnerProductCoefficient::Project(QuadratureFunction &qf)
|
||||
{
|
||||
MFEM_VERIFY(a->GetVDim() == b->GetVDim(),
|
||||
"Incompatible vector coefficients: a->GetVDim(): "
|
||||
<< a->GetVDim() << ", b->GetVDim(): " << b->GetVDim());
|
||||
|
||||
const int vdim = a->GetVDim();
|
||||
MFEM_VERIFY(vdim >= 1, "invalid vdim: " << vdim);
|
||||
|
||||
// When running on device, make sure the output data is allocated before any
|
||||
// local temporary data to reduce potential heap fragmentation:
|
||||
auto dot_d = qf.Write();
|
||||
|
||||
QuadratureFunction qf_a(qf.GetSpace(), vdim);
|
||||
QuadratureFunction qf_b(qf.GetSpace(), vdim);
|
||||
|
||||
a->Project(qf_a);
|
||||
b->Project(qf_b);
|
||||
|
||||
auto a_d = qf_a.Read();
|
||||
auto b_d = qf_b.Read();
|
||||
|
||||
mfem::forall(qf.GetSpace()->GetSize(), [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
const real_t *ai = a_d + i*vdim;
|
||||
const real_t *bi = b_d + i*vdim;
|
||||
real_t dot = ai[0]*bi[0];
|
||||
for (int d = 1; d < vdim; d++)
|
||||
{
|
||||
dot += ai[d]*bi[d];
|
||||
}
|
||||
dot_d[i] = dot;
|
||||
});
|
||||
}
|
||||
|
||||
VectorRotProductCoefficient::VectorRotProductCoefficient(VectorCoefficient &A,
|
||||
VectorCoefficient &B)
|
||||
: a(&A), b(&B), va(A.GetVDim()), vb(B.GetVDim())
|
||||
|
||||
@@ -897,6 +897,9 @@ public:
|
||||
void Eval(DenseMatrix &M, ElementTransformation &T,
|
||||
const IntegrationRule &ir) override;
|
||||
|
||||
/// @copydoc VectorCoefficient::Project(QuadratureFunction &)
|
||||
void Project(QuadratureFunction &qf) override;
|
||||
|
||||
virtual ~GradientGridFunctionCoefficient() { }
|
||||
};
|
||||
|
||||
@@ -1774,6 +1777,9 @@ public:
|
||||
/// Evaluate the coefficient at @a ip.
|
||||
real_t Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip) override;
|
||||
|
||||
/// @copydoc Coefficient::Project(QuadratureFunction &)
|
||||
void Project(QuadratureFunction &qf) override;
|
||||
};
|
||||
|
||||
/// Scalar coefficient defined as a cross product of two vectors in the xy-plane.
|
||||
|
||||
@@ -259,6 +259,30 @@ inline void FaceIdxToVolIdx3D(const int index, const int size1d,
|
||||
i = yz_plane ? level : _i;
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
inline int FaceIdxToVolIdx(int dim, int i, int size1d, int face0, int face1,
|
||||
int side, int orientation)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
int ix, iy;
|
||||
internal::FaceIdxToVolIdx2D(i, size1d, face0, face1, side, ix, iy);
|
||||
return ix + iy*size1d;
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
int ix, iy, iz;
|
||||
internal::FaceIdxToVolIdx3D(i, size1d, face0, face1, side, orientation,
|
||||
ix, iy, iz);
|
||||
return ix + size1d*iy + size1d*size1d*iz;
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT_KERNEL("Invalid dimension");
|
||||
return -1;
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace internal
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
+2
-4
@@ -2456,8 +2456,7 @@ RT_FECollection::RT_FECollection(const int order, const int dim,
|
||||
const char *cb_name = BasisType::Name(cb_type); // this may abort
|
||||
MFEM_ABORT("unknown closed BasisType: " << cb_name);
|
||||
}
|
||||
if (Quadrature1D::CheckOpen(op_type) == Quadrature1D::Invalid &&
|
||||
ob_type != BasisType::IntegratedGLL)
|
||||
if (Quadrature1D::CheckOpen(op_type) == Quadrature1D::Invalid)
|
||||
{
|
||||
const char *ob_name = BasisType::Name(ob_type); // this may abort
|
||||
MFEM_ABORT("unknown open BasisType: " << ob_name);
|
||||
@@ -2784,8 +2783,7 @@ ND_FECollection::ND_FECollection(const int p, const int dim,
|
||||
int cp_type = BasisType::GetQuadrature1D(cb_type);
|
||||
|
||||
// Error checking
|
||||
if (Quadrature1D::CheckOpen(op_type) == Quadrature1D::Invalid &&
|
||||
ob_type != BasisType::IntegratedGLL)
|
||||
if (Quadrature1D::CheckOpen(op_type) == Quadrature1D::Invalid)
|
||||
{
|
||||
const char *ob_name = BasisType::Name(ob_type);
|
||||
MFEM_ABORT("Invalid open basis point type: " << ob_name);
|
||||
|
||||
+1
-1
@@ -224,7 +224,7 @@ struct DerefineMatrixOpMultFunctor
|
||||
sum += sign * bsptr[boptr[k] + i + j * block_height] *
|
||||
xptr[this->IndexX(col, vdim, k)];
|
||||
}
|
||||
#if defined(__CUDA_ARCH__) or defined(__HIP_DEVICE_COMPILE__)
|
||||
#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)
|
||||
if (Atomic)
|
||||
{
|
||||
atomicAdd(yptr + this->IndexY(row, vdim), sum);
|
||||
|
||||
+6
-2
@@ -683,8 +683,12 @@ public:
|
||||
NURBSExtension *GetNURBSext() { return NURBSext; }
|
||||
NURBSExtension *StealNURBSext();
|
||||
|
||||
bool Conforming() const { return mesh->Conforming() && cP == NULL; }
|
||||
bool Nonconforming() const { return mesh->Nonconforming() || cP != NULL; }
|
||||
bool Conforming() const
|
||||
{
|
||||
return NURBSext != NULL ||
|
||||
(mesh->Conforming() && cP == NULL);
|
||||
}
|
||||
bool Nonconforming() const { return !Conforming(); }
|
||||
|
||||
/** Set the prolongation operator of the space to an arbitrary sparse matrix,
|
||||
creating a copy of the argument. */
|
||||
|
||||
+46
-2
@@ -68,7 +68,7 @@ GridFunction::GridFunction(Mesh *m, std::istream &input)
|
||||
Vector::Load(input, fes->GetVSize());
|
||||
|
||||
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
|
||||
if (fes->Nonconforming() &&
|
||||
if (fes->Nonconforming() && fes->GetMesh()->ncmesh &&
|
||||
fes->GetMesh()->ncmesh->IsLegacyLoaded())
|
||||
{
|
||||
LegacyNCReorder();
|
||||
@@ -1374,6 +1374,50 @@ void GridFunction::GetVectorGradientHat(
|
||||
MultAtB(loc_data_mat, dshape, gh);
|
||||
}
|
||||
|
||||
void GridFunction::GetGradients(const IntegrationRule &ir, Vector &grad,
|
||||
QVectorLayout ql, MemoryType d_mt) const
|
||||
{
|
||||
const FiniteElement &fe = *fes->GetTypicalFE();
|
||||
const int dim = fe.GetDim();
|
||||
const int vdim = fes->GetVDim();
|
||||
const int NE = fes->GetNE();
|
||||
const int ND = fe.GetDof();
|
||||
const int NQ = ir.GetNPoints();
|
||||
|
||||
MemoryType my_d_mt = (d_mt != MemoryType::DEFAULT) ? d_mt :
|
||||
Device::GetDeviceMemoryType();
|
||||
|
||||
// ql == QVectorLayout::byNODES : NQ x VDIM x DIM x NE
|
||||
// ql == QVectorLayout::byVDIM : VDIM x DIM x NQPT x NE
|
||||
grad.SetSize(dim*vdim*NQ*NE, my_d_mt);
|
||||
|
||||
const QuadratureInterpolator &qi = *fes->GetQuadratureInterpolator(ir);
|
||||
qi.SetOutputLayout(ql);
|
||||
|
||||
const bool use_tensor_products = UsesTensorBasis(*fes);
|
||||
qi.DisableTensorProducts(!use_tensor_products);
|
||||
const ElementDofOrdering e_ordering = use_tensor_products ?
|
||||
ElementDofOrdering::LEXICOGRAPHIC :
|
||||
ElementDofOrdering::NATIVE;
|
||||
const Operator *elem_restr = fes->GetElementRestriction(e_ordering);
|
||||
|
||||
// Pre-compute the geometric factors in order to set the desired MemoryType
|
||||
// they use:
|
||||
fes->GetMesh()->GetGeometricFactors(
|
||||
ir, GeometricFactors::JACOBIANS, my_d_mt);
|
||||
|
||||
if (elem_restr) // currently, always true
|
||||
{
|
||||
Vector f_e(vdim*ND*NE, my_d_mt);
|
||||
elem_restr->Mult(*this, f_e);
|
||||
qi.PhysDerivatives(f_e, grad);
|
||||
}
|
||||
else
|
||||
{
|
||||
qi.PhysDerivatives(*this, grad);
|
||||
}
|
||||
}
|
||||
|
||||
real_t GridFunction::GetDivergence(ElementTransformation &T) const
|
||||
{
|
||||
DofTransformation doftrans;
|
||||
@@ -2624,7 +2668,7 @@ void GridFunction::ProjectBdrCoefficient(Coefficient *coeff[],
|
||||
}
|
||||
for (int i = 0; i < values_counter.Size(); i++)
|
||||
{
|
||||
MFEM_ASSERT(bool(values_counter[i]) == ess_vdofs_marker[i],
|
||||
MFEM_ASSERT(bool(values_counter[i]) == bool(ess_vdofs_marker[i]),
|
||||
"internal error");
|
||||
}
|
||||
#endif
|
||||
|
||||
+31
-2
@@ -153,7 +153,8 @@ public:
|
||||
/// Shortcut for calling SetFromTrueDofs() with GetTrueVector() as argument.
|
||||
void SetFromTrueVector() { SetFromTrueDofs(GetTrueVector()); }
|
||||
|
||||
/// Returns the values in the vertices of i'th element for dimension vdim.
|
||||
/** @brief Returns the values at the vertices of element @a i for the 1-based
|
||||
dimension vdim. */
|
||||
void GetNodalValues(int i, Array<real_t> &nval, int vdim = 1) const;
|
||||
|
||||
/** @name Element index Get Value Methods
|
||||
@@ -308,7 +309,8 @@ public:
|
||||
/// For a vector grid function, makes sure that the ordering is byNODES.
|
||||
void ReorderByNodes();
|
||||
|
||||
/// Return the values as a vector on mesh vertices for dimension vdim.
|
||||
/** @brief Returns the values as a vector at mesh vertices, for the 1-based
|
||||
dimension vdim. */
|
||||
void GetNodalValues(Vector &nval, int vdim = 1) const;
|
||||
|
||||
void GetVectorFieldNodalValues(Vector &val, int comp) const;
|
||||
@@ -359,6 +361,33 @@ public:
|
||||
variable. */
|
||||
void GetVectorGradientHat(ElementTransformation &T, DenseMatrix &gh) const;
|
||||
|
||||
/** @brief Evaluate the gradients of the GridFunction at the given quadrature
|
||||
points, @a ir, in all mesh elements. */
|
||||
/** This method assumes that all mesh elements are the same type and that the
|
||||
IntegrationRule @a ir is consistent with that type of element.
|
||||
|
||||
@param[in] ir Quadrature points at which the gradients are to be
|
||||
evaluated.
|
||||
@param[out] grad Output vector of size `SDIM*VDIM*NQ*NE` where `SDIM` is
|
||||
the spatial dimention of the mesh, `VDIM` is the vector
|
||||
dimension of the GridFunction, `NQ` is the number of
|
||||
quadrature points in @a ir, and `NE` is the number of
|
||||
elements in the mesh. The layout of @a grad is
|
||||
determined by the parameter @a ql: when @a ql is
|
||||
QVectorLayout::byNODES, the layout is
|
||||
`NQ x VDIM x SDIM x NE`; when @a ql is
|
||||
QVectorLayout::byVDIM, the layout is
|
||||
`VDIM x SDIM x NQ x NE`.
|
||||
@param[in] ql Determines the layout of the output vector @a grad; see
|
||||
the description of @a grad for details.
|
||||
@param[in] d_mt MemoryType to use for allocating the output vector
|
||||
@a grad, as well the GeometricFactors and temporary
|
||||
vector used by the method. By default, the current
|
||||
device memory type is used. */
|
||||
void GetGradients(const IntegrationRule &ir, Vector &grad,
|
||||
QVectorLayout ql = QVectorLayout::byNODES,
|
||||
MemoryType d_mt = MemoryType::DEFAULT) const;
|
||||
|
||||
/** Compute $ (\int_{\Omega} (*this) \psi_i)/(\int_{\Omega} \psi_i) $,
|
||||
where $ \psi_i $ are the basis functions for the FE space of avgs.
|
||||
Both FE spaces should be scalar and on the same mesh. */
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -15,6 +15,86 @@
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
CurlCurlIntegrator::CurlCurlIntegrator() : Q(nullptr), DQ(nullptr), MQ(nullptr)
|
||||
{
|
||||
static Kernels kernels;
|
||||
}
|
||||
|
||||
CurlCurlIntegrator::CurlCurlIntegrator(Coefficient &q,
|
||||
const IntegrationRule *ir)
|
||||
: BilinearFormIntegrator(ir), Q(&q), DQ(nullptr), MQ(nullptr)
|
||||
{
|
||||
static Kernels kernels;
|
||||
}
|
||||
|
||||
CurlCurlIntegrator::CurlCurlIntegrator(DiagonalMatrixCoefficient &dq,
|
||||
const IntegrationRule *ir)
|
||||
: BilinearFormIntegrator(ir), Q(nullptr), DQ(&dq), MQ(nullptr)
|
||||
{
|
||||
static Kernels kernels;
|
||||
}
|
||||
|
||||
CurlCurlIntegrator::CurlCurlIntegrator(MatrixCoefficient &mq,
|
||||
const IntegrationRule *ir)
|
||||
: BilinearFormIntegrator(ir), Q(nullptr), DQ(nullptr), MQ(&mq)
|
||||
{
|
||||
static Kernels kernels;
|
||||
}
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
|
||||
CurlCurlIntegrator::Kernels::Kernels()
|
||||
{
|
||||
CurlCurlIntegrator::AddSpecialization<3, 2, 3>();
|
||||
CurlCurlIntegrator::AddSpecialization<3, 3, 4>();
|
||||
CurlCurlIntegrator::AddSpecialization<3, 4, 5>();
|
||||
CurlCurlIntegrator::AddSpecialization<3, 5, 6>();
|
||||
}
|
||||
|
||||
CurlCurlIntegrator::ApplyKernelType
|
||||
CurlCurlIntegrator::ApplyPAKernels::Fallback(int DIM, int, int)
|
||||
{
|
||||
if (DIM == 2) { return internal::PACurlCurlApply2D; }
|
||||
else if (DIM == 3)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
return internal::SmemPACurlCurlApply3D;
|
||||
}
|
||||
else
|
||||
{
|
||||
return internal::PACurlCurlApply3D;
|
||||
}
|
||||
}
|
||||
else { MFEM_ABORT(""); }
|
||||
}
|
||||
|
||||
CurlCurlIntegrator::DiagonalKernelType
|
||||
CurlCurlIntegrator::DiagonalPAKernels::Fallback(int DIM, int, int)
|
||||
{
|
||||
if (DIM == 2)
|
||||
{
|
||||
return internal::PACurlCurlAssembleDiagonal2D;
|
||||
}
|
||||
else if (DIM == 3)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
return internal::SmemPACurlCurlAssembleDiagonal3D;
|
||||
}
|
||||
else
|
||||
{
|
||||
return internal::PACurlCurlAssembleDiagonal3D;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
void CurlCurlIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
// Assumes tensor-product elements
|
||||
@@ -77,129 +157,16 @@ void CurlCurlIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
|
||||
void CurlCurlIntegrator::AssembleDiagonalPA(Vector& diag)
|
||||
{
|
||||
if (dim == 3)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
const int ID = (dofs1D << 4) | quad1D;
|
||||
switch (ID)
|
||||
{
|
||||
case 0x23:
|
||||
return internal::SmemPACurlCurlAssembleDiagonal3D<2,3>(
|
||||
dofs1D,
|
||||
quad1D,
|
||||
symmetric, ne,
|
||||
mapsO->B, mapsC->B,
|
||||
mapsO->G, mapsC->G,
|
||||
pa_data, diag);
|
||||
case 0x34:
|
||||
return internal::SmemPACurlCurlAssembleDiagonal3D<3,4>(
|
||||
dofs1D,
|
||||
quad1D,
|
||||
symmetric, ne,
|
||||
mapsO->B, mapsC->B,
|
||||
mapsO->G, mapsC->G,
|
||||
pa_data, diag);
|
||||
case 0x45:
|
||||
return internal::SmemPACurlCurlAssembleDiagonal3D<4,5>(
|
||||
dofs1D,
|
||||
quad1D,
|
||||
symmetric, ne,
|
||||
mapsO->B, mapsC->B,
|
||||
mapsO->G, mapsC->G,
|
||||
pa_data, diag);
|
||||
case 0x56:
|
||||
return internal::SmemPACurlCurlAssembleDiagonal3D<5,6>(
|
||||
dofs1D,
|
||||
quad1D,
|
||||
symmetric, ne,
|
||||
mapsO->B, mapsC->B,
|
||||
mapsO->G, mapsC->G,
|
||||
pa_data, diag);
|
||||
default:
|
||||
return internal::SmemPACurlCurlAssembleDiagonal3D(
|
||||
dofs1D, quad1D,
|
||||
symmetric, ne,
|
||||
mapsO->B, mapsC->B,
|
||||
mapsO->G, mapsC->G,
|
||||
pa_data, diag);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
internal::PACurlCurlAssembleDiagonal3D(dofs1D, quad1D, symmetric, ne,
|
||||
mapsO->B, mapsC->B,
|
||||
mapsO->G, mapsC->G,
|
||||
pa_data, diag);
|
||||
}
|
||||
}
|
||||
else if (dim == 2)
|
||||
{
|
||||
internal::PACurlCurlAssembleDiagonal2D(dofs1D, quad1D, ne,
|
||||
mapsO->B, mapsC->G, pa_data, diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unsupported dimension!");
|
||||
}
|
||||
DiagonalPAKernels::Run(dim, dofs1D, quad1D, dofs1D, quad1D, symmetric, ne,
|
||||
mapsO->B, mapsC->B, mapsO->G, mapsC->G, pa_data,
|
||||
diag);
|
||||
}
|
||||
|
||||
void CurlCurlIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (dim == 3)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
const int ID = (dofs1D << 4) | quad1D;
|
||||
switch (ID)
|
||||
{
|
||||
case 0x23:
|
||||
return internal::SmemPACurlCurlApply3D<2,3>(
|
||||
dofs1D, quad1D,
|
||||
symmetric, ne,
|
||||
mapsO->B, mapsC->B, mapsO->Bt, mapsC->Bt,
|
||||
mapsC->G, mapsC->Gt, pa_data, x, y);
|
||||
case 0x34:
|
||||
return internal::SmemPACurlCurlApply3D<3,4>(
|
||||
dofs1D, quad1D,
|
||||
symmetric, ne,
|
||||
mapsO->B, mapsC->B, mapsO->Bt, mapsC->Bt,
|
||||
mapsC->G, mapsC->Gt, pa_data, x, y);
|
||||
case 0x45:
|
||||
return internal::SmemPACurlCurlApply3D<4,5>(
|
||||
dofs1D, quad1D,
|
||||
symmetric, ne,
|
||||
mapsO->B, mapsC->B, mapsO->Bt, mapsC->Bt,
|
||||
mapsC->G, mapsC->Gt, pa_data, x, y);
|
||||
case 0x56:
|
||||
return internal::SmemPACurlCurlApply3D<5,6>(
|
||||
dofs1D, quad1D,
|
||||
symmetric, ne,
|
||||
mapsO->B, mapsC->B, mapsO->Bt, mapsC->Bt,
|
||||
mapsC->G, mapsC->Gt, pa_data, x, y);
|
||||
default:
|
||||
return internal::SmemPACurlCurlApply3D(
|
||||
dofs1D, quad1D, symmetric, ne,
|
||||
mapsO->B, mapsC->B, mapsO->Bt, mapsC->Bt,
|
||||
mapsC->G, mapsC->Gt, pa_data, x, y);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
internal::PACurlCurlApply3D(dofs1D, quad1D, symmetric, ne, mapsO->B, mapsC->B,
|
||||
mapsO->Bt, mapsC->Bt, mapsC->G, mapsC->Gt,
|
||||
pa_data, x, y);
|
||||
}
|
||||
}
|
||||
else if (dim == 2)
|
||||
{
|
||||
internal::PACurlCurlApply2D(dofs1D, quad1D, ne, mapsO->B, mapsO->Bt,
|
||||
mapsC->G, mapsC->Gt, pa_data, x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unsupported dimension!");
|
||||
}
|
||||
ApplyPAKernels::Run(dim, dofs1D, quad1D, dofs1D, quad1D, symmetric, ne,
|
||||
mapsO->B, mapsC->B, mapsO->Bt, mapsC->Bt, mapsC->G,
|
||||
mapsC->Gt, pa_data, x, y, false);
|
||||
}
|
||||
|
||||
void CurlCurlIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
|
||||
@@ -209,61 +176,9 @@ void CurlCurlIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
|
||||
auto absO = mapsO->Abs();
|
||||
auto absC = mapsC->Abs();
|
||||
|
||||
if (dim == 3)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
const int ID = (dofs1D << 4) | quad1D;
|
||||
switch (ID)
|
||||
{
|
||||
case 0x23:
|
||||
return internal::SmemPACurlCurlApply3D<2,3>(
|
||||
dofs1D, quad1D,
|
||||
symmetric, ne,
|
||||
absO.B, absC.B, absO.Bt, absC.Bt,
|
||||
absC.G, absC.Gt, abs_pa_data, x, y, true);
|
||||
case 0x34:
|
||||
return internal::SmemPACurlCurlApply3D<3,4>(
|
||||
dofs1D, quad1D,
|
||||
symmetric, ne,
|
||||
absO.B, absC.B, absO.Bt, absC.Bt,
|
||||
absC.G, absC.Gt, abs_pa_data, x, y, true);
|
||||
case 0x45:
|
||||
return internal::SmemPACurlCurlApply3D<4,5>(
|
||||
dofs1D, quad1D,
|
||||
symmetric, ne,
|
||||
absO.B, absC.B, absO.Bt, absC.Bt,
|
||||
absC.G, absC.Gt, abs_pa_data, x, y, true);
|
||||
case 0x56:
|
||||
return internal::SmemPACurlCurlApply3D<5,6>(
|
||||
dofs1D, quad1D,
|
||||
symmetric, ne,
|
||||
absO.B, absC.B, absO.Bt, absC.Bt,
|
||||
absC.G, absC.Gt, abs_pa_data, x, y, true);
|
||||
default:
|
||||
return internal::SmemPACurlCurlApply3D<0,0>(
|
||||
dofs1D, quad1D, symmetric, ne,
|
||||
absO.B, absC.B, absO.Bt, absC.Bt,
|
||||
absC.G, absC.Gt, abs_pa_data, x, y, true);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
internal::PACurlCurlApply3D<0,0>(
|
||||
dofs1D, quad1D, symmetric, ne,
|
||||
absO.B, absC.B, absO.Bt, absC.Bt, absC.G, absC.Gt,
|
||||
abs_pa_data, x, y, true);
|
||||
}
|
||||
}
|
||||
else if (dim == 2)
|
||||
{
|
||||
internal::PACurlCurlApply2D(dofs1D, quad1D, ne, absO.B, absO.Bt,
|
||||
absC.G, absC.Gt, abs_pa_data, x, y, true);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unsupported dimension!");
|
||||
}
|
||||
ApplyPAKernels::Run(dim, dofs1D, quad1D, dofs1D, quad1D, symmetric, ne,
|
||||
absO.B, absC.B, absO.Bt, absC.Bt, absC.G, absC.Gt,
|
||||
abs_pa_data, x, y, true);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -0,0 +1,500 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_BILININTEG_DGDIFFUSION_KERNELS_HPP
|
||||
#define MFEM_BILININTEG_DGDIFFUSION_KERNELS_HPP
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../mesh/face_nbr_geom.hpp"
|
||||
#include "../fe/face_map_utils.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace internal
|
||||
{
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PADGDiffusionApply2D(const int NF, const Array<real_t> &b,
|
||||
const Array<real_t> &bt,
|
||||
const Array<real_t> &g,
|
||||
const Array<real_t> >, const real_t sigma,
|
||||
const Vector &pa_data, const Vector &x_,
|
||||
const Vector &dxdn_, Vector &y_, Vector &dydn_,
|
||||
const int d1d = 0, const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
auto B_ = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G_ = Reshape(g.Read(), Q1D, D1D);
|
||||
|
||||
auto pa =
|
||||
Reshape(pa_data.Read(), 6, Q1D, NF); // (q, 1/h, J00, J01, J10, J11)
|
||||
|
||||
auto x = Reshape(x_.Read(), D1D, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, 2, NF);
|
||||
auto dxdn = Reshape(dxdn_.Read(), D1D, 2, NF);
|
||||
auto dydn = Reshape(dydn_.ReadWrite(), D1D, 2, NF);
|
||||
|
||||
const int NBX = std::max(D1D, Q1D);
|
||||
|
||||
mfem::forall_2D(NF, NBX, 2, [=] MFEM_HOST_DEVICE(int f) -> void
|
||||
{
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
|
||||
MFEM_SHARED real_t u0[max_D1D];
|
||||
MFEM_SHARED real_t u1[max_D1D];
|
||||
MFEM_SHARED real_t du0[max_D1D];
|
||||
MFEM_SHARED real_t du1[max_D1D];
|
||||
|
||||
MFEM_SHARED real_t Bu0[max_Q1D];
|
||||
MFEM_SHARED real_t Bu1[max_Q1D];
|
||||
MFEM_SHARED real_t Bdu0[max_Q1D];
|
||||
MFEM_SHARED real_t Bdu1[max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t r[max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t BG[2 * max_D1D * max_Q1D];
|
||||
DeviceMatrix B(BG, Q1D, D1D);
|
||||
DeviceMatrix G(BG + D1D * Q1D, Q1D, D1D);
|
||||
|
||||
if (MFEM_THREAD_ID(y) == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p, x, Q1D)
|
||||
{
|
||||
for (int d = 0; d < D1D; ++d)
|
||||
{
|
||||
B(p, d) = B_(p, d);
|
||||
G(p, d) = G_(p, d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// copy edge values to u0, u1 and copy edge normals to du0, du1
|
||||
MFEM_FOREACH_THREAD(side, y, 2)
|
||||
{
|
||||
real_t *u = (side == 0) ? u0 : u1;
|
||||
real_t *du = (side == 0) ? du0 : du1;
|
||||
MFEM_FOREACH_THREAD(d, x, D1D)
|
||||
{
|
||||
u[d] = x(d, side, f);
|
||||
du[d] = dxdn(d, side, f);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// eval @ quad points
|
||||
MFEM_FOREACH_THREAD(side, y, 2)
|
||||
{
|
||||
real_t *u = (side == 0) ? u0 : u1;
|
||||
real_t *du = (side == 0) ? du0 : du1;
|
||||
real_t *Bu = (side == 0) ? Bu0 : Bu1;
|
||||
real_t *Bdu = (side == 0) ? Bdu0 : Bdu1;
|
||||
|
||||
MFEM_FOREACH_THREAD(p, x, Q1D)
|
||||
{
|
||||
const real_t Je_side[] = {pa(2 + 2 * side, p, f),
|
||||
pa(2 + 2 * side + 1, p, f)
|
||||
};
|
||||
|
||||
Bu[p] = 0.0;
|
||||
Bdu[p] = 0.0;
|
||||
|
||||
for (int d = 0; d < D1D; ++d)
|
||||
{
|
||||
const real_t b = B(p, d);
|
||||
const real_t g = G(p, d);
|
||||
|
||||
Bu[p] += b * u[d];
|
||||
Bdu[p] += Je_side[0] * b * du[d] + Je_side[1] * g * u[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// term - < {Q du/dn}, [v] > + kappa * < {Q/h} [u], [v] >:
|
||||
if (MFEM_THREAD_ID(y) == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p, x, Q1D)
|
||||
{
|
||||
const real_t q = pa(0, p, f);
|
||||
const real_t hi = pa(1, p, f);
|
||||
const real_t jump = Bu0[p] - Bu1[p];
|
||||
const real_t avg = Bdu0[p] + Bdu1[p]; // = {Q du/dn} * w * det(J)
|
||||
r[p] = -avg + hi * q * jump;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(d, x, D1D)
|
||||
{
|
||||
real_t Br = 0.0;
|
||||
|
||||
for (int p = 0; p < Q1D; ++p)
|
||||
{
|
||||
Br += B(p, d) * r[p];
|
||||
}
|
||||
|
||||
u0[d] = Br; // overwrite u0, u1
|
||||
u1[d] = -Br;
|
||||
} // for d
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(side, y, 2)
|
||||
{
|
||||
real_t *du = (side == 0) ? du0 : du1;
|
||||
MFEM_FOREACH_THREAD(d, x, D1D) { du[d] = 0.0; }
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// term sigma * < [u], {Q dv/dn} >
|
||||
MFEM_FOREACH_THREAD(side, y, 2)
|
||||
{
|
||||
real_t *const du = (side == 0) ? du0 : du1;
|
||||
real_t *const u = (side == 0) ? u0 : u1;
|
||||
|
||||
MFEM_FOREACH_THREAD(d, x, D1D)
|
||||
{
|
||||
for (int p = 0; p < Q1D; ++p)
|
||||
{
|
||||
const real_t Je[] = {pa(2 + 2 * side, p, f),
|
||||
pa(2 + 2 * side + 1, p, f)
|
||||
};
|
||||
const real_t jump = Bu0[p] - Bu1[p];
|
||||
const real_t r_p = Je[0] * jump; // normal
|
||||
const real_t w_p = Je[1] * jump; // tangential
|
||||
du[d] += sigma * B(p, d) * r_p;
|
||||
u[d] += sigma * G(p, d) * w_p;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(side, y, 2)
|
||||
{
|
||||
real_t *u = (side == 0) ? u0 : u1;
|
||||
real_t *du = (side == 0) ? du0 : du1;
|
||||
MFEM_FOREACH_THREAD(d, x, D1D)
|
||||
{
|
||||
y(d, side, f) += u[d];
|
||||
dydn(d, side, f) += du[d];
|
||||
}
|
||||
}
|
||||
}); // mfem::forall
|
||||
}
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PADGDiffusionApply3D(const int NF, const Array<real_t> &b,
|
||||
const Array<real_t> &bt,
|
||||
const Array<real_t> &g,
|
||||
const Array<real_t> >, const real_t sigma,
|
||||
const Vector &pa_data, const Vector &x_,
|
||||
const Vector &dxdn_, Vector &y_, Vector &dydn_,
|
||||
const int d1d = 0, const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
auto B_ = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G_ = Reshape(g.Read(), Q1D, D1D);
|
||||
|
||||
// (J0[0], J0[1], J0[2], J1[0], J1[1], J1[2], q/h)
|
||||
auto pa = Reshape(pa_data.Read(), 7, Q1D, Q1D, NF);
|
||||
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, 2, NF);
|
||||
auto dxdn = Reshape(dxdn_.Read(), D1D, D1D, 2, NF);
|
||||
auto dydn = Reshape(dydn_.ReadWrite(), D1D, D1D, 2, NF);
|
||||
|
||||
const int NBX = std::max(D1D, Q1D);
|
||||
|
||||
mfem::forall_3D(NF, NBX, NBX, 2, [=] MFEM_HOST_DEVICE(int f) -> void
|
||||
{
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
|
||||
MFEM_SHARED real_t u0[max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t u1[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t du0[max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t du1[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t Gu0[max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t Gu1[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t Bu0[max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t Bu1[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t Bdu0[max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t Bdu1[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t kappa_Qh[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t nJe[2][max_Q1D][max_Q1D][3];
|
||||
MFEM_SHARED real_t BG[2 * max_D1D * max_Q1D];
|
||||
|
||||
// some buffers are reused multiple times, but for clarity have new names:
|
||||
real_t(*Bj0)[max_Q1D] = Bu0;
|
||||
real_t(*Bj1)[max_Q1D] = Bu1;
|
||||
real_t(*Bjn0)[max_Q1D] = Bdu0;
|
||||
real_t(*Bjn1)[max_Q1D] = Bdu1;
|
||||
real_t(*Gj0)[max_Q1D] = Gu0;
|
||||
real_t(*Gj1)[max_Q1D] = Gu1;
|
||||
|
||||
DeviceMatrix B(BG, Q1D, D1D);
|
||||
DeviceMatrix G(BG + D1D * Q1D, Q1D, D1D);
|
||||
|
||||
// copy face values to u0, u1 and copy normals to du0, du1
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
real_t(*u)[max_Q1D] = (side == 0) ? u0 : u1;
|
||||
real_t(*du)[max_Q1D] = (side == 0) ? du0 : du1;
|
||||
|
||||
MFEM_FOREACH_THREAD(d2, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d1, y, D1D)
|
||||
{
|
||||
u[d2][d1] = x(d1, d2, side,
|
||||
f); // copy transposed for better memory access
|
||||
du[d2][d1] = dxdn(d1, d2, side, f);
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_FOREACH_THREAD(p1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p2, y, Q1D)
|
||||
{
|
||||
for (int l = 0; l < 3; ++l)
|
||||
{
|
||||
nJe[side][p2][p1][l] = pa(3 * side + l, p1, p2, f);
|
||||
}
|
||||
|
||||
if (side == 0)
|
||||
{
|
||||
kappa_Qh[p2][p1] = pa(6, p1, p2, f);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (side == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d, y, D1D)
|
||||
{
|
||||
B(p, d) = B_(p, d);
|
||||
G(p, d) = G_(p, d);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// eval u and normal derivative @ quad points
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
real_t(*u)[max_Q1D] = (side == 0) ? u0 : u1;
|
||||
real_t(*du)[max_Q1D] = (side == 0) ? du0 : du1;
|
||||
real_t(*Bu)[max_Q1D] = (side == 0) ? Bu0 : Bu1;
|
||||
real_t(*Bdu)[max_Q1D] = (side == 0) ? Bdu0 : Bdu1;
|
||||
real_t(*Gu)[max_Q1D] = (side == 0) ? Gu0 : Gu1;
|
||||
|
||||
MFEM_FOREACH_THREAD(p1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d2, y, D1D)
|
||||
{
|
||||
real_t bu = 0.0;
|
||||
real_t bdu = 0.0;
|
||||
real_t gu = 0.0;
|
||||
|
||||
for (int d1 = 0; d1 < D1D; ++d1)
|
||||
{
|
||||
const real_t b = B(p1, d1);
|
||||
const real_t g = G(p1, d1);
|
||||
|
||||
bu += b * u[d2][d1];
|
||||
bdu += b * du[d2][d1];
|
||||
gu += g * u[d2][d1];
|
||||
}
|
||||
|
||||
Bu[p1][d2] = bu;
|
||||
Bdu[p1][d2] = bdu;
|
||||
Gu[p1][d2] = gu;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
real_t(*u)[max_Q1D] = (side == 0) ? u0 : u1;
|
||||
real_t(*du)[max_Q1D] = (side == 0) ? du0 : du1;
|
||||
real_t(*Bu)[max_Q1D] = (side == 0) ? Bu0 : Bu1;
|
||||
real_t(*Gu)[max_Q1D] = (side == 0) ? Gu0 : Gu1;
|
||||
real_t(*Bdu)[max_Q1D] = (side == 0) ? Bdu0 : Bdu1;
|
||||
|
||||
MFEM_FOREACH_THREAD(p2, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p1, y, Q1D)
|
||||
{
|
||||
const real_t *Je = nJe[side][p2][p1];
|
||||
|
||||
real_t bbu = 0.0;
|
||||
real_t bgu = 0.0;
|
||||
real_t gbu = 0.0;
|
||||
real_t bbdu = 0.0;
|
||||
|
||||
for (int d2 = 0; d2 < D1D; ++d2)
|
||||
{
|
||||
const real_t b = B(p2, d2);
|
||||
const real_t g = G(p2, d2);
|
||||
bbu += b * Bu[p1][d2];
|
||||
gbu += g * Bu[p1][d2];
|
||||
bgu += b * Gu[p1][d2];
|
||||
bbdu += b * Bdu[p1][d2];
|
||||
}
|
||||
|
||||
u[p2][p1] = bbu;
|
||||
// du <- Q du/dn * w * det(J)
|
||||
du[p2][p1] = Je[0] * bbdu + Je[1] * bgu + Je[2] * gbu;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
real_t(*Bj)[max_Q1D] = (side == 0) ? Bj0 : Bj1;
|
||||
real_t(*Bjn)[max_Q1D] = (side == 0) ? Bjn0 : Bjn1;
|
||||
real_t(*Gj)[max_Q1D] = (side == 0) ? Gj0 : Gj1;
|
||||
|
||||
MFEM_FOREACH_THREAD(d1, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p2, y, Q1D)
|
||||
{
|
||||
real_t bj = 0.0;
|
||||
real_t bjn = 0.0;
|
||||
real_t gj = 0.0;
|
||||
real_t br = 0.0;
|
||||
|
||||
for (int p1 = 0; p1 < Q1D; ++p1)
|
||||
{
|
||||
const real_t b = B(p1, d1);
|
||||
const real_t g = G(p1, d1);
|
||||
|
||||
const real_t *Je = nJe[side][p2][p1];
|
||||
|
||||
const real_t jump = u0[p2][p1] - u1[p2][p1];
|
||||
const real_t avg = du0[p2][p1] + du1[p2][p1];
|
||||
|
||||
// r = - < {Q du/dn}, [v] > + kappa * < {Q/h} [u], [v] >
|
||||
const real_t r = -avg + kappa_Qh[p2][p1] * jump;
|
||||
|
||||
// bj, gj, bjn contribute to sigma term
|
||||
bj += b * Je[0] * jump;
|
||||
gj += g * Je[1] * jump;
|
||||
bjn += b * Je[2] * jump;
|
||||
|
||||
br += b * r;
|
||||
}
|
||||
|
||||
Bj[d1][p2] = sigma * bj;
|
||||
Bjn[d1][p2] = sigma * bjn;
|
||||
|
||||
// group br and gj together since we will multiply them both by B
|
||||
// and then sum
|
||||
const real_t sgn = (side == 0) ? 1.0 : -1.0;
|
||||
Gj[d1][p2] = sgn * br + sigma * gj;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
real_t(*u)[max_Q1D] = (side == 0) ? u0 : u1;
|
||||
real_t(*du)[max_Q1D] = (side == 0) ? du0 : du1;
|
||||
real_t(*Bj)[max_Q1D] = (side == 0) ? Bj0 : Bj1;
|
||||
real_t(*Bjn)[max_Q1D] = (side == 0) ? Bjn0 : Bjn1;
|
||||
real_t(*Gj)[max_Q1D] = (side == 0) ? Gj0 : Gj1;
|
||||
|
||||
MFEM_FOREACH_THREAD(d2, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d1, y, D1D)
|
||||
{
|
||||
real_t bbj = 0.0;
|
||||
real_t gbj = 0.0;
|
||||
real_t bgj = 0.0;
|
||||
|
||||
for (int p2 = 0; p2 < Q1D; ++p2)
|
||||
{
|
||||
const real_t b = B(p2, d2);
|
||||
const real_t g = G(p2, d2);
|
||||
|
||||
bbj += b * Bj[d1][p2];
|
||||
bgj += b * Gj[d1][p2];
|
||||
gbj += g * Bjn[d1][p2];
|
||||
}
|
||||
|
||||
du[d2][d1] = bbj;
|
||||
u[d2][d1] = bgj + gbj;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// map back to y and dydn
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
const real_t(*u)[max_Q1D] = (side == 0) ? u0 : u1;
|
||||
const real_t(*du)[max_Q1D] = (side == 0) ? du0 : du1;
|
||||
|
||||
MFEM_FOREACH_THREAD(d2, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d1, y, D1D)
|
||||
{
|
||||
y(d1, d2, side, f) += u[d2][d1];
|
||||
dydn(d1, d2, side, f) += du[d2][d1];
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace internal
|
||||
|
||||
template <int DIM, int D1D, int Q1D>
|
||||
DGDiffusionIntegrator::ApplyKernelType
|
||||
DGDiffusionIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2)
|
||||
{
|
||||
return internal::PADGDiffusionApply2D<D1D, Q1D>;
|
||||
}
|
||||
else if constexpr (DIM == 3)
|
||||
{
|
||||
return internal::PADGDiffusionApply3D<D1D, Q1D>;
|
||||
}
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
} // namespace mfem
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
#endif
|
||||
@@ -11,42 +11,39 @@
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../mesh/face_nbr_geom.hpp"
|
||||
#include "../fe/face_map_utils.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
#include "../fe/face_map_utils.hpp"
|
||||
|
||||
using namespace std;
|
||||
#include "bilininteg_dgdiffusion_kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
static void PADGDiffusionSetup2D(const int Q1D,
|
||||
const int NE,
|
||||
const int NF,
|
||||
static void PADGDiffusionSetup2D(const int Q1D, const int NE, const int NF,
|
||||
const Array<real_t> &w,
|
||||
const GeometricFactors &el_geom,
|
||||
const FaceGeometricFactors &face_geom,
|
||||
const FaceNeighborGeometricFactors *nbr_geom,
|
||||
const Vector &q,
|
||||
const real_t sigma,
|
||||
const real_t kappa,
|
||||
Vector &pa_data,
|
||||
const Vector &q, const real_t sigma,
|
||||
const real_t kappa, Vector &pa_data,
|
||||
const Array<int> &face_info_)
|
||||
{
|
||||
const auto J_loc = Reshape(el_geom.J.Read(), Q1D, Q1D, 2, 2, NE);
|
||||
const auto detJe_loc = Reshape(el_geom.detJ.Read(), Q1D, Q1D, NE);
|
||||
|
||||
const int n_nbr = nbr_geom ? nbr_geom->num_neighbor_elems : 0;
|
||||
const auto J_shared = Reshape(nbr_geom ? nbr_geom->J.Read() : nullptr,
|
||||
Q1D, Q1D, 2, 2, n_nbr);
|
||||
const auto detJ_shared = Reshape(nbr_geom ? nbr_geom->detJ.Read() : nullptr,
|
||||
Q1D, Q1D, n_nbr);
|
||||
const auto J_shared =
|
||||
Reshape(nbr_geom ? nbr_geom->J.Read() : nullptr, Q1D, Q1D, 2, 2, n_nbr);
|
||||
const auto detJ_shared =
|
||||
Reshape(nbr_geom ? nbr_geom->detJ.Read() : nullptr, Q1D, Q1D, n_nbr);
|
||||
|
||||
const auto detJf = Reshape(face_geom.detJ.Read(), Q1D, NF);
|
||||
const auto n = Reshape(face_geom.normal.Read(), Q1D, 2, NF);
|
||||
|
||||
const bool const_q = (q.Size() == 1);
|
||||
const auto Q = const_q ? Reshape(q.Read(), 1,1) : Reshape(q.Read(), Q1D,NF);
|
||||
const auto Q =
|
||||
const_q ? Reshape(q.Read(), 1, 1) : Reshape(q.Read(), Q1D, NF);
|
||||
|
||||
const auto W = w.Read();
|
||||
|
||||
@@ -56,7 +53,7 @@ static void PADGDiffusionSetup2D(const int Q1D,
|
||||
// (q, 1/h, J0_0, J0_1, J1_0, J1_1)
|
||||
auto pa = Reshape(pa_data.Write(), 6, Q1D, NF);
|
||||
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE (int f) -> void
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE(int f) -> void
|
||||
{
|
||||
const int normal_dir[] = {face_info(0, f), face_info(1, f)};
|
||||
const int fid[] = {face_info(4, f), face_info(5, f)};
|
||||
@@ -74,7 +71,7 @@ static void PADGDiffusionSetup2D(const int Q1D,
|
||||
|
||||
for (int p = 0; p < Q1D; ++p)
|
||||
{
|
||||
const real_t Qp = const_q ? Q(0,0) : Q(p, f);
|
||||
const real_t Qp = const_q ? Q(0, 0) : Q(p, f);
|
||||
pa(0, p, f) = kappa * Qp * W[p] * detJf(p, f);
|
||||
|
||||
real_t hi = 0.0;
|
||||
@@ -85,17 +82,19 @@ static void PADGDiffusionSetup2D(const int Q1D,
|
||||
|
||||
// Always opposite direction in "native" ordering
|
||||
// Need to multiply the native=>lex0 with native=>lex1 and negate
|
||||
const int sgn = (side == 1) ? -1*sgn0*sgn1 : 1;
|
||||
const int sgn = (side == 1) ? -1 * sgn0 * sgn1 : 1;
|
||||
|
||||
const int e = el[side];
|
||||
const auto &J = (side == 1 && shared) ? J_shared : J_loc;
|
||||
const auto &detJ = (side == 1 && shared) ? detJ_shared : detJe_loc;
|
||||
|
||||
real_t nJi[2];
|
||||
nJi[0] = n(p,0,f)*J(i,j, 1,1, e) - n(p,1,f)*J(i,j,0,1,e);
|
||||
nJi[1] = -n(p,0,f)*J(i,j,1,0, e) + n(p,1,f)*J(i,j,0,0,e);
|
||||
nJi[0] =
|
||||
n(p, 0, f) * J(i, j, 1, 1, e) - n(p, 1, f) * J(i, j, 0, 1, e);
|
||||
nJi[1] =
|
||||
-n(p, 0, f) * J(i, j, 1, 0, e) + n(p, 1, f) * J(i, j, 0, 0, e);
|
||||
|
||||
const real_t dJe = detJ(i,j,e);
|
||||
const real_t dJe = detJ(i, j, e);
|
||||
const real_t dJf = detJf(p, f);
|
||||
|
||||
const real_t w = factor * Qp * W[p] * dJf / dJe;
|
||||
@@ -104,9 +103,9 @@ static void PADGDiffusionSetup2D(const int Q1D,
|
||||
const int ti = 1 - ni;
|
||||
|
||||
// Normal
|
||||
pa(2 + 2*side + 0, p, f) = w * nJi[ni];
|
||||
pa(2 + 2 * side + 0, p, f) = w * nJi[ni];
|
||||
// Tangential
|
||||
pa(2 + 2*side + 1, p, f) = sgn * w * nJi[ti];
|
||||
pa(2 + 2 * side + 1, p, f) = sgn * w * nJi[ti];
|
||||
|
||||
hi += factor * dJf / dJe;
|
||||
}
|
||||
@@ -122,47 +121,43 @@ static void PADGDiffusionSetup2D(const int Q1D,
|
||||
});
|
||||
}
|
||||
|
||||
static void PADGDiffusionSetup3D(const int Q1D,
|
||||
const int NE,
|
||||
const int NF,
|
||||
static void PADGDiffusionSetup3D(const int Q1D, const int NE, const int NF,
|
||||
const Array<real_t> &w,
|
||||
const GeometricFactors &el_geom,
|
||||
const FaceGeometricFactors &face_geom,
|
||||
const FaceNeighborGeometricFactors *nbr_geom,
|
||||
const Vector &q,
|
||||
const real_t sigma,
|
||||
const real_t kappa,
|
||||
Vector &pa_data,
|
||||
const Vector &q, const real_t sigma,
|
||||
const real_t kappa, Vector &pa_data,
|
||||
const Array<int> &face_info_)
|
||||
{
|
||||
const auto J_loc = Reshape(el_geom.J.Read(), Q1D, Q1D, Q1D, 3, 3, NE);
|
||||
const auto detJe_loc = Reshape(el_geom.detJ.Read(), Q1D, Q1D, Q1D, NE);
|
||||
|
||||
const int n_nbr = nbr_geom ? nbr_geom->num_neighbor_elems : 0;
|
||||
const auto J_shared = Reshape(nbr_geom ? nbr_geom->J.Read() : nullptr,
|
||||
Q1D, Q1D, Q1D, 3, 3, n_nbr);
|
||||
const auto detJ_shared = Reshape(nbr_geom ? nbr_geom->detJ.Read() : nullptr,
|
||||
Q1D, Q1D, Q1D, n_nbr);
|
||||
const auto J_shared = Reshape(nbr_geom ? nbr_geom->J.Read() : nullptr, Q1D,
|
||||
Q1D, Q1D, 3, 3, n_nbr);
|
||||
const auto detJ_shared =
|
||||
Reshape(nbr_geom ? nbr_geom->detJ.Read() : nullptr, Q1D, Q1D, Q1D, n_nbr);
|
||||
|
||||
const auto detJf = Reshape(face_geom.detJ.Read(), Q1D, Q1D, NF);
|
||||
const auto n = Reshape(face_geom.normal.Read(), Q1D, Q1D, 3, NF);
|
||||
|
||||
const bool const_q = (q.Size() == 1);
|
||||
const auto Q = const_q ? Reshape(q.Read(), 1, 1, 1)
|
||||
: Reshape(q.Read(), Q1D, Q1D, NF);
|
||||
const auto Q =
|
||||
const_q ? Reshape(q.Read(), 1, 1, 1) : Reshape(q.Read(), Q1D, Q1D, NF);
|
||||
|
||||
const auto W = Reshape(w.Read(), Q1D, Q1D);
|
||||
|
||||
// (perm[0], perm[1], perm[2], element_index, local_face_id, orientation)
|
||||
const auto face_info = Reshape(face_info_.Read(), 6, 2, NF);
|
||||
constexpr int _el_ = 3; // offset in face_info for element index
|
||||
constexpr int _el_ = 3; // offset in face_info for element index
|
||||
constexpr int _fid_ = 4; // offset in face_info for local face id
|
||||
constexpr int _or_ = 5; // offset in face_info for orientation
|
||||
constexpr int _or_ = 5; // offset in face_info for orientation
|
||||
|
||||
// (J00, J01, J02, J10, J11, J12, q/h)
|
||||
const auto pa = Reshape(pa_data.Write(), 7, Q1D, Q1D, NF);
|
||||
|
||||
mfem::forall_2D(NF, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int f) -> void
|
||||
mfem::forall_2D(NF, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int f) -> void
|
||||
{
|
||||
MFEM_SHARED int perm[2][3];
|
||||
MFEM_SHARED int el[2];
|
||||
@@ -172,10 +167,7 @@ static void PADGDiffusionSetup3D(const int Q1D,
|
||||
|
||||
MFEM_FOREACH_THREAD(side, x, 2)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(i, y, 3)
|
||||
{
|
||||
perm[side][i] = face_info(i, side, f);
|
||||
}
|
||||
MFEM_FOREACH_THREAD(i, y, 3) { perm[side][i] = face_info(i, side, f); }
|
||||
|
||||
if (MFEM_THREAD_ID(y) == 0)
|
||||
{
|
||||
@@ -200,16 +192,16 @@ static void PADGDiffusionSetup3D(const int Q1D,
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p2, y, Q1D)
|
||||
{
|
||||
const real_t Qp = const_q ? Q(0,0,0) : Q(p1, p2, f);
|
||||
const real_t dJf = detJf(p1,p2,f);
|
||||
const real_t Qp = const_q ? Q(0, 0, 0) : Q(p1, p2, f);
|
||||
const real_t dJf = detJf(p1, p2, f);
|
||||
|
||||
real_t hi = 0.0;
|
||||
|
||||
for (int side = 0; side < nsides; ++side)
|
||||
{
|
||||
int i, j, k;
|
||||
internal::FaceIdxToVolIdx3D(
|
||||
p1 + Q1D*p2, Q1D, fid[0], fid[1], side, ortn[1], i, j, k);
|
||||
internal::FaceIdxToVolIdx3D(p1 + Q1D * p2, Q1D, fid[0], fid[1],
|
||||
side, ortn[1], i, j, k);
|
||||
|
||||
const int e = el[side];
|
||||
const auto &J = shared[side] ? J_shared : J_loc;
|
||||
@@ -217,27 +209,45 @@ static void PADGDiffusionSetup3D(const int Q1D,
|
||||
|
||||
// *INDENT-OFF*
|
||||
real_t nJi[3];
|
||||
nJi[0] = ( -J(i,j,k, 1,2, e)*J(i,j,k, 2,1, e) + J(i,j,k, 1,1, e)*J(i,j,k, 2,2, e)) * n(p1,p2, 0, f)
|
||||
+ ( J(i,j,k, 0,2, e)*J(i,j,k, 2,1, e) - J(i,j,k, 0,1, e)*J(i,j,k, 2,2, e)) * n(p1,p2, 1, f)
|
||||
+ (-J(i,j,k, 0,2, e)*J(i,j,k, 1,1, e) + J(i,j,k, 0,1, e)*J(i,j,k, 1,2, e)) * n(p1,p2, 2, f);
|
||||
nJi[0] = (-J(i, j, k, 1, 2, e) * J(i, j, k, 2, 1, e) +
|
||||
J(i, j, k, 1, 1, e) * J(i, j, k, 2, 2, e)) *
|
||||
n(p1, p2, 0, f) +
|
||||
(J(i, j, k, 0, 2, e) * J(i, j, k, 2, 1, e) -
|
||||
J(i, j, k, 0, 1, e) * J(i, j, k, 2, 2, e)) *
|
||||
n(p1, p2, 1, f) +
|
||||
(-J(i, j, k, 0, 2, e) * J(i, j, k, 1, 1, e) +
|
||||
J(i, j, k, 0, 1, e) * J(i, j, k, 1, 2, e)) *
|
||||
n(p1, p2, 2, f);
|
||||
|
||||
nJi[1] = ( J(i,j,k, 1,2, e)*J(i,j,k, 2,0, e) - J(i,j,k, 1,0, e)*J(i,j,k, 2,2, e)) * n(p1,p2, 0, f)
|
||||
+ (-J(i,j,k, 0,2, e)*J(i,j,k, 2,0, e) + J(i,j,k, 0,0, e)*J(i,j,k, 2,2, e)) * n(p1,p2, 1, f)
|
||||
+ ( J(i,j,k, 0,2, e)*J(i,j,k, 1,0, e) - J(i,j,k, 0,0, e)*J(i,j,k, 1,2, e)) * n(p1,p2, 2, f);
|
||||
nJi[1] = (J(i, j, k, 1, 2, e) * J(i, j, k, 2, 0, e) -
|
||||
J(i, j, k, 1, 0, e) * J(i, j, k, 2, 2, e)) *
|
||||
n(p1, p2, 0, f) +
|
||||
(-J(i, j, k, 0, 2, e) * J(i, j, k, 2, 0, e) +
|
||||
J(i, j, k, 0, 0, e) * J(i, j, k, 2, 2, e)) *
|
||||
n(p1, p2, 1, f) +
|
||||
(J(i, j, k, 0, 2, e) * J(i, j, k, 1, 0, e) -
|
||||
J(i, j, k, 0, 0, e) * J(i, j, k, 1, 2, e)) *
|
||||
n(p1, p2, 2, f);
|
||||
|
||||
nJi[2] = ( -J(i,j,k, 1,1, e)*J(i,j,k, 2,0, e) + J(i,j,k, 1,0, e)*J(i,j,k, 2,1, e)) * n(p1,p2, 0, f)
|
||||
+ ( J(i,j,k, 0,1, e)*J(i,j,k, 2,0, e) - J(i,j,k, 0,0, e)*J(i,j,k, 2,1, e)) * n(p1,p2, 1, f)
|
||||
+ (-J(i,j,k, 0,1, e)*J(i,j,k, 1,0, e) + J(i,j,k, 0,0, e)*J(i,j,k, 1,1, e)) * n(p1,p2, 2, f);
|
||||
nJi[2] = (-J(i, j, k, 1, 1, e) * J(i, j, k, 2, 0, e) +
|
||||
J(i, j, k, 1, 0, e) * J(i, j, k, 2, 1, e)) *
|
||||
n(p1, p2, 0, f) +
|
||||
(J(i, j, k, 0, 1, e) * J(i, j, k, 2, 0, e) -
|
||||
J(i, j, k, 0, 0, e) * J(i, j, k, 2, 1, e)) *
|
||||
n(p1, p2, 1, f) +
|
||||
(-J(i, j, k, 0, 1, e) * J(i, j, k, 1, 0, e) +
|
||||
J(i, j, k, 0, 0, e) * J(i, j, k, 1, 1, e)) *
|
||||
n(p1, p2, 2, f);
|
||||
// *INDENT-ON*
|
||||
|
||||
const real_t dJe = detJe(i,j,k,e);
|
||||
const real_t dJe = detJe(i, j, k, e);
|
||||
const real_t val = factor * Qp * W(p1, p2) * dJf / dJe;
|
||||
|
||||
for (int d = 0; d < 3; ++d)
|
||||
{
|
||||
const int idx = std::abs(perm[side][d]) - 1;
|
||||
const int sgn = (perm[side][d] < 0) ? -1 : 1;
|
||||
pa(3*side + d, p1, p2, f) = sgn * val * nJi[idx];
|
||||
pa(3 * side + d, p1, p2, f) = sgn * val * nJi[idx];
|
||||
}
|
||||
|
||||
hi += factor * dJf / dJe;
|
||||
@@ -257,7 +267,8 @@ static void PADGDiffusionSetup3D(const int Q1D,
|
||||
}
|
||||
|
||||
static void PADGDiffusionSetupFaceInfo2D(const int nf, const Mesh &mesh,
|
||||
const FaceType type, Array<int> &face_info_)
|
||||
const FaceType type,
|
||||
Array<int> &face_info_)
|
||||
{
|
||||
const int ne = mesh.GetNE();
|
||||
|
||||
@@ -326,8 +337,7 @@ inline void FaceNormalPermutation(int perm[3], const int face_id)
|
||||
|
||||
// Assigns to perm the permutation as in FaceNormalPermutation for the second
|
||||
// element on the face but signed to indicate the sign of the normal derivative.
|
||||
inline void SignedFaceNormalPermutation(int perm[3],
|
||||
const int face_id1,
|
||||
inline void SignedFaceNormalPermutation(int perm[3], const int face_id1,
|
||||
const int face_id2,
|
||||
const int orientation)
|
||||
{
|
||||
@@ -386,17 +396,19 @@ inline void SignedFaceNormalPermutation(int perm[3],
|
||||
}
|
||||
|
||||
static void PADGDiffusionSetupFaceInfo3D(const int nf, const Mesh &mesh,
|
||||
const FaceType type, Array<int> &face_info_)
|
||||
const FaceType type,
|
||||
Array<int> &face_info_)
|
||||
{
|
||||
const int ne = mesh.GetNE();
|
||||
|
||||
int fidx = 0;
|
||||
// face_info array has 12 entries per face, 6 for each of the adjacent elements:
|
||||
// (perm[0], perm[1], perm[2], element_index, local_face_id, orientation)
|
||||
// face_info array has 12 entries per face, 6 for each of the adjacent
|
||||
// elements: (perm[0], perm[1], perm[2], element_index, local_face_id,
|
||||
// orientation)
|
||||
face_info_.SetSize(nf * 12);
|
||||
constexpr int _e_ = 3; // offset for element index
|
||||
constexpr int _e_ = 3; // offset for element index
|
||||
constexpr int _fid_ = 4; // offset for local face id
|
||||
constexpr int _or_ = 5; // offset for orientation
|
||||
constexpr int _or_ = 5; // offset for orientation
|
||||
|
||||
auto face_info = Reshape(face_info_.HostWrite(), 6, 2, nf);
|
||||
for (int f = 0; f < mesh.GetNumFaces(); ++f)
|
||||
@@ -408,9 +420,9 @@ static void PADGDiffusionSetupFaceInfo3D(const int nf, const Mesh &mesh,
|
||||
const int fid0 = f_info.element[0].local_face_id;
|
||||
const int or0 = f_info.element[0].orientation;
|
||||
|
||||
face_info( _e_, 0, fidx) = f_info.element[0].index;
|
||||
face_info(_e_, 0, fidx) = f_info.element[0].index;
|
||||
face_info(_fid_, 0, fidx) = fid0;
|
||||
face_info( _or_, 0, fidx) = or0;
|
||||
face_info(_or_, 0, fidx) = or0;
|
||||
|
||||
FaceNormalPermutation(&face_info(0, 0, fidx), fid0);
|
||||
|
||||
@@ -421,16 +433,17 @@ static void PADGDiffusionSetupFaceInfo3D(const int nf, const Mesh &mesh,
|
||||
|
||||
if (f_info.IsShared())
|
||||
{
|
||||
face_info( _e_, 1, fidx) = ne + f_info.element[1].index;
|
||||
face_info(_e_, 1, fidx) = ne + f_info.element[1].index;
|
||||
}
|
||||
else
|
||||
{
|
||||
face_info( _e_, 1, fidx) = f_info.element[1].index;
|
||||
face_info(_e_, 1, fidx) = f_info.element[1].index;
|
||||
}
|
||||
face_info(_fid_, 1, fidx) = fid1;
|
||||
face_info( _or_, 1, fidx) = or1;
|
||||
face_info(_or_, 1, fidx) = or1;
|
||||
|
||||
SignedFaceNormalPermutation(&face_info(0, 1, fidx), fid0, fid1, or1);
|
||||
SignedFaceNormalPermutation(&face_info(0, 1, fidx), fid0, fid1,
|
||||
or1);
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -448,8 +461,8 @@ static void PADGDiffusionSetupFaceInfo3D(const int nf, const Mesh &mesh,
|
||||
void DGDiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
|
||||
FaceType type)
|
||||
{
|
||||
const MemoryType mt = (pa_mt == MemoryType::DEFAULT) ?
|
||||
Device::GetDeviceMemoryType() : pa_mt;
|
||||
const MemoryType mt =
|
||||
(pa_mt == MemoryType::DEFAULT) ? Device::GetDeviceMemoryType() : pa_mt;
|
||||
|
||||
const int ne = fes.GetNE();
|
||||
nf = fes.GetNFbyType(type);
|
||||
@@ -458,16 +471,17 @@ void DGDiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
|
||||
Mesh &mesh = *fes.GetMesh();
|
||||
const Geometry::Type face_geom_type = mesh.GetTypicalFaceGeometry();
|
||||
const FiniteElement &el = *fes.GetTypicalTraceElement();
|
||||
const int ir_order = IntRule ? IntRule->GetOrder()
|
||||
const int ir_order = IntRule
|
||||
? IntRule->GetOrder()
|
||||
: GetRule(el.GetOrder(), face_geom_type).GetOrder();
|
||||
const IntegrationRule &ir = irs.Get(face_geom_type, ir_order);
|
||||
dim = mesh.Dimension();
|
||||
const int q1d = (ir.GetOrder() + 3)/2;
|
||||
MFEM_ASSERT(q1d == pow(real_t(ir.Size()), 1.0/(dim - 1)), "");
|
||||
const int q1d = (ir.GetOrder() + 3) / 2;
|
||||
MFEM_ASSERT(q1d == pow(real_t(ir.Size()), 1.0 / (dim - 1)), "");
|
||||
|
||||
const auto vol_ir = irs.Get(mesh.GetTypicalElementGeometry(), ir_order);
|
||||
const auto geom_flags = GeometricFactors::JACOBIANS |
|
||||
GeometricFactors::DETERMINANTS;
|
||||
const auto geom_flags =
|
||||
GeometricFactors::JACOBIANS | GeometricFactors::DETERMINANTS;
|
||||
const auto el_geom = mesh.GetGeometricFactors(vol_ir, geom_flags, mt);
|
||||
|
||||
std::unique_ptr<FaceNeighborGeometricFactors> nbr_geom;
|
||||
@@ -476,8 +490,8 @@ void DGDiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
|
||||
nbr_geom.reset(new FaceNeighborGeometricFactors(*el_geom));
|
||||
}
|
||||
|
||||
const auto face_geom_flags = FaceGeometricFactors::DETERMINANTS |
|
||||
FaceGeometricFactors::NORMALS;
|
||||
const auto face_geom_flags =
|
||||
FaceGeometricFactors::DETERMINANTS | FaceGeometricFactors::NORMALS;
|
||||
auto face_geom = mesh.GetFaceGeometricFactors(ir, face_geom_flags, type, mt);
|
||||
maps = &el.GetDofToQuad(ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
@@ -489,9 +503,18 @@ void DGDiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
|
||||
// Evaluate the coefficient at the face quadrature points.
|
||||
FaceQuadratureSpace fqs(mesh, ir, type);
|
||||
CoefficientVector q(fqs, CoefficientStorage::COMPRESSED);
|
||||
if (Q) { q.Project(*Q); }
|
||||
else if (MQ) { MFEM_ABORT("Not yet implemented"); /* q.Project(*MQ); */ }
|
||||
else { q.SetConstant(1.0); }
|
||||
if (Q)
|
||||
{
|
||||
q.Project(*Q);
|
||||
}
|
||||
else if (MQ)
|
||||
{
|
||||
MFEM_ABORT("Not yet implemented"); /* q.Project(*MQ); */
|
||||
}
|
||||
else
|
||||
{
|
||||
q.SetConstant(1.0);
|
||||
}
|
||||
|
||||
Array<int> face_info;
|
||||
if (dim == 1)
|
||||
@@ -501,14 +524,16 @@ void DGDiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
|
||||
else if (dim == 2)
|
||||
{
|
||||
PADGDiffusionSetupFaceInfo2D(nf, mesh, type, face_info);
|
||||
PADGDiffusionSetup2D(quad1D, ne, nf, ir.GetWeights(), *el_geom, *face_geom,
|
||||
nbr_geom.get(), q, sigma, kappa, pa_data, face_info);
|
||||
PADGDiffusionSetup2D(quad1D, ne, nf, ir.GetWeights(), *el_geom,
|
||||
*face_geom, nbr_geom.get(), q, sigma, kappa, pa_data,
|
||||
face_info);
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
PADGDiffusionSetupFaceInfo3D(nf, mesh, type, face_info);
|
||||
PADGDiffusionSetup3D(quad1D, ne, nf, ir.GetWeights(), *el_geom, *face_geom,
|
||||
nbr_geom.get(), q, sigma, kappa, pa_data, face_info);
|
||||
PADGDiffusionSetup3D(quad1D, ne, nf, ir.GetWeights(), *el_geom,
|
||||
*face_geom, nbr_geom.get(), q, sigma, kappa, pa_data,
|
||||
face_info);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -524,529 +549,76 @@ void DGDiffusionIntegrator::AssemblePABoundaryFaces(
|
||||
SetupPA(fes, FaceType::Boundary);
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0> static
|
||||
void PADGDiffusionApply2D(const int NF,
|
||||
const Array<real_t> &b,
|
||||
const Array<real_t> &bt,
|
||||
const Array<real_t>& g,
|
||||
const Array<real_t>& gt,
|
||||
const real_t sigma,
|
||||
const Vector &pa_data,
|
||||
const Vector &x_,
|
||||
const Vector &dxdn_,
|
||||
Vector &y_,
|
||||
Vector &dydn_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
void DGDiffusionIntegrator::AddMultPAFaceNormalDerivatives(const Vector &x,
|
||||
const Vector &dxdn,
|
||||
Vector &y,
|
||||
Vector &dydn) const
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
auto B_ = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G_ = Reshape(g.Read(), Q1D, D1D);
|
||||
|
||||
auto pa = Reshape(pa_data.Read(), 6, Q1D, NF); // (q, 1/h, J00, J01, J10, J11)
|
||||
|
||||
auto x = Reshape(x_.Read(), D1D, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, 2, NF);
|
||||
auto dxdn = Reshape(dxdn_.Read(), D1D, 2, NF);
|
||||
auto dydn = Reshape(dydn_.ReadWrite(), D1D, 2, NF);
|
||||
|
||||
const int NBX = std::max(D1D, Q1D);
|
||||
|
||||
mfem::forall_2D(NF, NBX, 2, [=] MFEM_HOST_DEVICE (int f) -> void
|
||||
{
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
|
||||
MFEM_SHARED real_t u0[max_D1D];
|
||||
MFEM_SHARED real_t u1[max_D1D];
|
||||
MFEM_SHARED real_t du0[max_D1D];
|
||||
MFEM_SHARED real_t du1[max_D1D];
|
||||
|
||||
MFEM_SHARED real_t Bu0[max_Q1D];
|
||||
MFEM_SHARED real_t Bu1[max_Q1D];
|
||||
MFEM_SHARED real_t Bdu0[max_Q1D];
|
||||
MFEM_SHARED real_t Bdu1[max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t r[max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t BG[2*max_D1D*max_Q1D];
|
||||
DeviceMatrix B(BG, Q1D, D1D);
|
||||
DeviceMatrix G(BG + D1D*Q1D, Q1D, D1D);
|
||||
|
||||
if (MFEM_THREAD_ID(y) == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p,x,Q1D)
|
||||
{
|
||||
for (int d = 0; d < D1D; ++d)
|
||||
{
|
||||
B(p,d) = B_(p,d);
|
||||
G(p,d) = G_(p,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// copy edge values to u0, u1 and copy edge normals to du0, du1
|
||||
MFEM_FOREACH_THREAD(side,y,2)
|
||||
{
|
||||
real_t *u = (side == 0) ? u0 : u1;
|
||||
real_t *du = (side == 0) ? du0 : du1;
|
||||
MFEM_FOREACH_THREAD(d,x,D1D)
|
||||
{
|
||||
u[d] = x(d, side, f);
|
||||
du[d] = dxdn(d, side, f);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// eval @ quad points
|
||||
MFEM_FOREACH_THREAD(side,y,2)
|
||||
{
|
||||
real_t *u = (side == 0) ? u0 : u1;
|
||||
real_t *du = (side == 0) ? du0 : du1;
|
||||
real_t *Bu = (side == 0) ? Bu0 : Bu1;
|
||||
real_t *Bdu = (side == 0) ? Bdu0 : Bdu1;
|
||||
|
||||
MFEM_FOREACH_THREAD(p,x,Q1D)
|
||||
{
|
||||
const real_t Je_side[] = {pa(2 + 2*side, p, f), pa(2 + 2*side + 1, p, f)};
|
||||
|
||||
Bu[p] = 0.0;
|
||||
Bdu[p] = 0.0;
|
||||
|
||||
for (int d = 0; d < D1D; ++d)
|
||||
{
|
||||
const real_t b = B(p,d);
|
||||
const real_t g = G(p,d);
|
||||
|
||||
Bu[p] += b*u[d];
|
||||
Bdu[p] += Je_side[0] * b * du[d] + Je_side[1] * g * u[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// term - < {Q du/dn}, [v] > + kappa * < {Q/h} [u], [v] >:
|
||||
if (MFEM_THREAD_ID(y) == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p,x,Q1D)
|
||||
{
|
||||
const real_t q = pa(0, p, f);
|
||||
const real_t hi = pa(1, p, f);
|
||||
const real_t jump = Bu0[p] - Bu1[p];
|
||||
const real_t avg = Bdu0[p] + Bdu1[p]; // = {Q du/dn} * w * det(J)
|
||||
r[p] = -avg + hi * q * jump;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(d,x,D1D)
|
||||
{
|
||||
real_t Br = 0.0;
|
||||
|
||||
for (int p = 0; p < Q1D; ++p)
|
||||
{
|
||||
Br += B(p, d) * r[p];
|
||||
}
|
||||
|
||||
u0[d] = Br; // overwrite u0, u1
|
||||
u1[d] = -Br;
|
||||
} // for d
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
|
||||
MFEM_FOREACH_THREAD(side,y,2)
|
||||
{
|
||||
real_t *du = (side == 0) ? du0 : du1;
|
||||
MFEM_FOREACH_THREAD(d,x,D1D)
|
||||
{
|
||||
du[d] = 0.0;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// term sigma * < [u], {Q dv/dn} >
|
||||
MFEM_FOREACH_THREAD(side,y,2)
|
||||
{
|
||||
real_t * const du = (side == 0) ? du0 : du1;
|
||||
real_t * const u = (side == 0) ? u0 : u1;
|
||||
|
||||
MFEM_FOREACH_THREAD(d,x,D1D)
|
||||
{
|
||||
for (int p = 0; p < Q1D; ++p)
|
||||
{
|
||||
const real_t Je[] = {pa(2 + 2*side, p, f), pa(2 + 2*side + 1, p, f)};
|
||||
const real_t jump = Bu0[p] - Bu1[p];
|
||||
const real_t r_p = Je[0] * jump; // normal
|
||||
const real_t w_p = Je[1] * jump; // tangential
|
||||
du[d] += sigma * B(p, d) * r_p;
|
||||
u[d] += sigma * G(p, d) * w_p;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(side,y,2)
|
||||
{
|
||||
real_t *u = (side == 0) ? u0 : u1;
|
||||
real_t *du = (side == 0) ? du0 : du1;
|
||||
MFEM_FOREACH_THREAD(d,x,D1D)
|
||||
{
|
||||
y(d, side, f) += u[d];
|
||||
dydn(d, side, f) += du[d];
|
||||
}
|
||||
}
|
||||
}); // mfem::forall
|
||||
ApplyPAKernels::Run(dim, dofs1D, quad1D, nf, maps->B, maps->Bt, maps->G,
|
||||
maps->Gt, sigma, pa_data, x, dxdn, y, dydn, dofs1D,
|
||||
quad1D);
|
||||
}
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PADGDiffusionApply3D(const int NF,
|
||||
const Array<real_t>& b,
|
||||
const Array<real_t>& bt,
|
||||
const Array<real_t>& g,
|
||||
const Array<real_t>& gt,
|
||||
const real_t sigma,
|
||||
const Vector& pa_data,
|
||||
const Vector& x_,
|
||||
const Vector& dxdn_,
|
||||
Vector& y_,
|
||||
Vector& dydn_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
DGDiffusionIntegrator::DGDiffusionIntegrator(const real_t s, const real_t k)
|
||||
: sigma(s), kappa(k)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
auto B_ = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G_ = Reshape(g.Read(), Q1D, D1D);
|
||||
|
||||
// (J0[0], J0[1], J0[2], J1[0], J1[1], J1[2], q/h)
|
||||
auto pa = Reshape(pa_data.Read(), 7, Q1D, Q1D, NF);
|
||||
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, 2, NF);
|
||||
auto dxdn = Reshape(dxdn_.Read(), D1D, D1D, 2, NF);
|
||||
auto dydn = Reshape(dydn_.ReadWrite(), D1D, D1D, 2, NF);
|
||||
|
||||
const int NBX = std::max(D1D, Q1D);
|
||||
|
||||
mfem::forall_3D(NF, NBX, NBX, 2, [=] MFEM_HOST_DEVICE (int f) -> void
|
||||
{
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
|
||||
MFEM_SHARED real_t u0[max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t u1[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t du0[max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t du1[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t Gu0[max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t Gu1[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t Bu0[max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t Bu1[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t Bdu0[max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t Bdu1[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t kappa_Qh[max_Q1D][max_Q1D];
|
||||
|
||||
MFEM_SHARED real_t nJe[2][max_Q1D][max_Q1D][3];
|
||||
MFEM_SHARED real_t BG[2*max_D1D*max_Q1D];
|
||||
|
||||
// some buffers are reused multiple times, but for clarity have new names:
|
||||
real_t (*Bj0)[max_Q1D] = Bu0;
|
||||
real_t (*Bj1)[max_Q1D] = Bu1;
|
||||
real_t (*Bjn0)[max_Q1D] = Bdu0;
|
||||
real_t (*Bjn1)[max_Q1D] = Bdu1;
|
||||
real_t (*Gj0)[max_Q1D] = Gu0;
|
||||
real_t (*Gj1)[max_Q1D] = Gu1;
|
||||
|
||||
DeviceMatrix B(BG, Q1D, D1D);
|
||||
DeviceMatrix G(BG + D1D*Q1D, Q1D, D1D);
|
||||
|
||||
// copy face values to u0, u1 and copy normals to du0, du1
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
real_t (*u)[max_Q1D] = (side == 0) ? u0 : u1;
|
||||
real_t (*du)[max_Q1D] = (side == 0) ? du0 : du1;
|
||||
|
||||
MFEM_FOREACH_THREAD(d2, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d1, y, D1D)
|
||||
{
|
||||
u[d2][d1] = x(d1, d2, side, f); // copy transposed for better memory access
|
||||
du[d2][d1] = dxdn(d1, d2, side, f);
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_FOREACH_THREAD(p1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p2, y, Q1D)
|
||||
{
|
||||
for (int l=0; l < 3; ++l)
|
||||
{
|
||||
nJe[side][p2][p1][l] = pa(3*side + l, p1, p2, f);
|
||||
}
|
||||
|
||||
if (side == 0)
|
||||
{
|
||||
kappa_Qh[p2][p1] = pa(6, p1, p2, f);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (side == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d, y, D1D)
|
||||
{
|
||||
B(p, d) = B_(p, d);
|
||||
G(p, d) = G_(p, d);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// eval u and normal derivative @ quad points
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
real_t (*u)[max_Q1D] = (side == 0) ? u0 : u1;
|
||||
real_t (*du)[max_Q1D] = (side == 0) ? du0 : du1;
|
||||
real_t (*Bu)[max_Q1D] = (side == 0) ? Bu0 : Bu1;
|
||||
real_t (*Bdu)[max_Q1D] = (side == 0) ? Bdu0 : Bdu1;
|
||||
real_t (*Gu)[max_Q1D] = (side == 0) ? Gu0 : Gu1;
|
||||
|
||||
MFEM_FOREACH_THREAD(p1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d2, y, D1D)
|
||||
{
|
||||
real_t bu = 0.0;
|
||||
real_t bdu = 0.0;
|
||||
real_t gu = 0.0;
|
||||
|
||||
for (int d1=0; d1 < D1D; ++d1)
|
||||
{
|
||||
const real_t b = B(p1, d1);
|
||||
const real_t g = G(p1, d1);
|
||||
|
||||
bu += b * u[d2][d1];
|
||||
bdu += b * du[d2][d1];
|
||||
gu += g * u[d2][d1];
|
||||
}
|
||||
|
||||
Bu[p1][d2] = bu;
|
||||
Bdu[p1][d2] = bdu;
|
||||
Gu[p1][d2] = gu;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
real_t (*u)[max_Q1D] = (side == 0) ? u0 : u1;
|
||||
real_t (*du)[max_Q1D] = (side == 0) ? du0 : du1;
|
||||
real_t (*Bu)[max_Q1D] = (side == 0) ? Bu0 : Bu1;
|
||||
real_t (*Gu)[max_Q1D] = (side == 0) ? Gu0 : Gu1;
|
||||
real_t (*Bdu)[max_Q1D] = (side == 0) ? Bdu0 : Bdu1;
|
||||
|
||||
MFEM_FOREACH_THREAD(p2, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p1, y, Q1D)
|
||||
{
|
||||
const real_t * Je = nJe[side][p2][p1];
|
||||
|
||||
real_t bbu = 0.0;
|
||||
real_t bgu = 0.0;
|
||||
real_t gbu = 0.0;
|
||||
real_t bbdu = 0.0;
|
||||
|
||||
for (int d2 = 0; d2 < D1D; ++d2)
|
||||
{
|
||||
const real_t b = B(p2, d2);
|
||||
const real_t g = G(p2, d2);
|
||||
bbu += b * Bu[p1][d2];
|
||||
gbu += g * Bu[p1][d2];
|
||||
bgu += b * Gu[p1][d2];
|
||||
bbdu += b * Bdu[p1][d2];
|
||||
}
|
||||
|
||||
u[p2][p1] = bbu;
|
||||
// du <- Q du/dn * w * det(J)
|
||||
du[p2][p1] = Je[0] * bbdu + Je[1] * bgu + Je[2] * gbu;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
real_t (*Bj)[max_Q1D] = (side == 0) ? Bj0 : Bj1;
|
||||
real_t (*Bjn)[max_Q1D] = (side == 0) ? Bjn0 : Bjn1;
|
||||
real_t (*Gj)[max_Q1D] = (side == 0) ? Gj0 : Gj1;
|
||||
|
||||
MFEM_FOREACH_THREAD(d1, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(p2, y, Q1D)
|
||||
{
|
||||
real_t bj = 0.0;
|
||||
real_t bjn = 0.0;
|
||||
real_t gj = 0.0;
|
||||
real_t br = 0.0;
|
||||
|
||||
for (int p1 = 0; p1 < Q1D; ++p1)
|
||||
{
|
||||
const real_t b = B(p1, d1);
|
||||
const real_t g = G(p1, d1);
|
||||
|
||||
const real_t * Je = nJe[side][p2][p1];
|
||||
|
||||
const real_t jump = u0[p2][p1] - u1[p2][p1];
|
||||
const real_t avg = du0[p2][p1] + du1[p2][p1];
|
||||
|
||||
// r = - < {Q du/dn}, [v] > + kappa * < {Q/h} [u], [v] >
|
||||
const real_t r = -avg + kappa_Qh[p2][p1] * jump;
|
||||
|
||||
// bj, gj, bjn contribute to sigma term
|
||||
bj += b * Je[0] * jump;
|
||||
gj += g * Je[1] * jump;
|
||||
bjn += b * Je[2] * jump;
|
||||
|
||||
br += b * r;
|
||||
}
|
||||
|
||||
Bj[d1][p2] = sigma * bj;
|
||||
Bjn[d1][p2] = sigma * bjn;
|
||||
|
||||
// group br and gj together since we will multiply them both by B
|
||||
// and then sum
|
||||
const real_t sgn = (side == 0) ? 1.0 : -1.0;
|
||||
Gj[d1][p2] = sgn * br + sigma * gj;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
real_t (*u)[max_Q1D] = (side == 0) ? u0 : u1;
|
||||
real_t (*du)[max_Q1D] = (side == 0) ? du0 : du1;
|
||||
real_t (*Bj)[max_Q1D] = (side == 0) ? Bj0 : Bj1;
|
||||
real_t (*Bjn)[max_Q1D] = (side == 0) ? Bjn0 : Bjn1;
|
||||
real_t (*Gj)[max_Q1D] = (side == 0) ? Gj0 : Gj1;
|
||||
|
||||
MFEM_FOREACH_THREAD(d2, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d1, y, D1D)
|
||||
{
|
||||
real_t bbj = 0.0;
|
||||
real_t gbj = 0.0;
|
||||
real_t bgj = 0.0;
|
||||
|
||||
for (int p2 = 0; p2 < Q1D; ++p2)
|
||||
{
|
||||
const real_t b = B(p2, d2);
|
||||
const real_t g = G(p2, d2);
|
||||
|
||||
bbj += b * Bj[d1][p2];
|
||||
bgj += b * Gj[d1][p2];
|
||||
gbj += g * Bjn[d1][p2];
|
||||
}
|
||||
|
||||
du[d2][d1] = bbj;
|
||||
u[d2][d1] = bgj + gbj;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// map back to y and dydn
|
||||
MFEM_FOREACH_THREAD(side, z, 2)
|
||||
{
|
||||
const real_t (*u)[max_Q1D] = (side == 0) ? u0 : u1;
|
||||
const real_t (*du)[max_Q1D] = (side == 0) ? du0 : du1;
|
||||
|
||||
MFEM_FOREACH_THREAD(d2, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d1, y, D1D)
|
||||
{
|
||||
y(d1, d2, side, f) += u[d2][d1];
|
||||
dydn(d1, d2, side, f) += du[d2][d1];
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
static Kernels kernels;
|
||||
}
|
||||
|
||||
static void PADGDiffusionApply(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NF,
|
||||
const Array<real_t> &B,
|
||||
const Array<real_t> &Bt,
|
||||
const Array<real_t> &G,
|
||||
const Array<real_t> &Gt,
|
||||
const real_t sigma,
|
||||
const Vector &pa_data,
|
||||
const Vector &x,
|
||||
const Vector &dxdn,
|
||||
Vector &y,
|
||||
Vector &dydn)
|
||||
DGDiffusionIntegrator::DGDiffusionIntegrator(Coefficient &q, const real_t s,
|
||||
const real_t k)
|
||||
: DGDiffusionIntegrator(s, k)
|
||||
{
|
||||
Q = &q;
|
||||
}
|
||||
|
||||
DGDiffusionIntegrator::DGDiffusionIntegrator(MatrixCoefficient &q,
|
||||
const real_t s, const real_t k)
|
||||
: DGDiffusionIntegrator(s, k)
|
||||
{
|
||||
MQ = &q;
|
||||
}
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
|
||||
DGDiffusionIntegrator::ApplyKernelType
|
||||
DGDiffusionIntegrator::ApplyPAKernels::Fallback(int dim, int, int)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
auto kernel = PADGDiffusionApply2D<0,0>;
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x23: kernel = PADGDiffusionApply2D<2,3>; break;
|
||||
case 0x34: kernel = PADGDiffusionApply2D<3,4>; break;
|
||||
case 0x45: kernel = PADGDiffusionApply2D<4,5>; break;
|
||||
case 0x56: kernel = PADGDiffusionApply2D<5,6>; break;
|
||||
case 0x67: kernel = PADGDiffusionApply2D<6,7>; break;
|
||||
case 0x78: kernel = PADGDiffusionApply2D<7,8>; break;
|
||||
case 0x89: kernel = PADGDiffusionApply2D<8,9>; break;
|
||||
case 0x9A: kernel = PADGDiffusionApply2D<9,10>; break;
|
||||
}
|
||||
kernel(NF, B, Bt, G, Gt, sigma, pa_data, x, dxdn, y, dydn, D1D, Q1D);
|
||||
return internal::PADGDiffusionApply2D;
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
auto kernel = PADGDiffusionApply3D<0,0>;
|
||||
switch ((D1D << 4) | Q1D)
|
||||
{
|
||||
case 0x24: kernel = PADGDiffusionApply3D<2,4>; break;
|
||||
case 0x35: kernel = PADGDiffusionApply3D<3,5>; break;
|
||||
case 0x46: kernel = PADGDiffusionApply3D<4,6>; break;
|
||||
case 0x57: kernel = PADGDiffusionApply3D<5,7>; break;
|
||||
case 0x68: kernel = PADGDiffusionApply3D<6,8>; break;
|
||||
case 0x79: kernel = PADGDiffusionApply3D<7,9>; break;
|
||||
case 0x8A: kernel = PADGDiffusionApply3D<8,10>; break;
|
||||
case 0x9B: kernel = PADGDiffusionApply3D<9,11>; break;
|
||||
}
|
||||
kernel(NF, B, Bt, G, Gt, sigma, pa_data, x, dxdn, y, dydn, D1D, Q1D);
|
||||
return internal::PADGDiffusionApply3D;
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unsupported dimension");
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
}
|
||||
|
||||
void DGDiffusionIntegrator::AddMultPAFaceNormalDerivatives(
|
||||
const Vector &x, const Vector &dxdn, Vector &y, Vector &dydn) const
|
||||
DGDiffusionIntegrator::Kernels::Kernels()
|
||||
{
|
||||
PADGDiffusionApply(dim, dofs1D, quad1D, nf,
|
||||
maps->B, maps->Bt, maps->G, maps->Gt,
|
||||
sigma, pa_data, x, dxdn, y, dydn);
|
||||
DGDiffusionIntegrator::AddSpecialization<2, 2, 3>();
|
||||
DGDiffusionIntegrator::AddSpecialization<2, 3, 4>();
|
||||
DGDiffusionIntegrator::AddSpecialization<2, 4, 5>();
|
||||
DGDiffusionIntegrator::AddSpecialization<2, 5, 6>();
|
||||
DGDiffusionIntegrator::AddSpecialization<2, 6, 7>();
|
||||
DGDiffusionIntegrator::AddSpecialization<2, 7, 8>();
|
||||
DGDiffusionIntegrator::AddSpecialization<2, 8, 9>();
|
||||
DGDiffusionIntegrator::AddSpecialization<2, 9, 10>();
|
||||
|
||||
DGDiffusionIntegrator::AddSpecialization<3, 2, 4>();
|
||||
DGDiffusionIntegrator::AddSpecialization<3, 3, 5>();
|
||||
DGDiffusionIntegrator::AddSpecialization<3, 4, 6>();
|
||||
DGDiffusionIntegrator::AddSpecialization<3, 5, 7>();
|
||||
DGDiffusionIntegrator::AddSpecialization<3, 6, 8>();
|
||||
DGDiffusionIntegrator::AddSpecialization<3, 7, 9>();
|
||||
DGDiffusionIntegrator::AddSpecialization<3, 8, 10>();
|
||||
DGDiffusionIntegrator::AddSpecialization<3, 9, 11>();
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -0,0 +1,793 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef BILININTEG_DGTRACE_KERNELS_HPP
|
||||
#define BILININTEG_DGTRACE_KERNELS_HPP
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
#include "../restriction.hpp"
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace internal
|
||||
{
|
||||
|
||||
// PA DGTrace Apply 2D kernel for Gauss-Lobatto/Bernstein
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PADGTraceApply2D(const int NF, const Array<real_t> &b,
|
||||
const Array<real_t> &bt, const Vector &op_,
|
||||
const Vector &x_, Vector &y_, const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(bt.Read(), D1D, Q1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, 2, 2, NF);
|
||||
auto x = Reshape(x_.Read(), D1D, VDIM, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, VDIM, 2, NF);
|
||||
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE(int f)
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
// the following variables are evaluated at compile time
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
real_t u0[max_D1D][VDIM];
|
||||
real_t u1[max_D1D][VDIM];
|
||||
for (int d = 0; d < D1D; d++)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
u0[d][c] = x(d, c, 0, f);
|
||||
u1[d][c] = x(d, c, 1, f);
|
||||
}
|
||||
}
|
||||
real_t Bu0[max_Q1D][VDIM];
|
||||
real_t Bu1[max_Q1D][VDIM];
|
||||
for (int q = 0; q < Q1D; ++q)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Bu0[q][c] = 0.0;
|
||||
Bu1[q][c] = 0.0;
|
||||
}
|
||||
for (int d = 0; d < D1D; ++d)
|
||||
{
|
||||
const real_t b = B(q, d);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Bu0[q][c] += b * u0[d][c];
|
||||
Bu1[q][c] += b * u1[d][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t DBu[max_Q1D][VDIM];
|
||||
for (int q = 0; q < Q1D; ++q)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
DBu[q][c] = op(q, 0, 0, f) * Bu0[q][c] + op(q, 1, 0, f) * Bu1[q][c];
|
||||
}
|
||||
}
|
||||
real_t BDBu[max_D1D][VDIM];
|
||||
for (int d = 0; d < D1D; ++d)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BDBu[d][c] = 0.0;
|
||||
}
|
||||
for (int q = 0; q < Q1D; ++q)
|
||||
{
|
||||
const real_t b = Bt(d, q);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BDBu[d][c] += b * DBu[q][c];
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
y(d, c, 0, f) += BDBu[d][c];
|
||||
y(d, c, 1, f) += -BDBu[d][c];
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// PA DGTrace Apply 3D kernel for Gauss-Lobatto/Bernstein
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PADGTraceApply3D(const int NF, const Array<real_t> &b,
|
||||
const Array<real_t> &bt, const Vector &op_,
|
||||
const Vector &x_, Vector &y_, const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(bt.Read(), D1D, Q1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, 2, 2, NF);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, 2, NF);
|
||||
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE(int f)
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
// the following variables are evaluated at compile time
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
real_t u0[max_D1D][max_D1D][VDIM];
|
||||
real_t u1[max_D1D][max_D1D][VDIM];
|
||||
for (int d1 = 0; d1 < D1D; d1++)
|
||||
{
|
||||
for (int d2 = 0; d2 < D1D; d2++)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
u0[d1][d2][c] = x(d1, d2, c, 0, f);
|
||||
u1[d1][d2][c] = x(d1, d2, c, 1, f);
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t Bu0[max_Q1D][max_D1D][VDIM];
|
||||
real_t Bu1[max_Q1D][max_D1D][VDIM];
|
||||
for (int q = 0; q < Q1D; ++q)
|
||||
{
|
||||
for (int d2 = 0; d2 < D1D; d2++)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Bu0[q][d2][c] = 0.0;
|
||||
Bu1[q][d2][c] = 0.0;
|
||||
}
|
||||
for (int d1 = 0; d1 < D1D; ++d1)
|
||||
{
|
||||
const real_t b = B(q, d1);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Bu0[q][d2][c] += b * u0[d1][d2][c];
|
||||
Bu1[q][d2][c] += b * u1[d1][d2][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t BBu0[max_Q1D][max_Q1D][VDIM];
|
||||
real_t BBu1[max_Q1D][max_Q1D][VDIM];
|
||||
for (int q1 = 0; q1 < Q1D; ++q1)
|
||||
{
|
||||
for (int q2 = 0; q2 < Q1D; q2++)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BBu0[q1][q2][c] = 0.0;
|
||||
BBu1[q1][q2][c] = 0.0;
|
||||
}
|
||||
for (int d2 = 0; d2 < D1D; ++d2)
|
||||
{
|
||||
const real_t b = B(q2, d2);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BBu0[q1][q2][c] += b * Bu0[q1][d2][c];
|
||||
BBu1[q1][q2][c] += b * Bu1[q1][d2][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t DBBu[max_Q1D][max_Q1D][VDIM];
|
||||
for (int q1 = 0; q1 < Q1D; ++q1)
|
||||
{
|
||||
for (int q2 = 0; q2 < Q1D; q2++)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
DBBu[q1][q2][c] = op(q1, q2, 0, 0, f) * BBu0[q1][q2][c] +
|
||||
op(q1, q2, 1, 0, f) * BBu1[q1][q2][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t BDBBu[max_Q1D][max_D1D][VDIM];
|
||||
for (int q1 = 0; q1 < Q1D; ++q1)
|
||||
{
|
||||
for (int d2 = 0; d2 < D1D; d2++)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BDBBu[q1][d2][c] = 0.0;
|
||||
}
|
||||
for (int q2 = 0; q2 < Q1D; ++q2)
|
||||
{
|
||||
const real_t b = Bt(d2, q2);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BDBBu[q1][d2][c] += b * DBBu[q1][q2][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t BBDBBu[max_D1D][max_D1D][VDIM];
|
||||
for (int d1 = 0; d1 < D1D; ++d1)
|
||||
{
|
||||
for (int d2 = 0; d2 < D1D; d2++)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BBDBBu[d1][d2][c] = 0.0;
|
||||
}
|
||||
for (int q1 = 0; q1 < Q1D; ++q1)
|
||||
{
|
||||
const real_t b = Bt(d1, q1);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BBDBBu[d1][d2][c] += b * BDBBu[q1][d2][c];
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
y(d1, d2, c, 0, f) += BBDBBu[d1][d2][c];
|
||||
y(d1, d2, c, 1, f) += -BBDBBu[d1][d2][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Optimized PA DGTrace Apply 3D kernel for Gauss-Lobatto/Bernstein
|
||||
template <int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
|
||||
static void SmemPADGTraceApply3D(const int NF, const Array<real_t> &b,
|
||||
const Array<real_t> &bt, const Vector &op_,
|
||||
const Vector &x_, Vector &y_,
|
||||
const int d1d = 0, const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(bt.Read(), D1D, Q1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, 2, 2, NF);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, 2, NF);
|
||||
|
||||
mfem::forall_2D_batch(NF, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE(int f)
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
// the following variables are evaluated at compile time
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
MFEM_SHARED real_t u0[NBZ][max_D1D][max_D1D];
|
||||
MFEM_SHARED real_t u1[NBZ][max_D1D][max_D1D];
|
||||
MFEM_FOREACH_THREAD(d1, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d2, y, D1D)
|
||||
{
|
||||
u0[tidz][d1][d2] = x(d1, d2, 0, f);
|
||||
u1[tidz][d1][d2] = x(d1, d2, 1, f);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_SHARED real_t Bu0[NBZ][max_Q1D][max_D1D];
|
||||
MFEM_SHARED real_t Bu1[NBZ][max_Q1D][max_D1D];
|
||||
MFEM_FOREACH_THREAD(q1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d2, y, D1D)
|
||||
{
|
||||
real_t Bu0_ = 0.0;
|
||||
real_t Bu1_ = 0.0;
|
||||
for (int d1 = 0; d1 < D1D; ++d1)
|
||||
{
|
||||
const real_t b = B(q1, d1);
|
||||
Bu0_ += b * u0[tidz][d1][d2];
|
||||
Bu1_ += b * u1[tidz][d1][d2];
|
||||
}
|
||||
Bu0[tidz][q1][d2] = Bu0_;
|
||||
Bu1[tidz][q1][d2] = Bu1_;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_SHARED real_t BBu0[NBZ][max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t BBu1[NBZ][max_Q1D][max_Q1D];
|
||||
MFEM_FOREACH_THREAD(q1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q2, y, Q1D)
|
||||
{
|
||||
real_t BBu0_ = 0.0;
|
||||
real_t BBu1_ = 0.0;
|
||||
for (int d2 = 0; d2 < D1D; ++d2)
|
||||
{
|
||||
const real_t b = B(q2, d2);
|
||||
BBu0_ += b * Bu0[tidz][q1][d2];
|
||||
BBu1_ += b * Bu1[tidz][q1][d2];
|
||||
}
|
||||
BBu0[tidz][q1][q2] = BBu0_;
|
||||
BBu1[tidz][q1][q2] = BBu1_;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_SHARED real_t DBBu[NBZ][max_Q1D][max_Q1D];
|
||||
MFEM_FOREACH_THREAD(q1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q2, y, Q1D)
|
||||
{
|
||||
DBBu[tidz][q1][q2] = op(q1, q2, 0, 0, f) * BBu0[tidz][q1][q2] +
|
||||
op(q1, q2, 1, 0, f) * BBu1[tidz][q1][q2];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_SHARED real_t BDBBu[NBZ][max_Q1D][max_D1D];
|
||||
MFEM_FOREACH_THREAD(q1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d2, y, D1D)
|
||||
{
|
||||
real_t BDBBu_ = 0.0;
|
||||
for (int q2 = 0; q2 < Q1D; ++q2)
|
||||
{
|
||||
const real_t b = Bt(d2, q2);
|
||||
BDBBu_ += b * DBBu[tidz][q1][q2];
|
||||
}
|
||||
BDBBu[tidz][q1][d2] = BDBBu_;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(d1, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d2, y, D1D)
|
||||
{
|
||||
real_t BBDBBu_ = 0.0;
|
||||
for (int q1 = 0; q1 < Q1D; ++q1)
|
||||
{
|
||||
const real_t b = Bt(d1, q1);
|
||||
BBDBBu_ += b * BDBBu[tidz][q1][d2];
|
||||
}
|
||||
y(d1, d2, 0, f) += BBDBBu_;
|
||||
y(d1, d2, 1, f) += -BBDBBu_;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// PA DGTrace Apply 2D kernel for Gauss-Lobatto/Bernstein
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PADGTraceApplyTranspose2D(const int NF, const Array<real_t> &b,
|
||||
const Array<real_t> &bt,
|
||||
const Vector &op_, const Vector &x_,
|
||||
Vector &y_, const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(bt.Read(), D1D, Q1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, 2, 2, NF);
|
||||
auto x = Reshape(x_.Read(), D1D, VDIM, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, VDIM, 2, NF);
|
||||
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE(int f)
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
// the following variables are evaluated at compile time
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
real_t u0[max_D1D][VDIM];
|
||||
real_t u1[max_D1D][VDIM];
|
||||
for (int d = 0; d < D1D; d++)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
u0[d][c] = x(d, c, 0, f);
|
||||
u1[d][c] = x(d, c, 1, f);
|
||||
}
|
||||
}
|
||||
real_t Bu0[max_Q1D][VDIM];
|
||||
real_t Bu1[max_Q1D][VDIM];
|
||||
for (int q = 0; q < Q1D; ++q)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Bu0[q][c] = 0.0;
|
||||
Bu1[q][c] = 0.0;
|
||||
}
|
||||
for (int d = 0; d < D1D; ++d)
|
||||
{
|
||||
const real_t b = B(q, d);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Bu0[q][c] += b * u0[d][c];
|
||||
Bu1[q][c] += b * u1[d][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t DBu0[max_Q1D][VDIM];
|
||||
real_t DBu1[max_Q1D][VDIM];
|
||||
for (int q = 0; q < Q1D; ++q)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
DBu0[q][c] =
|
||||
op(q, 0, 0, f) * Bu0[q][c] + op(q, 0, 1, f) * Bu1[q][c];
|
||||
DBu1[q][c] =
|
||||
op(q, 1, 0, f) * Bu0[q][c] + op(q, 1, 1, f) * Bu1[q][c];
|
||||
}
|
||||
}
|
||||
real_t BDBu0[max_D1D][VDIM];
|
||||
real_t BDBu1[max_D1D][VDIM];
|
||||
for (int d = 0; d < D1D; ++d)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BDBu0[d][c] = 0.0;
|
||||
BDBu1[d][c] = 0.0;
|
||||
}
|
||||
for (int q = 0; q < Q1D; ++q)
|
||||
{
|
||||
const real_t b = Bt(d, q);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BDBu0[d][c] += b * DBu0[q][c];
|
||||
BDBu1[d][c] += b * DBu1[q][c];
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
y(d, c, 0, f) += BDBu0[d][c];
|
||||
y(d, c, 1, f) += BDBu1[d][c];
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// PA DGTrace Apply Transpose 3D kernel for Gauss-Lobatto/Bernstein
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PADGTraceApplyTranspose3D(const int NF, const Array<real_t> &b,
|
||||
const Array<real_t> &bt,
|
||||
const Vector &op_, const Vector &x_,
|
||||
Vector &y_, const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(bt.Read(), D1D, Q1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, 2, 2, NF);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, 2, NF);
|
||||
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE(int f)
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
// the following variables are evaluated at compile time
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
real_t u0[max_D1D][max_D1D][VDIM];
|
||||
real_t u1[max_D1D][max_D1D][VDIM];
|
||||
for (int d1 = 0; d1 < D1D; d1++)
|
||||
{
|
||||
for (int d2 = 0; d2 < D1D; d2++)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
u0[d1][d2][c] = x(d1, d2, c, 0, f);
|
||||
u1[d1][d2][c] = x(d1, d2, c, 1, f);
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t Bu0[max_Q1D][max_D1D][VDIM];
|
||||
real_t Bu1[max_Q1D][max_D1D][VDIM];
|
||||
for (int q1 = 0; q1 < Q1D; ++q1)
|
||||
{
|
||||
for (int d2 = 0; d2 < D1D; ++d2)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Bu0[q1][d2][c] = 0.0;
|
||||
Bu1[q1][d2][c] = 0.0;
|
||||
}
|
||||
for (int d1 = 0; d1 < D1D; ++d1)
|
||||
{
|
||||
const real_t b = B(q1, d1);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Bu0[q1][d2][c] += b * u0[d1][d2][c];
|
||||
Bu1[q1][d2][c] += b * u1[d1][d2][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t BBu0[max_Q1D][max_Q1D][VDIM];
|
||||
real_t BBu1[max_Q1D][max_Q1D][VDIM];
|
||||
for (int q1 = 0; q1 < Q1D; ++q1)
|
||||
{
|
||||
for (int q2 = 0; q2 < Q1D; ++q2)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BBu0[q1][q2][c] = 0.0;
|
||||
BBu1[q1][q2][c] = 0.0;
|
||||
}
|
||||
for (int d2 = 0; d2 < D1D; ++d2)
|
||||
{
|
||||
const real_t b = B(q2, d2);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BBu0[q1][q2][c] += b * Bu0[q1][d2][c];
|
||||
BBu1[q1][q2][c] += b * Bu1[q1][d2][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t DBu0[max_Q1D][max_Q1D][VDIM];
|
||||
real_t DBu1[max_Q1D][max_Q1D][VDIM];
|
||||
for (int q1 = 0; q1 < Q1D; ++q1)
|
||||
{
|
||||
for (int q2 = 0; q2 < Q1D; ++q2)
|
||||
{
|
||||
const real_t D00 = op(q1, q2, 0, 0, f);
|
||||
const real_t D01 = op(q1, q2, 0, 1, f);
|
||||
const real_t D10 = op(q1, q2, 1, 0, f);
|
||||
const real_t D11 = op(q1, q2, 1, 1, f);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
DBu0[q1][q2][c] = D00 * BBu0[q1][q2][c] + D01 * BBu1[q1][q2][c];
|
||||
DBu1[q1][q2][c] = D10 * BBu0[q1][q2][c] + D11 * BBu1[q1][q2][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t BDBu0[max_D1D][max_Q1D][VDIM];
|
||||
real_t BDBu1[max_D1D][max_Q1D][VDIM];
|
||||
for (int d1 = 0; d1 < D1D; ++d1)
|
||||
{
|
||||
for (int q2 = 0; q2 < Q1D; ++q2)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BDBu0[d1][q2][c] = 0.0;
|
||||
BDBu1[d1][q2][c] = 0.0;
|
||||
}
|
||||
for (int q1 = 0; q1 < Q1D; ++q1)
|
||||
{
|
||||
const real_t b = Bt(d1, q1);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BDBu0[d1][q2][c] += b * DBu0[q1][q2][c];
|
||||
BDBu1[d1][q2][c] += b * DBu1[q1][q2][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
real_t BBDBu0[max_D1D][max_D1D][VDIM];
|
||||
real_t BBDBu1[max_D1D][max_D1D][VDIM];
|
||||
for (int d1 = 0; d1 < D1D; ++d1)
|
||||
{
|
||||
for (int d2 = 0; d2 < D1D; ++d2)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BBDBu0[d1][d2][c] = 0.0;
|
||||
BBDBu1[d1][d2][c] = 0.0;
|
||||
}
|
||||
for (int q2 = 0; q2 < Q1D; ++q2)
|
||||
{
|
||||
const real_t b = Bt(d2, q2);
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
BBDBu0[d1][d2][c] += b * BDBu0[d1][q2][c];
|
||||
BBDBu1[d1][d2][c] += b * BDBu1[d1][q2][c];
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
y(d1, d2, c, 0, f) += BBDBu0[d1][d2][c];
|
||||
y(d1, d2, c, 1, f) += BBDBu1[d1][d2][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Optimized PA DGTrace Apply Transpose 3D kernel for Gauss-Lobatto/Bernstein
|
||||
template <int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
|
||||
static void SmemPADGTraceApplyTranspose3D(const int NF, const Array<real_t> &b,
|
||||
const Array<real_t> &bt,
|
||||
const Vector &op_, const Vector &x_,
|
||||
Vector &y_, const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(bt.Read(), D1D, Q1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, 2, 2, NF);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, 2, NF);
|
||||
|
||||
mfem::forall_2D_batch(NF, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE(int f)
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
// the following variables are evaluated at compile time
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
MFEM_SHARED real_t u0[NBZ][max_D1D][max_D1D];
|
||||
MFEM_SHARED real_t u1[NBZ][max_D1D][max_D1D];
|
||||
MFEM_FOREACH_THREAD(d1, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d2, y, D1D)
|
||||
{
|
||||
u0[tidz][d1][d2] = x(d1, d2, 0, f);
|
||||
u1[tidz][d1][d2] = x(d1, d2, 1, f);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_SHARED real_t Bu0[NBZ][max_Q1D][max_D1D];
|
||||
MFEM_SHARED real_t Bu1[NBZ][max_Q1D][max_D1D];
|
||||
MFEM_FOREACH_THREAD(q1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d2, y, D1D)
|
||||
{
|
||||
real_t Bu0_ = 0.0;
|
||||
real_t Bu1_ = 0.0;
|
||||
for (int d1 = 0; d1 < D1D; ++d1)
|
||||
{
|
||||
const real_t b = B(q1, d1);
|
||||
Bu0_ += b * u0[tidz][d1][d2];
|
||||
Bu1_ += b * u1[tidz][d1][d2];
|
||||
}
|
||||
Bu0[tidz][q1][d2] = Bu0_;
|
||||
Bu1[tidz][q1][d2] = Bu1_;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_SHARED real_t BBu0[NBZ][max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t BBu1[NBZ][max_Q1D][max_Q1D];
|
||||
MFEM_FOREACH_THREAD(q1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q2, y, Q1D)
|
||||
{
|
||||
real_t BBu0_ = 0.0;
|
||||
real_t BBu1_ = 0.0;
|
||||
for (int d2 = 0; d2 < D1D; ++d2)
|
||||
{
|
||||
const real_t b = B(q2, d2);
|
||||
BBu0_ += b * Bu0[tidz][q1][d2];
|
||||
BBu1_ += b * Bu1[tidz][q1][d2];
|
||||
}
|
||||
BBu0[tidz][q1][q2] = BBu0_;
|
||||
BBu1[tidz][q1][q2] = BBu1_;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_SHARED real_t DBBu0[NBZ][max_Q1D][max_Q1D];
|
||||
MFEM_SHARED real_t DBBu1[NBZ][max_Q1D][max_Q1D];
|
||||
MFEM_FOREACH_THREAD(q1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q2, y, Q1D)
|
||||
{
|
||||
const real_t D00 = op(q1, q2, 0, 0, f);
|
||||
const real_t D01 = op(q1, q2, 0, 1, f);
|
||||
const real_t D10 = op(q1, q2, 1, 0, f);
|
||||
const real_t D11 = op(q1, q2, 1, 1, f);
|
||||
const real_t u0q = BBu0[tidz][q1][q2];
|
||||
const real_t u1q = BBu1[tidz][q1][q2];
|
||||
DBBu0[tidz][q1][q2] = D00 * u0q + D01 * u1q;
|
||||
DBBu1[tidz][q1][q2] = D10 * u0q + D11 * u1q;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_SHARED real_t BDBBu0[NBZ][max_Q1D][max_D1D];
|
||||
MFEM_SHARED real_t BDBBu1[NBZ][max_Q1D][max_D1D];
|
||||
MFEM_FOREACH_THREAD(q1, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d2, y, D1D)
|
||||
{
|
||||
real_t BDBBu0_ = 0.0;
|
||||
real_t BDBBu1_ = 0.0;
|
||||
for (int q2 = 0; q2 < Q1D; ++q2)
|
||||
{
|
||||
const real_t b = Bt(d2, q2);
|
||||
BDBBu0_ += b * DBBu0[tidz][q1][q2];
|
||||
BDBBu1_ += b * DBBu1[tidz][q1][q2];
|
||||
}
|
||||
BDBBu0[tidz][q1][d2] = BDBBu0_;
|
||||
BDBBu1[tidz][q1][d2] = BDBBu1_;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(d1, x, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d2, y, D1D)
|
||||
{
|
||||
real_t BBDBBu0_ = 0.0;
|
||||
real_t BBDBBu1_ = 0.0;
|
||||
for (int q1 = 0; q1 < Q1D; ++q1)
|
||||
{
|
||||
const real_t b = Bt(d1, q1);
|
||||
BBDBBu0_ += b * BDBBu0[tidz][q1][d2];
|
||||
BBDBBu1_ += b * BDBBu1[tidz][q1][d2];
|
||||
}
|
||||
y(d1, d2, 0, f) += BBDBBu0_;
|
||||
y(d1, d2, 1, f) += BBDBBu1_;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace internal
|
||||
|
||||
template <int DIM, int D1D, int Q1D>
|
||||
DGTraceIntegrator::ApplyKernelType DGTraceIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2)
|
||||
{
|
||||
return internal::PADGTraceApply2D<D1D, Q1D>;
|
||||
}
|
||||
else if constexpr (DIM == 3)
|
||||
{
|
||||
if constexpr (D1D == 3 || D1D == 4)
|
||||
{
|
||||
return internal::SmemPADGTraceApply3D<D1D, Q1D, 2>;
|
||||
}
|
||||
else
|
||||
{
|
||||
return internal::SmemPADGTraceApply3D<D1D, Q1D>;
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
template <int DIM, int D1D, int Q1D>
|
||||
DGTraceIntegrator::ApplyKernelType DGTraceIntegrator::ApplyPATKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2)
|
||||
{
|
||||
return internal::PADGTraceApplyTranspose2D<D1D, Q1D>;
|
||||
}
|
||||
else if constexpr (DIM == 3)
|
||||
{
|
||||
return internal::SmemPADGTraceApplyTranspose3D<D1D, Q1D>;
|
||||
}
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
} // namespace mfem
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
#endif
|
||||
+163
-922
File diff suppressed because it is too large
Load Diff
@@ -19,6 +19,8 @@ namespace mfem
|
||||
DiffusionIntegrator::Kernels::Kernels()
|
||||
{
|
||||
// 2D
|
||||
// Q = P+1
|
||||
DiffusionIntegrator::AddSpecialization<2,1,1>();
|
||||
DiffusionIntegrator::AddSpecialization<2,2,2>();
|
||||
DiffusionIntegrator::AddSpecialization<2,3,3>();
|
||||
DiffusionIntegrator::AddSpecialization<2,4,4>();
|
||||
@@ -27,17 +29,39 @@ DiffusionIntegrator::Kernels::Kernels()
|
||||
DiffusionIntegrator::AddSpecialization<2,7,7>();
|
||||
DiffusionIntegrator::AddSpecialization<2,8,8>();
|
||||
DiffusionIntegrator::AddSpecialization<2,9,9>();
|
||||
// Q = P+2
|
||||
DiffusionIntegrator::AddSpecialization<2,1,2>();
|
||||
DiffusionIntegrator::AddSpecialization<2,2,3>();
|
||||
DiffusionIntegrator::AddSpecialization<2,3,4>();
|
||||
DiffusionIntegrator::AddSpecialization<2,4,5>();
|
||||
DiffusionIntegrator::AddSpecialization<2,5,6>();
|
||||
DiffusionIntegrator::AddSpecialization<2,6,7>();
|
||||
DiffusionIntegrator::AddSpecialization<2,7,8>();
|
||||
DiffusionIntegrator::AddSpecialization<2,8,9>();
|
||||
DiffusionIntegrator::AddSpecialization<2,9,10>();
|
||||
// others
|
||||
// 3D
|
||||
// Q = P+1
|
||||
DiffusionIntegrator::AddSpecialization<3,1,1>();
|
||||
DiffusionIntegrator::AddSpecialization<3,2,2>();
|
||||
DiffusionIntegrator::AddSpecialization<3,3,3>();
|
||||
DiffusionIntegrator::AddSpecialization<3,4,4>();
|
||||
DiffusionIntegrator::AddSpecialization<3,5,5>();
|
||||
DiffusionIntegrator::AddSpecialization<3,6,6>();
|
||||
DiffusionIntegrator::AddSpecialization<3,7,7>();
|
||||
DiffusionIntegrator::AddSpecialization<3,8,8>();
|
||||
// Q = P+2
|
||||
DiffusionIntegrator::AddSpecialization<3,1,2>();
|
||||
DiffusionIntegrator::AddSpecialization<3,2,3>();
|
||||
DiffusionIntegrator::AddSpecialization<3,3,4>();
|
||||
DiffusionIntegrator::AddSpecialization<3,4,5>();
|
||||
DiffusionIntegrator::AddSpecialization<3,4,6>();
|
||||
DiffusionIntegrator::AddSpecialization<3,5,6>();
|
||||
DiffusionIntegrator::AddSpecialization<3,5,8>();
|
||||
DiffusionIntegrator::AddSpecialization<3,6,7>();
|
||||
DiffusionIntegrator::AddSpecialization<3,7,8>();
|
||||
DiffusionIntegrator::AddSpecialization<3,8,9>();
|
||||
// others
|
||||
DiffusionIntegrator::AddSpecialization<3,4,6>();
|
||||
DiffusionIntegrator::AddSpecialization<3,5,8>();
|
||||
}
|
||||
|
||||
namespace internal
|
||||
|
||||
@@ -672,12 +672,12 @@ inline void SmemPADiffusionApply2D(const int NE,
|
||||
real_t (*Gt)[MQ1] = (real_t (*)[MQ1]) (sBG+1);
|
||||
MFEM_SHARED real_t Xz[NBZ][MD1][MD1];
|
||||
MFEM_SHARED real_t GD[2][NBZ][MD1][MQ1];
|
||||
MFEM_SHARED real_t GQ[2][NBZ][MD1][MQ1];
|
||||
MFEM_SHARED real_t GQ[2][NBZ][MQ1][MQ1];
|
||||
real_t (*X)[MD1] = (real_t (*)[MD1])(Xz + tidz);
|
||||
real_t (*DQ0)[MD1] = (real_t (*)[MD1])(GD[0] + tidz);
|
||||
real_t (*DQ1)[MD1] = (real_t (*)[MD1])(GD[1] + tidz);
|
||||
real_t (*QQ0)[MD1] = (real_t (*)[MD1])(GQ[0] + tidz);
|
||||
real_t (*QQ1)[MD1] = (real_t (*)[MD1])(GQ[1] + tidz);
|
||||
real_t (*DQ0)[MQ1] = (real_t (*)[MQ1])(GD[0] + tidz);
|
||||
real_t (*DQ1)[MQ1] = (real_t (*)[MQ1])(GD[1] + tidz);
|
||||
real_t (*QQ0)[MQ1] = (real_t (*)[MQ1])(GQ[0] + tidz);
|
||||
real_t (*QQ1)[MQ1] = (real_t (*)[MQ1])(GQ[1] + tidz);
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
@@ -1221,9 +1221,9 @@ using DiagonalKernelType = DiffusionIntegrator::DiagonalKernelType;
|
||||
template<int DIM, int T_D1D, int T_Q1D>
|
||||
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if (DIM == 2) { return internal::SmemPADiffusionApply2D<T_D1D,T_Q1D>; }
|
||||
else if (DIM == 3) { return internal::SmemPADiffusionApply3D<T_D1D, T_Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<T_D1D,T_Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<T_D1D, T_Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
inline
|
||||
@@ -1237,9 +1237,9 @@ ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int DIM, int, int)
|
||||
template<int DIM, int D1D, int Q1D>
|
||||
DiagonalKernelType DiffusionIntegrator::DiagonalPAKernels::Kernel()
|
||||
{
|
||||
if (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D,Q1D>; }
|
||||
else if (DIM == 3) { return internal::SmemPADiffusionDiagonal3D<D1D, Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D,Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPADiffusionDiagonal3D<D1D, Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
inline DiagonalKernelType
|
||||
|
||||
@@ -599,13 +599,11 @@ void PACurlCurlSetup3D(const int Q1D,
|
||||
});
|
||||
}
|
||||
|
||||
void PACurlCurlAssembleDiagonal2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<real_t> &bo,
|
||||
void PACurlCurlAssembleDiagonal2D(const int D1D, const int Q1D, const bool,
|
||||
const int NE, const Array<real_t> &bo,
|
||||
const Array<real_t> &, const Array<real_t> &,
|
||||
const Array<real_t> &gc,
|
||||
const Vector &pa_data,
|
||||
Vector &diag)
|
||||
const Vector &pa_data, Vector &diag)
|
||||
{
|
||||
auto Bo = Reshape(bo.Read(), Q1D, D1D-1);
|
||||
auto Gc = Reshape(gc.Read(), Q1D, D1D);
|
||||
@@ -653,16 +651,11 @@ void PACurlCurlAssembleDiagonal2D(const int D1D,
|
||||
}); // end of element loop
|
||||
}
|
||||
|
||||
void PACurlCurlApply2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<real_t> &bo,
|
||||
const Array<real_t> &bot,
|
||||
const Array<real_t> &gc,
|
||||
const Array<real_t> &gct,
|
||||
const Vector &pa_data,
|
||||
const Vector &x,
|
||||
Vector &y,
|
||||
void PACurlCurlApply2D(const int D1D, const int Q1D, const bool, const int NE,
|
||||
const Array<real_t> &bo, const Array<real_t> &,
|
||||
const Array<real_t> &bot, const Array<real_t> &,
|
||||
const Array<real_t> &gc, const Array<real_t> &gct,
|
||||
const Vector &pa_data, const Vector &x, Vector &y,
|
||||
const bool useAbs)
|
||||
{
|
||||
|
||||
|
||||
@@ -24,7 +24,7 @@
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
namespace internal
|
||||
{
|
||||
|
||||
@@ -426,8 +426,11 @@ void PACurlCurlSetup3D(const int Q1D,
|
||||
// PA H(curl) curl-curl Diagonal 2D kernel
|
||||
void PACurlCurlAssembleDiagonal2D(const int D1D,
|
||||
const int Q1D,
|
||||
const bool symmetric, // unused
|
||||
const int NE,
|
||||
const Array<real_t> &bo,
|
||||
const Array<real_t> &bc, // unused
|
||||
const Array<real_t> &go, // unused
|
||||
const Array<real_t> &gc,
|
||||
const Vector &pa_data,
|
||||
Vector &diag);
|
||||
@@ -831,9 +834,12 @@ inline void SmemPACurlCurlAssembleDiagonal3D(const int d1d,
|
||||
// PA H(curl) curl-curl Apply/AbsApply 2D kernel
|
||||
void PACurlCurlApply2D(const int D1D,
|
||||
const int Q1D,
|
||||
const bool symmetric, // unused
|
||||
const int NE,
|
||||
const Array<real_t> &bo,
|
||||
const Array<real_t> &bc, // unused
|
||||
const Array<real_t> &bot,
|
||||
const Array<real_t> &bct, // unused
|
||||
const Array<real_t> &gc,
|
||||
const Array<real_t> &gct,
|
||||
const Vector &pa_data,
|
||||
@@ -3158,6 +3164,49 @@ inline void SmemPAHcurlL2ApplyTranspose3D(const int d1d,
|
||||
|
||||
} // namespace internal
|
||||
|
||||
template<int DIM, int T_D1D, int T_Q1D>
|
||||
CurlCurlIntegrator::ApplyKernelType CurlCurlIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2)
|
||||
{
|
||||
return internal::PACurlCurlApply2D;
|
||||
}
|
||||
else if constexpr (DIM == 3)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
return internal::SmemPACurlCurlApply3D<T_D1D, T_Q1D>;
|
||||
}
|
||||
else
|
||||
{
|
||||
return internal::PACurlCurlApply3D;
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
template <int DIM, int T_D1D, int T_Q1D>
|
||||
CurlCurlIntegrator::DiagonalKernelType
|
||||
CurlCurlIntegrator::DiagonalPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2)
|
||||
{
|
||||
return internal::PACurlCurlAssembleDiagonal2D;
|
||||
}
|
||||
else if constexpr (DIM == 3)
|
||||
{
|
||||
if (Device::Allows(Backend::DEVICE_MASK))
|
||||
{
|
||||
return internal::SmemPACurlCurlAssembleDiagonal3D<T_D1D, T_Q1D>;
|
||||
}
|
||||
else
|
||||
{
|
||||
return internal::PACurlCurlAssembleDiagonal3D;
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
|
||||
@@ -19,6 +19,7 @@
|
||||
#include "../../linalg/vector.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
@@ -819,4 +820,6 @@ inline void PAHcurlHdivApplyTranspose3D(const int d1d,
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
#endif
|
||||
|
||||
@@ -17,6 +17,8 @@ namespace mfem
|
||||
MassIntegrator::Kernels::Kernels()
|
||||
{
|
||||
// 2D
|
||||
// Q=P+1
|
||||
MassIntegrator::AddSpecialization<2,1,1>();
|
||||
MassIntegrator::AddSpecialization<2,2,2>();
|
||||
MassIntegrator::AddSpecialization<2,3,3>();
|
||||
MassIntegrator::AddSpecialization<2,4,4>();
|
||||
@@ -25,17 +27,45 @@ MassIntegrator::Kernels::Kernels()
|
||||
MassIntegrator::AddSpecialization<2,7,7>();
|
||||
MassIntegrator::AddSpecialization<2,8,8>();
|
||||
MassIntegrator::AddSpecialization<2,9,9>();
|
||||
// Q=P+2
|
||||
MassIntegrator::AddSpecialization<2,1,2>();
|
||||
MassIntegrator::AddSpecialization<2,2,3>();
|
||||
MassIntegrator::AddSpecialization<2,3,4>();
|
||||
MassIntegrator::AddSpecialization<2,4,5>();
|
||||
MassIntegrator::AddSpecialization<2,5,6>();
|
||||
MassIntegrator::AddSpecialization<2,6,7>();
|
||||
MassIntegrator::AddSpecialization<2,7,8>();
|
||||
MassIntegrator::AddSpecialization<2,8,9>();
|
||||
MassIntegrator::AddSpecialization<2,9,10>();
|
||||
// others
|
||||
MassIntegrator::AddSpecialization<2,2,4>();
|
||||
MassIntegrator::AddSpecialization<2,3,6>();
|
||||
MassIntegrator::AddSpecialization<2,4,6>();
|
||||
// 3D
|
||||
// Q=P+1
|
||||
MassIntegrator::AddSpecialization<3,1,1>();
|
||||
MassIntegrator::AddSpecialization<3,2,2>();
|
||||
MassIntegrator::AddSpecialization<3,3,3>();
|
||||
MassIntegrator::AddSpecialization<3,4,4>();
|
||||
MassIntegrator::AddSpecialization<3,5,5>();
|
||||
MassIntegrator::AddSpecialization<3,6,6>();
|
||||
MassIntegrator::AddSpecialization<3,7,7>();
|
||||
MassIntegrator::AddSpecialization<3,8,8>();
|
||||
MassIntegrator::AddSpecialization<3,9,9>();
|
||||
// Q=P+2
|
||||
MassIntegrator::AddSpecialization<3,1,2>();
|
||||
MassIntegrator::AddSpecialization<3,2,3>();
|
||||
MassIntegrator::AddSpecialization<3,3,4>();
|
||||
MassIntegrator::AddSpecialization<3,3,6>();
|
||||
MassIntegrator::AddSpecialization<3,4,5>();
|
||||
MassIntegrator::AddSpecialization<3,4,6>();
|
||||
MassIntegrator::AddSpecialization<3,5,6>();
|
||||
MassIntegrator::AddSpecialization<3,5,8>();
|
||||
MassIntegrator::AddSpecialization<3,6,7>();
|
||||
MassIntegrator::AddSpecialization<3,7,8>();
|
||||
MassIntegrator::AddSpecialization<3,8,9>();
|
||||
// others
|
||||
MassIntegrator::AddSpecialization<3,2,4>();
|
||||
MassIntegrator::AddSpecialization<3,4,6>();
|
||||
MassIntegrator::AddSpecialization<3,5,8>();
|
||||
}
|
||||
|
||||
namespace internal
|
||||
|
||||
@@ -1392,10 +1392,10 @@ using DiagonalKernelType = MassIntegrator::DiagonalKernelType;
|
||||
template<int DIM, int T_D1D, int T_Q1D>
|
||||
ApplyKernelType MassIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if (DIM == 1) { return internal::PAMassApply1D; }
|
||||
else if (DIM == 2) { return internal::SmemPAMassApply2D<T_D1D,T_Q1D>; }
|
||||
else if (DIM == 3) { return internal::SmemPAMassApply3D<T_D1D, T_Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
if constexpr (DIM == 1) { return internal::PAMassApply1D; }
|
||||
else if constexpr (DIM == 2) { return internal::SmemPAMassApply2D<T_D1D,T_Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPAMassApply3D<T_D1D, T_Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
inline ApplyKernelType MassIntegrator::ApplyPAKernels::Fallback(
|
||||
@@ -1410,10 +1410,10 @@ inline ApplyKernelType MassIntegrator::ApplyPAKernels::Fallback(
|
||||
template<int DIM, int T_D1D, int T_Q1D>
|
||||
DiagonalKernelType MassIntegrator::DiagonalPAKernels::Kernel()
|
||||
{
|
||||
if (DIM == 1) { return internal::PAMassAssembleDiagonal1D; }
|
||||
else if (DIM == 2) { return internal::SmemPAMassAssembleDiagonal2D<T_D1D,T_Q1D>; }
|
||||
else if (DIM == 3) { return internal::SmemPAMassAssembleDiagonal3D<T_D1D, T_Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
if constexpr (DIM == 1) { return internal::PAMassAssembleDiagonal1D; }
|
||||
else if constexpr (DIM == 2) { return internal::SmemPAMassAssembleDiagonal2D<T_D1D,T_Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPAMassAssembleDiagonal3D<T_D1D, T_Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
inline DiagonalKernelType MassIntegrator::DiagonalPAKernels::Fallback(
|
||||
|
||||
@@ -0,0 +1,355 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_BILININTEG_VECDIFFUSION_KERNELS_HPP
|
||||
#define MFEM_BILININTEG_VECDIFFUSION_KERNELS_HPP
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../ceed/integrators/diffusion/diffusion.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace internal
|
||||
{
|
||||
|
||||
// PA Diffusion Apply 2D kernel
|
||||
template <int T_D1D = 0, int T_Q1D = 0, int T_VDIM = 0>
|
||||
static void
|
||||
PAVectorDiffusionApply2D(const int NE, const Array<real_t> &b,
|
||||
const Array<real_t> &g, const Array<real_t> &bt,
|
||||
const Array<real_t> >, const Vector &d_,
|
||||
const Vector &x_, Vector &y_, const int d1d = 0,
|
||||
const int q1d = 0, const int vdim = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(bt.Read(), D1D, Q1D);
|
||||
auto Gt = Reshape(gt.Read(), D1D, Q1D);
|
||||
auto D = Reshape(d_.Read(), Q1D * Q1D, 3, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
|
||||
real_t grad[max_Q1D][max_Q1D][2];
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
grad[qy][qx][0] = 0.0;
|
||||
grad[qy][qx][1] = 0.0;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
real_t gradX[max_Q1D][2];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
gradX[qx][0] = 0.0;
|
||||
gradX[qx][1] = 0.0;
|
||||
}
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const real_t s = x(dx, dy, c, e);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
gradX[qx][0] += s * B(qx, dx);
|
||||
gradX[qx][1] += s * G(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t wy = B(qy, dy);
|
||||
const real_t wDy = G(qy, dy);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
grad[qy][qx][0] += gradX[qx][1] * wy;
|
||||
grad[qy][qx][1] += gradX[qx][0] * wDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxy, xDy in plane
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const int q = qx + qy * Q1D;
|
||||
const real_t O11 = D(q, 0, e);
|
||||
const real_t O12 = D(q, 1, e);
|
||||
const real_t O22 = D(q, 2, e);
|
||||
const real_t gradX = grad[qy][qx][0];
|
||||
const real_t gradY = grad[qy][qx][1];
|
||||
grad[qy][qx][0] = (O11 * gradX) + (O12 * gradY);
|
||||
grad[qy][qx][1] = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
real_t gradX[max_D1D][2];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
gradX[dx][0] = 0.0;
|
||||
gradX[dx][1] = 0.0;
|
||||
}
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t gX = grad[qy][qx][0];
|
||||
const real_t gY = grad[qy][qx][1];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const real_t wx = Bt(dx, qx);
|
||||
const real_t wDx = Gt(dx, qx);
|
||||
gradX[dx][0] += gX * wDx;
|
||||
gradX[dx][1] += gY * wx;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const real_t wy = Bt(dy, qy);
|
||||
const real_t wDy = Gt(dy, qy);
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
y(dx, dy, c, e) +=
|
||||
((gradX[dx][0] * wy) + (gradX[dx][1] * wDy));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// PA Diffusion Apply 3D kernel
|
||||
template <const int T_D1D = 0, const int T_Q1D = 0>
|
||||
static void
|
||||
PAVectorDiffusionApply3D(const int NE, const Array<real_t> &b,
|
||||
const Array<real_t> &g, const Array<real_t> &bt,
|
||||
const Array<real_t> >, const Vector &op_,
|
||||
const Vector &x_, Vector &y_, const int d1d = 0,
|
||||
const int q1d = 0, const int sdim = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int VDIM = 3;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(bt.Read(), D1D, Q1D);
|
||||
auto Gt = Reshape(gt.Read(), D1D, Q1D);
|
||||
auto op = Reshape(op_.Read(), Q1D * Q1D * Q1D, 6, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
real_t grad[max_Q1D][max_Q1D][max_Q1D][3];
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
grad[qz][qy][qx][0] = 0.0;
|
||||
grad[qz][qy][qx][1] = 0.0;
|
||||
grad[qz][qy][qx][2] = 0.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
real_t gradXY[max_Q1D][max_Q1D][3];
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
gradXY[qy][qx][0] = 0.0;
|
||||
gradXY[qy][qx][1] = 0.0;
|
||||
gradXY[qy][qx][2] = 0.0;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
real_t gradX[max_Q1D][2];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
gradX[qx][0] = 0.0;
|
||||
gradX[qx][1] = 0.0;
|
||||
}
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const real_t s = x(dx, dy, dz, c, e);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
gradX[qx][0] += s * B(qx, dx);
|
||||
gradX[qx][1] += s * G(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t wy = B(qy, dy);
|
||||
const real_t wDy = G(qy, dy);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t wx = gradX[qx][0];
|
||||
const real_t wDx = gradX[qx][1];
|
||||
gradXY[qy][qx][0] += wDx * wy;
|
||||
gradXY[qy][qx][1] += wx * wDy;
|
||||
gradXY[qy][qx][2] += wx * wy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const real_t wz = B(qz, dz);
|
||||
const real_t wDz = G(qz, dz);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
grad[qz][qy][qx][0] += gradXY[qy][qx][0] * wz;
|
||||
grad[qz][qy][qx][1] += gradXY[qy][qx][1] * wz;
|
||||
grad[qz][qy][qx][2] += gradXY[qy][qx][2] * wDz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const int q = qx + (qy + qz * Q1D) * Q1D;
|
||||
const real_t O11 = op(q, 0, e);
|
||||
const real_t O12 = op(q, 1, e);
|
||||
const real_t O13 = op(q, 2, e);
|
||||
const real_t O22 = op(q, 3, e);
|
||||
const real_t O23 = op(q, 4, e);
|
||||
const real_t O33 = op(q, 5, e);
|
||||
const real_t gradX = grad[qz][qy][qx][0];
|
||||
const real_t gradY = grad[qz][qy][qx][1];
|
||||
const real_t gradZ = grad[qz][qy][qx][2];
|
||||
grad[qz][qy][qx][0] =
|
||||
(O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
|
||||
grad[qz][qy][qx][1] =
|
||||
(O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
|
||||
grad[qz][qy][qx][2] =
|
||||
(O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
real_t gradXY[max_D1D][max_D1D][3];
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
gradXY[dy][dx][0] = 0;
|
||||
gradXY[dy][dx][1] = 0;
|
||||
gradXY[dy][dx][2] = 0;
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
real_t gradX[max_D1D][3];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
gradX[dx][0] = 0;
|
||||
gradX[dx][1] = 0;
|
||||
gradX[dx][2] = 0;
|
||||
}
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t gX = grad[qz][qy][qx][0];
|
||||
const real_t gY = grad[qz][qy][qx][1];
|
||||
const real_t gZ = grad[qz][qy][qx][2];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const real_t wx = Bt(dx, qx);
|
||||
const real_t wDx = Gt(dx, qx);
|
||||
gradX[dx][0] += gX * wDx;
|
||||
gradX[dx][1] += gY * wx;
|
||||
gradX[dx][2] += gZ * wx;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const real_t wy = Bt(dy, qy);
|
||||
const real_t wDy = Gt(dy, qy);
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
gradXY[dy][dx][0] += gradX[dx][0] * wy;
|
||||
gradXY[dy][dx][1] += gradX[dx][1] * wDy;
|
||||
gradXY[dy][dx][2] += gradX[dx][2] * wy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
const real_t wz = Bt(dz, qz);
|
||||
const real_t wDz = Gt(dz, qz);
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
y(dx, dy, dz, c, e) +=
|
||||
((gradXY[dy][dx][0] * wz) + (gradXY[dy][dx][1] * wz) +
|
||||
(gradXY[dy][dx][2] * wDz));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
} // namespace internal
|
||||
|
||||
template <int DIM, int VDIM, int T_D1D, int T_Q1D>
|
||||
VectorDiffusionIntegrator::ApplyKernelType
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2)
|
||||
{
|
||||
return internal::PAVectorDiffusionApply2D<T_D1D, T_Q1D, VDIM>;
|
||||
}
|
||||
else if constexpr (DIM == 3)
|
||||
{
|
||||
return internal::PAVectorDiffusionApply3D;
|
||||
}
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
#endif
|
||||
@@ -15,9 +15,58 @@
|
||||
#include "../qfunction.hpp"
|
||||
#include "../ceed/integrators/diffusion/diffusion.hpp"
|
||||
|
||||
#include "bilininteg_vecdiffusion_kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
VectorDiffusionIntegrator::VectorDiffusionIntegrator(const IntegrationRule *ir)
|
||||
: BilinearFormIntegrator(ir)
|
||||
{
|
||||
static Kernels kernels;
|
||||
}
|
||||
|
||||
VectorDiffusionIntegrator::VectorDiffusionIntegrator(Coefficient &q)
|
||||
: VectorDiffusionIntegrator()
|
||||
{
|
||||
Q = &q;
|
||||
}
|
||||
|
||||
VectorDiffusionIntegrator::VectorDiffusionIntegrator(int vector_dimension)
|
||||
: VectorDiffusionIntegrator()
|
||||
{
|
||||
vdim = vector_dimension;
|
||||
}
|
||||
|
||||
VectorDiffusionIntegrator::VectorDiffusionIntegrator(Coefficient &q,
|
||||
const IntegrationRule *ir)
|
||||
: VectorDiffusionIntegrator(ir)
|
||||
{
|
||||
Q = &q;
|
||||
}
|
||||
|
||||
VectorDiffusionIntegrator::VectorDiffusionIntegrator(Coefficient &q,
|
||||
int vector_dimension)
|
||||
: VectorDiffusionIntegrator()
|
||||
{
|
||||
Q = &q;
|
||||
vdim = vector_dimension;
|
||||
}
|
||||
|
||||
VectorDiffusionIntegrator::VectorDiffusionIntegrator(VectorCoefficient &vq)
|
||||
: VectorDiffusionIntegrator()
|
||||
{
|
||||
VQ = &vq;
|
||||
vdim = vq.GetVDim();
|
||||
}
|
||||
|
||||
VectorDiffusionIntegrator::VectorDiffusionIntegrator(MatrixCoefficient &mq)
|
||||
: VectorDiffusionIntegrator()
|
||||
{
|
||||
MQ = &mq;
|
||||
vdim = mq.GetVDim();
|
||||
}
|
||||
|
||||
// PA Diffusion Assemble 2D kernel
|
||||
static void PAVectorDiffusionSetup2D(const int Q1D,
|
||||
const int NE,
|
||||
@@ -425,322 +474,6 @@ void VectorDiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
}
|
||||
}
|
||||
|
||||
// PA Diffusion Apply 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_VDIM = 0> static
|
||||
void PAVectorDiffusionApply2D(const int NE,
|
||||
const Array<real_t> &b,
|
||||
const Array<real_t> &g,
|
||||
const Array<real_t> &bt,
|
||||
const Array<real_t> >,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0,
|
||||
const int vdim = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(bt.Read(), D1D, Q1D);
|
||||
auto Gt = Reshape(gt.Read(), D1D, Q1D);
|
||||
auto D = Reshape(d_.Read(), Q1D*Q1D, 3, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
|
||||
real_t grad[max_Q1D][max_Q1D][2];
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
grad[qy][qx][0] = 0.0;
|
||||
grad[qy][qx][1] = 0.0;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
real_t gradX[max_Q1D][2];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
gradX[qx][0] = 0.0;
|
||||
gradX[qx][1] = 0.0;
|
||||
}
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const real_t s = x(dx,dy,c,e);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
gradX[qx][0] += s * B(qx,dx);
|
||||
gradX[qx][1] += s * G(qx,dx);
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t wy = B(qy,dy);
|
||||
const real_t wDy = G(qy,dy);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
grad[qy][qx][0] += gradX[qx][1] * wy;
|
||||
grad[qy][qx][1] += gradX[qx][0] * wDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxy, xDy in plane
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const int q = qx + qy * Q1D;
|
||||
const real_t O11 = D(q,0,e);
|
||||
const real_t O12 = D(q,1,e);
|
||||
const real_t O22 = D(q,2,e);
|
||||
const real_t gradX = grad[qy][qx][0];
|
||||
const real_t gradY = grad[qy][qx][1];
|
||||
grad[qy][qx][0] = (O11 * gradX) + (O12 * gradY);
|
||||
grad[qy][qx][1] = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
real_t gradX[max_D1D][2];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
gradX[dx][0] = 0.0;
|
||||
gradX[dx][1] = 0.0;
|
||||
}
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t gX = grad[qy][qx][0];
|
||||
const real_t gY = grad[qy][qx][1];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const real_t wx = Bt(dx,qx);
|
||||
const real_t wDx = Gt(dx,qx);
|
||||
gradX[dx][0] += gX * wDx;
|
||||
gradX[dx][1] += gY * wx;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const real_t wy = Bt(dy,qy);
|
||||
const real_t wDy = Gt(dy,qy);
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
y(dx,dy,c,e) += ((gradX[dx][0] * wy) + (gradX[dx][1] * wDy));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// PA Diffusion Apply 3D kernel
|
||||
template<const int T_D1D = 0,
|
||||
const int T_Q1D = 0> static
|
||||
void PAVectorDiffusionApply3D(const int NE,
|
||||
const Array<real_t> &b,
|
||||
const Array<real_t> &g,
|
||||
const Array<real_t> &bt,
|
||||
const Array<real_t> >,
|
||||
const Vector &op_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
int d1d = 0, int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int VDIM = 3;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(bt.Read(), D1D, Q1D);
|
||||
auto Gt = Reshape(gt.Read(), D1D, Q1D);
|
||||
auto op = Reshape(op_.Read(), Q1D*Q1D*Q1D, 6, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
for (int c = 0; c < VDIM; ++ c)
|
||||
{
|
||||
real_t grad[max_Q1D][max_Q1D][max_Q1D][3];
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
grad[qz][qy][qx][0] = 0.0;
|
||||
grad[qz][qy][qx][1] = 0.0;
|
||||
grad[qz][qy][qx][2] = 0.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
real_t gradXY[max_Q1D][max_Q1D][3];
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
gradXY[qy][qx][0] = 0.0;
|
||||
gradXY[qy][qx][1] = 0.0;
|
||||
gradXY[qy][qx][2] = 0.0;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
real_t gradX[max_Q1D][2];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
gradX[qx][0] = 0.0;
|
||||
gradX[qx][1] = 0.0;
|
||||
}
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const real_t s = x(dx,dy,dz,c,e);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
gradX[qx][0] += s * B(qx,dx);
|
||||
gradX[qx][1] += s * G(qx,dx);
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t wy = B(qy,dy);
|
||||
const real_t wDy = G(qy,dy);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t wx = gradX[qx][0];
|
||||
const real_t wDx = gradX[qx][1];
|
||||
gradXY[qy][qx][0] += wDx * wy;
|
||||
gradXY[qy][qx][1] += wx * wDy;
|
||||
gradXY[qy][qx][2] += wx * wy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const real_t wz = B(qz,dz);
|
||||
const real_t wDz = G(qz,dz);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
grad[qz][qy][qx][0] += gradXY[qy][qx][0] * wz;
|
||||
grad[qz][qy][qx][1] += gradXY[qy][qx][1] * wz;
|
||||
grad[qz][qy][qx][2] += gradXY[qy][qx][2] * wDz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const int q = qx + (qy + qz * Q1D) * Q1D;
|
||||
const real_t O11 = op(q,0,e);
|
||||
const real_t O12 = op(q,1,e);
|
||||
const real_t O13 = op(q,2,e);
|
||||
const real_t O22 = op(q,3,e);
|
||||
const real_t O23 = op(q,4,e);
|
||||
const real_t O33 = op(q,5,e);
|
||||
const real_t gradX = grad[qz][qy][qx][0];
|
||||
const real_t gradY = grad[qz][qy][qx][1];
|
||||
const real_t gradZ = grad[qz][qy][qx][2];
|
||||
grad[qz][qy][qx][0] = (O11*gradX)+(O12*gradY)+(O13*gradZ);
|
||||
grad[qz][qy][qx][1] = (O12*gradX)+(O22*gradY)+(O23*gradZ);
|
||||
grad[qz][qy][qx][2] = (O13*gradX)+(O23*gradY)+(O33*gradZ);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
real_t gradXY[max_D1D][max_D1D][3];
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
gradXY[dy][dx][0] = 0;
|
||||
gradXY[dy][dx][1] = 0;
|
||||
gradXY[dy][dx][2] = 0;
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
real_t gradX[max_D1D][3];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
gradX[dx][0] = 0;
|
||||
gradX[dx][1] = 0;
|
||||
gradX[dx][2] = 0;
|
||||
}
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t gX = grad[qz][qy][qx][0];
|
||||
const real_t gY = grad[qz][qy][qx][1];
|
||||
const real_t gZ = grad[qz][qy][qx][2];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const real_t wx = Bt(dx,qx);
|
||||
const real_t wDx = Gt(dx,qx);
|
||||
gradX[dx][0] += gX * wDx;
|
||||
gradX[dx][1] += gY * wx;
|
||||
gradX[dx][2] += gZ * wx;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const real_t wy = Bt(dy,qy);
|
||||
const real_t wDy = Gt(dy,qy);
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
gradXY[dy][dx][0] += gradX[dx][0] * wy;
|
||||
gradXY[dy][dx][1] += gradX[dx][1] * wDy;
|
||||
gradXY[dy][dx][2] += gradX[dx][2] * wy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
const real_t wz = Bt(dz,qz);
|
||||
const real_t wDz = Gt(dz,qz);
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
y(dx,dy,dz,c,e) +=
|
||||
((gradXY[dy][dx][0] * wz) +
|
||||
(gradXY[dy][dx][1] * wz) +
|
||||
(gradXY[dy][dx][2] * wDz));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// PA Diffusion Apply kernel
|
||||
void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
@@ -757,27 +490,29 @@ void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
const Array<real_t> &Bt = maps->Bt;
|
||||
const Array<real_t> &Gt = maps->Gt;
|
||||
const Vector &D = pa_data;
|
||||
|
||||
if (dim == 2 && sdim == 3)
|
||||
{
|
||||
switch ((dofs1D << 4 ) | quad1D)
|
||||
{
|
||||
case 0x22: return PAVectorDiffusionApply2D<2,2,3>(ne,B,G,Bt,Gt,D,x,y);
|
||||
case 0x33: return PAVectorDiffusionApply2D<3,3,3>(ne,B,G,Bt,Gt,D,x,y);
|
||||
case 0x44: return PAVectorDiffusionApply2D<4,4,3>(ne,B,G,Bt,Gt,D,x,y);
|
||||
case 0x55: return PAVectorDiffusionApply2D<5,5,3>(ne,B,G,Bt,Gt,D,x,y);
|
||||
default:
|
||||
return PAVectorDiffusionApply2D(ne,B,G,Bt,Gt,D,x,y,D1D,Q1D,sdim);
|
||||
}
|
||||
}
|
||||
if (dim == 2 && sdim == 2)
|
||||
{ return PAVectorDiffusionApply2D(ne,B,G,Bt,Gt,D,x,y,D1D,Q1D,sdim); }
|
||||
|
||||
if (dim == 3 && sdim == 3)
|
||||
{ return PAVectorDiffusionApply3D(ne,B,G,Bt,Gt,D,x,y,D1D,Q1D); }
|
||||
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
ApplyPAKernels::Run(dim, sdim, D1D, Q1D, ne, B, G, Bt, Gt, D, x, y, D1D,
|
||||
Q1D, sdim);
|
||||
}
|
||||
}
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
|
||||
VectorDiffusionIntegrator::ApplyKernelType
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Fallback(int DIM, int, int, int)
|
||||
{
|
||||
if (DIM == 2) { return internal::PAVectorDiffusionApply2D; }
|
||||
else if (DIM == 3) { return internal::PAVectorDiffusionApply3D; }
|
||||
else { MFEM_ABORT(""); }
|
||||
}
|
||||
|
||||
VectorDiffusionIntegrator::Kernels::Kernels()
|
||||
{
|
||||
VectorDiffusionIntegrator::AddSpecialization<2, 3, 2, 2>();
|
||||
VectorDiffusionIntegrator::AddSpecialization<2, 3, 3, 3>();
|
||||
VectorDiffusionIntegrator::AddSpecialization<2, 3, 4, 4>();
|
||||
VectorDiffusionIntegrator::AddSpecialization<2, 3, 5, 5>();
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
+55
-204
@@ -9,183 +9,19 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../fem/kernels.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../fem.hpp"
|
||||
|
||||
#include "lininteg_domain_kernels.hpp"
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void DLFEvalAssemble2D(const int vdim, const int ne, const int d,
|
||||
const int q,
|
||||
const int map_type, const int *markers, const real_t *b,
|
||||
const real_t *detj, const real_t *weights,
|
||||
const Vector &coeff, real_t *y)
|
||||
{
|
||||
const auto F = coeff.Read();
|
||||
const auto M = Reshape(markers, ne);
|
||||
const auto B = Reshape(b, q, d);
|
||||
const auto DETJ = Reshape(detj, q, q, ne);
|
||||
const auto W = Reshape(weights, q, q);
|
||||
const bool cst = coeff.Size() == vdim;
|
||||
const auto C = cst ? Reshape(F,vdim,1,1,1) : Reshape(F,vdim,q,q,ne);
|
||||
auto Y = Reshape(y, d,d, vdim, ne);
|
||||
|
||||
mfem::forall_2D(ne, q, q, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
if (M(e) == 0) { return; } // ignore
|
||||
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
|
||||
MFEM_SHARED real_t sBt[Q*D];
|
||||
MFEM_SHARED real_t sQQ[Q*Q];
|
||||
MFEM_SHARED real_t sQD[Q*D];
|
||||
|
||||
const DeviceMatrix Bt(sBt, d, q);
|
||||
kernels::internal::LoadB<D,Q>(d, q, B, sBt);
|
||||
|
||||
const DeviceMatrix QQ(sQQ, q, q);
|
||||
const DeviceMatrix QD(sQD, q, d);
|
||||
|
||||
for (int c = 0; c < vdim; ++c)
|
||||
{
|
||||
const real_t cst_val = C(c,0,0,0);
|
||||
MFEM_FOREACH_THREAD(x,x,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(y,y,q)
|
||||
{
|
||||
const real_t detJ = (map_type == FiniteElement::VALUE) ? DETJ(x,y,e) : 1.0;
|
||||
const real_t coeff_val = cst ? cst_val : C(c,x,y,e);
|
||||
QQ(y,x) = W(x,y) * coeff_val * detJ;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,d)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < q; ++qx) { u += QQ(qy,qx) * Bt(dx,qx); }
|
||||
QD(qy,dx) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,d)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < q; ++qy) { u += QD(qy,dx) * Bt(dy,qy); }
|
||||
Y(dx,dy,c,e) += u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void DLFEvalAssemble3D(const int vdim, const int ne, const int d,
|
||||
const int q,
|
||||
const int map_type, const int *markers, const real_t *b,
|
||||
const real_t *detj, const real_t *weights,
|
||||
const Vector &coeff, real_t *y)
|
||||
{
|
||||
const auto F = coeff.Read();
|
||||
const auto M = Reshape(markers, ne);
|
||||
const auto B = Reshape(b, q,d);
|
||||
const auto DETJ = Reshape(detj, q, q, q, ne);
|
||||
const auto W = Reshape(weights, q,q,q);
|
||||
const bool cst_coeff = coeff.Size() == vdim;
|
||||
const auto C = cst_coeff ? Reshape(F,vdim,1,1,1,1):Reshape(F,vdim,q,q,q,ne);
|
||||
|
||||
auto Y = Reshape(y, d,d,d, vdim, ne);
|
||||
|
||||
mfem::forall_2D(ne, q, q, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
if (M(e) == 0) { return; } // ignore
|
||||
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int MQD = (Q >= D) ? Q : D;
|
||||
|
||||
real_t u[D];
|
||||
|
||||
MFEM_SHARED real_t sBt[Q*D];
|
||||
const DeviceMatrix Bt(sBt, d,q);
|
||||
kernels::internal::LoadB<D,Q>(d,q,B,sBt);
|
||||
|
||||
MFEM_SHARED real_t sQQQ[MQD*MQD*MQD];
|
||||
const DeviceCube QQQ(sQQQ, MQD, MQD, MQD);
|
||||
|
||||
for (int c = 0; c < vdim; ++c)
|
||||
{
|
||||
const real_t cst_val = C(c,0,0,0,0);
|
||||
MFEM_FOREACH_THREAD(x,x,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(y,y,q)
|
||||
{
|
||||
for (int z = 0; z < q; ++z)
|
||||
{
|
||||
const real_t detJ = (map_type == FiniteElement::VALUE) ? DETJ(x,y,z,e) : 1.0;
|
||||
const real_t coeff_val = cst_coeff ? cst_val : C(c,x,y,z,e);
|
||||
QQQ(z,y,x) = W(x,y,z) * coeff_val * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qx,x,q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,q)
|
||||
{
|
||||
for (int dz = 0; dz < d; ++dz) { u[dz] = 0.0; }
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
const real_t ZYX = QQQ(qz,qy,qx);
|
||||
for (int dz = 0; dz < d; ++dz) { u[dz] += ZYX * Bt(dz,qz); }
|
||||
}
|
||||
for (int dz = 0; dz < d; ++dz) { QQQ(dz,qy,qx) = u[dz]; }
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,y,d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,q)
|
||||
{
|
||||
for (int dy = 0; dy < d; ++dy) { u[dy] = 0.0; }
|
||||
for (int qy = 0; qy < q; ++qy)
|
||||
{
|
||||
const real_t zYX = QQQ(dz,qy,qx);
|
||||
for (int dy = 0; dy < d; ++dy) { u[dy] += zYX * Bt(dy,qy); }
|
||||
}
|
||||
for (int dy = 0; dy < d; ++dy) { QQQ(dz,dy,qx) = u[dy]; }
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,y,d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,x,d)
|
||||
{
|
||||
for (int dx = 0; dx < d; ++dx) { u[dx] = 0.0; }
|
||||
for (int qx = 0; qx < q; ++qx)
|
||||
{
|
||||
const real_t zyX = QQQ(dz,dy,qx);
|
||||
for (int dx = 0; dx < d; ++dx) { u[dx] += zyX * Bt(dx,qx); }
|
||||
}
|
||||
for (int dx = 0; dx < d; ++dx) { Y(dx,dy,dz,c,e) += u[dx]; }
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static void DLFEvalAssemble(const FiniteElementSpace &fes,
|
||||
const IntegrationRule *ir,
|
||||
const Array<int> &markers,
|
||||
const Vector &coeff,
|
||||
const Array<int> &markers, const Vector &coeff,
|
||||
Vector &y)
|
||||
{
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
@@ -197,50 +33,20 @@ static void DLFEvalAssemble(const FiniteElementSpace &fes,
|
||||
constexpr int flags = GeometricFactors::DETERMINANTS;
|
||||
const GeometricFactors *geom = mesh->GetGeometricFactors(*ir, flags, mt);
|
||||
const int map_type = fes.GetTypicalFE()->GetMapType();
|
||||
decltype(&DLFEvalAssemble2D<>) ker =
|
||||
dim == 2 ? DLFEvalAssemble2D<> : DLFEvalAssemble3D<>;
|
||||
|
||||
if (dim==2)
|
||||
{
|
||||
if (d==1 && q==1) { ker=DLFEvalAssemble2D<1,1>; }
|
||||
if (d==2 && q==2) { ker=DLFEvalAssemble2D<2,2>; }
|
||||
if (d==3 && q==3) { ker=DLFEvalAssemble2D<3,3>; }
|
||||
if (d==4 && q==4) { ker=DLFEvalAssemble2D<4,4>; }
|
||||
if (d==5 && q==5) { ker=DLFEvalAssemble2D<5,5>; }
|
||||
if (d==2 && q==3) { ker=DLFEvalAssemble2D<2,3>; }
|
||||
if (d==3 && q==4) { ker=DLFEvalAssemble2D<3,4>; }
|
||||
if (d==4 && q==5) { ker=DLFEvalAssemble2D<4,5>; }
|
||||
if (d==5 && q==6) { ker=DLFEvalAssemble2D<5,6>; }
|
||||
}
|
||||
|
||||
if (dim==3)
|
||||
{
|
||||
if (d==1 && q==1) { ker=DLFEvalAssemble3D<1,1>; }
|
||||
if (d==2 && q==2) { ker=DLFEvalAssemble3D<2,2>; }
|
||||
if (d==3 && q==3) { ker=DLFEvalAssemble3D<3,3>; }
|
||||
if (d==4 && q==4) { ker=DLFEvalAssemble3D<4,4>; }
|
||||
if (d==5 && q==5) { ker=DLFEvalAssemble3D<5,5>; }
|
||||
if (d==2 && q==3) { ker=DLFEvalAssemble3D<2,3>; }
|
||||
if (d==3 && q==4) { ker=DLFEvalAssemble3D<3,4>; }
|
||||
if (d==4 && q==5) { ker=DLFEvalAssemble3D<4,5>; }
|
||||
if (d==5 && q==6) { ker=DLFEvalAssemble3D<5,6>; }
|
||||
}
|
||||
|
||||
MFEM_VERIFY(ker, "No kernel ndof " << d << " nqpt " << q);
|
||||
|
||||
const int vdim = fes.GetVDim();
|
||||
const int ne = fes.GetMesh()->GetNE();
|
||||
const int *M = markers.Read();
|
||||
const real_t *B = maps.B.Read();
|
||||
const int *M = markers.Read();
|
||||
const real_t *detJ = geom->detJ.Read();
|
||||
const real_t *W = ir->GetWeights().Read();
|
||||
real_t *Y = y.ReadWrite();
|
||||
ker(vdim, ne, d, q, map_type, M, B, detJ, W, coeff, Y);
|
||||
DomainLFIntegrator::AssembleKernels::Run(dim, d, q, vdim, ne, d, q, map_type,
|
||||
M, B, detJ, W, coeff, Y);
|
||||
}
|
||||
|
||||
void DomainLFIntegrator::AssembleDevice(const FiniteElementSpace &fes,
|
||||
const Array<int> &markers,
|
||||
Vector &b)
|
||||
const Array<int> &markers, Vector &b)
|
||||
{
|
||||
const FiniteElement &fe = *fes.GetTypicalFE();
|
||||
const int qorder = oa * fe.GetOrder() + ob;
|
||||
@@ -266,4 +72,49 @@ void VectorDomainLFIntegrator::AssembleDevice(const FiniteElementSpace &fes,
|
||||
DLFEvalAssemble(fes, ir, markers, coeff, b);
|
||||
}
|
||||
|
||||
DomainLFIntegrator::AssembleKernelType
|
||||
DomainLFIntegrator::AssembleKernels::Fallback(int DIM, int, int)
|
||||
{
|
||||
switch (DIM)
|
||||
{
|
||||
case 1:
|
||||
return DLFEvalAssemble1D<0, 0>;
|
||||
case 2:
|
||||
return DLFEvalAssemble2D<0, 0>;
|
||||
case 3:
|
||||
return DLFEvalAssemble3D<0, 0>;
|
||||
}
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
DomainLFIntegrator::Kernels::Kernels()
|
||||
{
|
||||
// 2D
|
||||
// Q = P+1
|
||||
DomainLFIntegrator::AddSpecialization<2, 1, 1>();
|
||||
DomainLFIntegrator::AddSpecialization<2, 2, 2>();
|
||||
DomainLFIntegrator::AddSpecialization<2, 3, 3>();
|
||||
DomainLFIntegrator::AddSpecialization<2, 4, 4>();
|
||||
DomainLFIntegrator::AddSpecialization<2, 5, 5>();
|
||||
// Q = P+2
|
||||
DomainLFIntegrator::AddSpecialization<2, 2, 3>();
|
||||
DomainLFIntegrator::AddSpecialization<2, 3, 4>();
|
||||
DomainLFIntegrator::AddSpecialization<2, 4, 5>();
|
||||
DomainLFIntegrator::AddSpecialization<2, 5, 6>();
|
||||
// 3D
|
||||
// Q = P+1
|
||||
DomainLFIntegrator::AddSpecialization<3, 1, 1>();
|
||||
DomainLFIntegrator::AddSpecialization<3, 2, 2>();
|
||||
DomainLFIntegrator::AddSpecialization<3, 3, 3>();
|
||||
DomainLFIntegrator::AddSpecialization<3, 4, 4>();
|
||||
DomainLFIntegrator::AddSpecialization<3, 5, 5>();
|
||||
// Q = P+2
|
||||
DomainLFIntegrator::AddSpecialization<3, 2, 3>();
|
||||
DomainLFIntegrator::AddSpecialization<3, 3, 4>();
|
||||
DomainLFIntegrator::AddSpecialization<3, 4, 5>();
|
||||
DomainLFIntegrator::AddSpecialization<3, 5, 6>();
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -0,0 +1,318 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_LININTEG_DOMAIN_KERNELS_HPP
|
||||
#define MFEM_LININTEG_DOMAIN_KERNELS_HPP
|
||||
|
||||
#include "../../fem/kernels.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../fem.hpp"
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void DLFEvalAssemble1D(const int vdim, const int ne, const int d,
|
||||
const int q, const int map_type,
|
||||
const int *markers, const real_t *b,
|
||||
const real_t *detj, const real_t *weights,
|
||||
const Vector &coeff, real_t *y)
|
||||
{
|
||||
{
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
MFEM_VERIFY(q <= Q, "");
|
||||
MFEM_VERIFY(d <= D, "");
|
||||
}
|
||||
|
||||
const auto F = coeff.Read();
|
||||
const auto B = Reshape(b, q, d);
|
||||
const auto DETJ = Reshape(detj, q, ne);
|
||||
const bool cst = coeff.Size() == vdim;
|
||||
const auto C = cst ? Reshape(F, vdim, 1, 1) : Reshape(F, vdim, q, ne);
|
||||
auto Y = Reshape(y, d, vdim, ne);
|
||||
|
||||
mfem::forall_2D(ne, d, 1, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
if (markers[e] == 0)
|
||||
{
|
||||
return;
|
||||
} // ignore
|
||||
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
|
||||
MFEM_SHARED real_t sBt[Q * D];
|
||||
const DeviceMatrix Bt(sBt, d, q);
|
||||
kernels::internal::LoadB<D, Q>(d, q, B, sBt);
|
||||
|
||||
for (int c = 0; c < vdim; ++c)
|
||||
{
|
||||
const real_t cst_val = C(c, 0, 0);
|
||||
MFEM_FOREACH_THREAD(dx, x, d)
|
||||
{
|
||||
real_t u = 0;
|
||||
for (int qx = 0; qx < q; ++qx)
|
||||
{
|
||||
const real_t detJ =
|
||||
(map_type == FiniteElement::VALUE) ? DETJ(qx, e) : 1.0;
|
||||
const real_t coeff_val = cst ? cst_val : C(c, qx, e);
|
||||
u += weights[qx] * coeff_val * detJ * Bt(dx, qx);
|
||||
}
|
||||
Y(dx, c, e) += u;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void DLFEvalAssemble2D(const int vdim, const int ne, const int d,
|
||||
const int q, const int map_type,
|
||||
const int *markers, const real_t *b,
|
||||
const real_t *detj, const real_t *weights,
|
||||
const Vector &coeff, real_t *y)
|
||||
{
|
||||
{
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
MFEM_VERIFY(q <= Q, "");
|
||||
MFEM_VERIFY(d <= D, "");
|
||||
}
|
||||
|
||||
const auto F = coeff.Read();
|
||||
const auto B = Reshape(b, q, d);
|
||||
const auto DETJ = Reshape(detj, q, q, ne);
|
||||
const auto W = Reshape(weights, q, q);
|
||||
const bool cst = coeff.Size() == vdim;
|
||||
const auto C = cst ? Reshape(F, vdim, 1, 1, 1) : Reshape(F, vdim, q, q, ne);
|
||||
auto Y = Reshape(y, d, d, vdim, ne);
|
||||
|
||||
mfem::forall_2D(ne, q, q, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
if (markers[e] == 0)
|
||||
{
|
||||
return;
|
||||
} // ignore
|
||||
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
|
||||
MFEM_SHARED real_t sBt[Q * D];
|
||||
MFEM_SHARED real_t sQQ[Q * Q];
|
||||
MFEM_SHARED real_t sQD[Q * D];
|
||||
|
||||
const DeviceMatrix Bt(sBt, d, q);
|
||||
kernels::internal::LoadB<D, Q>(d, q, B, sBt);
|
||||
|
||||
const DeviceMatrix QQ(sQQ, q, q);
|
||||
const DeviceMatrix QD(sQD, q, d);
|
||||
|
||||
for (int c = 0; c < vdim; ++c)
|
||||
{
|
||||
const real_t cst_val = C(c, 0, 0, 0);
|
||||
MFEM_FOREACH_THREAD(x, x, q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(y, y, q)
|
||||
{
|
||||
const real_t detJ =
|
||||
(map_type == FiniteElement::VALUE) ? DETJ(x, y, e) : 1.0;
|
||||
const real_t coeff_val = cst ? cst_val : C(c, x, y, e);
|
||||
QQ(y, x) = W(x, y) * coeff_val * detJ;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy, y, q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < q; ++qx)
|
||||
{
|
||||
u += QQ(qy, qx) * Bt(dx, qx);
|
||||
}
|
||||
QD(qy, dx) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy, y, d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < q; ++qy)
|
||||
{
|
||||
u += QD(qy, dx) * Bt(dy, qy);
|
||||
}
|
||||
Y(dx, dy, c, e) += u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void DLFEvalAssemble3D(const int vdim, const int ne, const int d,
|
||||
const int q, const int map_type,
|
||||
const int* markers, const real_t *b,
|
||||
const real_t *detj, const real_t *weights,
|
||||
const Vector &coeff, real_t *y)
|
||||
{
|
||||
{
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
MFEM_VERIFY(q <= Q, "");
|
||||
MFEM_VERIFY(d <= D, "");
|
||||
}
|
||||
|
||||
const auto F = coeff.Read();
|
||||
const auto B = Reshape(b, q, d);
|
||||
const auto DETJ = Reshape(detj, q, q, q, ne);
|
||||
const auto W = Reshape(weights, q, q, q);
|
||||
const bool cst_coeff = coeff.Size() == vdim;
|
||||
const auto C =
|
||||
cst_coeff ? Reshape(F, vdim, 1, 1, 1, 1) : Reshape(F, vdim, q, q, q, ne);
|
||||
|
||||
auto Y = Reshape(y, d, d, d, vdim, ne);
|
||||
|
||||
mfem::forall_2D(ne, q, q, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
if (markers[e] == 0)
|
||||
{
|
||||
return;
|
||||
} // ignore
|
||||
|
||||
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int MQD = (Q >= D) ? Q : D;
|
||||
|
||||
real_t u[D];
|
||||
|
||||
MFEM_SHARED real_t sBt[Q * D];
|
||||
const DeviceMatrix Bt(sBt, d, q);
|
||||
kernels::internal::LoadB<D, Q>(d, q, B, sBt);
|
||||
|
||||
MFEM_SHARED real_t sQQQ[MQD * MQD * MQD];
|
||||
const DeviceCube QQQ(sQQQ, MQD, MQD, MQD);
|
||||
|
||||
for (int c = 0; c < vdim; ++c)
|
||||
{
|
||||
const real_t cst_val = C(c, 0, 0, 0, 0);
|
||||
MFEM_FOREACH_THREAD(x, x, q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(y, y, q)
|
||||
{
|
||||
for (int z = 0; z < q; ++z)
|
||||
{
|
||||
const real_t detJ = (map_type == FiniteElement::VALUE)
|
||||
? DETJ(x, y, z, e)
|
||||
: 1.0;
|
||||
const real_t coeff_val =
|
||||
cst_coeff ? cst_val : C(c, x, y, z, e);
|
||||
QQQ(z, y, x) = W(x, y, z) * coeff_val * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qx, x, q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q)
|
||||
{
|
||||
for (int dz = 0; dz < d; ++dz)
|
||||
{
|
||||
u[dz] = 0.0;
|
||||
}
|
||||
for (int qz = 0; qz < q; ++qz)
|
||||
{
|
||||
const real_t ZYX = QQQ(qz, qy, qx);
|
||||
for (int dz = 0; dz < d; ++dz)
|
||||
{
|
||||
u[dz] += ZYX * Bt(dz, qz);
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < d; ++dz)
|
||||
{
|
||||
QQQ(dz, qy, qx) = u[dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz, y, d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q)
|
||||
{
|
||||
for (int dy = 0; dy < d; ++dy)
|
||||
{
|
||||
u[dy] = 0.0;
|
||||
}
|
||||
for (int qy = 0; qy < q; ++qy)
|
||||
{
|
||||
const real_t zYX = QQQ(dz, qy, qx);
|
||||
for (int dy = 0; dy < d; ++dy)
|
||||
{
|
||||
u[dy] += zYX * Bt(dy, qy);
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < d; ++dy)
|
||||
{
|
||||
QQQ(dz, dy, qx) = u[dy];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz, y, d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, x, d)
|
||||
{
|
||||
for (int dx = 0; dx < d; ++dx)
|
||||
{
|
||||
u[dx] = 0.0;
|
||||
}
|
||||
for (int qx = 0; qx < q; ++qx)
|
||||
{
|
||||
const real_t zyX = QQQ(dz, dy, qx);
|
||||
for (int dx = 0; dx < d; ++dx)
|
||||
{
|
||||
u[dx] += zyX * Bt(dx, qx);
|
||||
}
|
||||
}
|
||||
for (int dx = 0; dx < d; ++dx)
|
||||
{
|
||||
Y(dx, dy, dz, c, e) += u[dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <int DIM, int T_D1D, int T_Q1D>
|
||||
DomainLFIntegrator::AssembleKernelType
|
||||
DomainLFIntegrator::AssembleKernels::Kernel()
|
||||
{
|
||||
switch (DIM)
|
||||
{
|
||||
case 1:
|
||||
return DLFEvalAssemble1D<T_D1D, T_Q1D>;
|
||||
case 2:
|
||||
return DLFEvalAssemble2D<T_D1D, T_Q1D>;
|
||||
case 3:
|
||||
return DLFEvalAssemble3D<T_D1D, T_Q1D>;
|
||||
}
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
#endif
|
||||
@@ -947,6 +947,7 @@ int Quadrature1D::CheckOpen(int type)
|
||||
case OpenUniform:
|
||||
case ClosedUniform:
|
||||
case OpenHalfUniform:
|
||||
case ClosedGL:
|
||||
return type; // all types can work as open
|
||||
default:
|
||||
return Invalid;
|
||||
|
||||
@@ -78,7 +78,7 @@ namespace mfem
|
||||
public: \
|
||||
const char *kernel_name = MFEM_KERNEL_NAME(KernelName); \
|
||||
using KernelSignature = KernelType; \
|
||||
template <MFEM_PARAM_LIST P3> static MFEM_EXPORT KernelSignature Kernel(); \
|
||||
template <MFEM_PARAM_LIST P3> static KernelSignature Kernel(); \
|
||||
static MFEM_EXPORT KernelSignature Fallback(MFEM_PARAM_LIST P1); \
|
||||
static MFEM_EXPORT KernelName &Get() { \
|
||||
static KernelName table; \
|
||||
|
||||
@@ -35,6 +35,19 @@ void LinearFormIntegrator::AssembleRHSElementVect(
|
||||
mfem_error("LinearFormIntegrator::AssembleRHSElementVect(...)");
|
||||
}
|
||||
|
||||
DomainLFIntegrator::DomainLFIntegrator(Coefficient &QF, int a, int b)
|
||||
: DeltaLFIntegrator(QF), Q(QF), oa(a), ob(b)
|
||||
{
|
||||
static Kernels kernels;
|
||||
}
|
||||
|
||||
DomainLFIntegrator::DomainLFIntegrator(Coefficient &QF,
|
||||
const IntegrationRule *ir)
|
||||
: DeltaLFIntegrator(QF, ir), Q(QF), oa(1), ob(1)
|
||||
{
|
||||
static Kernels kernels;
|
||||
}
|
||||
|
||||
void DomainLFIntegrator::AssembleRHSElementVect(const FiniteElement &el,
|
||||
ElementTransformation &Tr,
|
||||
Vector &elvect)
|
||||
@@ -266,6 +279,13 @@ void BoundaryTangentialLFIntegrator::AssembleRHSElementVect(
|
||||
}
|
||||
}
|
||||
|
||||
VectorDomainLFIntegrator::VectorDomainLFIntegrator(VectorCoefficient &QF,
|
||||
const IntegrationRule *ir)
|
||||
: DeltaLFIntegrator(QF, ir), Q(QF)
|
||||
{
|
||||
static DomainLFIntegrator::Kernels kernels;
|
||||
}
|
||||
|
||||
void VectorDomainLFIntegrator::AssembleRHSElementVect(
|
||||
const FiniteElement &el, ElementTransformation &Tr, Vector &elvect)
|
||||
{
|
||||
|
||||
+32
-11
@@ -18,6 +18,8 @@
|
||||
#include <random>
|
||||
#include "integrator.hpp"
|
||||
|
||||
#include "kernel_dispatch.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
@@ -109,14 +111,12 @@ class DomainLFIntegrator : public DeltaLFIntegrator
|
||||
int oa, ob;
|
||||
public:
|
||||
/// Constructs a domain integrator with a given Coefficient
|
||||
DomainLFIntegrator(Coefficient &QF, int a = 2, int b = 0)
|
||||
// the old default was a = 1, b = 1
|
||||
// for simple elliptic problems a = 2, b = -2 is OK
|
||||
: DeltaLFIntegrator(QF), Q(QF), oa(a), ob(b) { }
|
||||
/// the old default was a = 1, b = 1
|
||||
/// for simple elliptic problems a = 2, b = -2 is OK
|
||||
DomainLFIntegrator(Coefficient &QF, int a = 2, int b = 0);
|
||||
|
||||
/// Constructs a domain integrator with a given Coefficient
|
||||
DomainLFIntegrator(Coefficient &QF, const IntegrationRule *ir)
|
||||
: DeltaLFIntegrator(QF, ir), Q(QF), oa(1), ob(1) { }
|
||||
DomainLFIntegrator(Coefficient &QF, const IntegrationRule *ir);
|
||||
|
||||
bool SupportsDevice() const override { return true; }
|
||||
|
||||
@@ -136,6 +136,22 @@ public:
|
||||
Vector &elvect) override;
|
||||
|
||||
using LinearFormIntegrator::AssembleRHSElementVect;
|
||||
|
||||
/// args: vdim, ne, d1d, q1d, map_type, markers, B, detJ, W, coeff, y
|
||||
using AssembleKernelType = void (*)(const int, const int, const int,
|
||||
const int, const int, const int *,
|
||||
const real_t *, const real_t *,
|
||||
const real_t *, const Vector &coeff,
|
||||
real_t *y);
|
||||
|
||||
/// parameters: use DIM, T_D1D, T_Q1D
|
||||
MFEM_REGISTER_KERNELS(AssembleKernels, AssembleKernelType, (int, int, int));
|
||||
struct Kernels { Kernels(); };
|
||||
|
||||
template <int DIM, int D1D, int Q1D> static void AddSpecialization()
|
||||
{
|
||||
AssembleKernels::Specialization<DIM, D1D, Q1D>::Add();
|
||||
}
|
||||
};
|
||||
|
||||
/// Class for domain integrator $ L(v) := (f, \nabla v) $
|
||||
@@ -256,14 +272,13 @@ private:
|
||||
|
||||
public:
|
||||
/// Constructs a domain integrator with a given VectorCoefficient
|
||||
VectorDomainLFIntegrator(VectorCoefficient &QF)
|
||||
: DeltaLFIntegrator(QF), Q(QF) { }
|
||||
VectorDomainLFIntegrator(VectorCoefficient &QF,
|
||||
const IntegrationRule *ir = nullptr);
|
||||
|
||||
bool SupportsDevice() const override { return true; }
|
||||
|
||||
/// Method defining assembly on device
|
||||
void AssembleDevice(const FiniteElementSpace &fes,
|
||||
const Array<int> &markers,
|
||||
void AssembleDevice(const FiniteElementSpace &fes, const Array<int> &markers,
|
||||
Vector &b) override;
|
||||
|
||||
/** Given a particular Finite Element and a transformation (Tr)
|
||||
@@ -277,6 +292,12 @@ public:
|
||||
Vector &elvect) override;
|
||||
|
||||
using LinearFormIntegrator::AssembleRHSElementVect;
|
||||
|
||||
template <int DIM, int D1D, int Q1D> static void AddSpecialization()
|
||||
{
|
||||
// uses the same kernels for assembly
|
||||
DomainLFIntegrator::AssembleKernels::Specialization<DIM, D1D, Q1D>::Add();
|
||||
}
|
||||
};
|
||||
|
||||
/** Class for domain integrator $ L(v) := (f, \nabla v) $, where
|
||||
@@ -544,7 +565,7 @@ public:
|
||||
Specifically, given the Dirichlet data $u_D$, the linear form assembles the
|
||||
following integrals on the boundary:
|
||||
$$
|
||||
\sigma \langle u_D, (Q \nabla v)) \cdot n \rangle + \kappa \langle {h^{-1} Q} u_D, v \rangle,
|
||||
\sigma \langle u_D, (Q \nabla v) \cdot n \rangle + \kappa \langle {h^{-1} Q} u_D, v \rangle,
|
||||
$$
|
||||
where Q is a scalar or matrix diffusion coefficient and v is the test
|
||||
function. The parameters $\sigma$ and $\kappa$ should be the same as the ones
|
||||
|
||||
+264
-22
@@ -14,9 +14,11 @@
|
||||
#include "../../general/forall.hpp"
|
||||
#include <climits>
|
||||
#include "../pbilinearform.hpp"
|
||||
#include "../../fem/fe/face_map_utils.hpp"
|
||||
|
||||
// Specializations
|
||||
#include "lor_h1.hpp"
|
||||
#include "lor_dg.hpp"
|
||||
#include "lor_nd.hpp"
|
||||
#include "lor_rt.hpp"
|
||||
|
||||
@@ -54,17 +56,18 @@ bool BatchedLORAssembly::FormIsSupported(BilinearForm &a)
|
||||
// Batched LOR requires all tensor elements
|
||||
if (!UsesTensorBasis(*a.FESpace())) { return false; }
|
||||
|
||||
if (dynamic_cast<const H1_FECollection*>(fec))
|
||||
if (dynamic_cast<const H1_FECollection*>(fec) ||
|
||||
dynamic_cast<const DG_FECollection*>(fec))
|
||||
{
|
||||
if (HasIntegrators<DiffusionIntegrator, MassIntegrator>(a)) { return true; }
|
||||
return HasIntegrators<DiffusionIntegrator, MassIntegrator>(a);
|
||||
}
|
||||
else if (dynamic_cast<const ND_FECollection*>(fec))
|
||||
{
|
||||
if (HasIntegrators<CurlCurlIntegrator, VectorFEMassIntegrator>(a)) { return true; }
|
||||
return HasIntegrators<CurlCurlIntegrator, VectorFEMassIntegrator>(a);
|
||||
}
|
||||
else if (dynamic_cast<const RT_FECollection*>(fec))
|
||||
{
|
||||
if (HasIntegrators<DivDivIntegrator, VectorFEMassIntegrator>(a)) { return true; }
|
||||
return HasIntegrators<DivDivIntegrator, VectorFEMassIntegrator>(a);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -75,12 +78,14 @@ void BatchedLORAssembly::FormLORVertexCoordinates(FiniteElementSpace &fes_ho,
|
||||
Mesh &mesh_ho = *fes_ho.GetMesh();
|
||||
mesh_ho.EnsureNodes();
|
||||
|
||||
const bool dg = fes_ho.IsDGSpace();
|
||||
|
||||
// Get nodal points at the LOR vertices
|
||||
const int dim = mesh_ho.Dimension();
|
||||
const int sdim = mesh_ho.SpaceDimension();
|
||||
const int nel_ho = mesh_ho.GetNE();
|
||||
const int order = fes_ho.GetMaxElementOrder();
|
||||
const int nd1d = order + 1;
|
||||
const int nd1d = dg ? order + 2 : order + 1;
|
||||
const int ndof_per_el = static_cast<int>(pow(nd1d, dim));
|
||||
|
||||
const GridFunction *nodal_gf = mesh_ho.GetNodes();
|
||||
@@ -92,7 +97,8 @@ void BatchedLORAssembly::FormLORVertexCoordinates(FiniteElementSpace &fes_ho,
|
||||
Vector nodal_evec(nodal_restriction->Height());
|
||||
nodal_restriction->Mult(*nodal_gf, nodal_evec);
|
||||
|
||||
IntegrationRule ir = GetCollocatedIntRule(fes_ho);
|
||||
const IntegrationRule ir = GetLobattoIntRule(
|
||||
mesh_ho.GetTypicalElementGeometry(), nd1d);
|
||||
|
||||
// Map from nodal E-vector to Q-vector at the LOR vertex points
|
||||
X_vert.SetSize(sdim*ndof_per_el*nel_ho);
|
||||
@@ -159,6 +165,7 @@ int BatchedLORAssembly::FillI(SparseMatrix &A) const
|
||||
const auto K = dof_glob2loc_offsets_.Read();
|
||||
const auto map = Reshape(sparse_mapping.Read(), nnz_per_row, ndof_per_el);
|
||||
|
||||
|
||||
auto I = A.WriteI();
|
||||
|
||||
mfem::forall(nvdof + 1, [=] MFEM_HOST_DEVICE (int ii) { I[ii] = 0; });
|
||||
@@ -358,6 +365,177 @@ void BatchedLORAssembly::FillJAndData(SparseMatrix &A) const
|
||||
});
|
||||
}
|
||||
|
||||
void BatchedLORAssembly::SparseIJToCSR_DG(OperatorHandle &A) const
|
||||
{
|
||||
const int ndof_per_el = fes_ho.GetFE(0)->GetDof();
|
||||
const int nel_ho = fes_ho.GetNE();
|
||||
const int nnz_per_row = sparse_ij.Size()/ndof_per_el/nel_ho;
|
||||
const int dim = fes_ho.GetMesh()->Dimension();
|
||||
const int nrows = nel_ho*ndof_per_el;
|
||||
const int p = fes_ho.GetMaxElementOrder();
|
||||
const int pp1 = p + 1;
|
||||
const int nnz = nrows*nnz_per_row;
|
||||
|
||||
const int face_nbr_vsize = [&]()
|
||||
{
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (auto *par_fes = dynamic_cast<ParFiniteElementSpace*>(&fes_ho))
|
||||
{
|
||||
return par_fes->GetFaceNbrVSize();
|
||||
}
|
||||
#endif
|
||||
return 0;
|
||||
}();
|
||||
|
||||
// If A contains an existing SparseMatrix, reuse it (and try to reuse its
|
||||
// I, J, A arrays if they are big enough)
|
||||
SparseMatrix *A_mat = A.Is<SparseMatrix>();
|
||||
if (!A_mat)
|
||||
{
|
||||
A_mat = new SparseMatrix;
|
||||
A.Reset(A_mat);
|
||||
}
|
||||
|
||||
// The second argument (nrows + face_nbr_vsize) accounts for additional
|
||||
// columns contributed by DG face neighbors in parallel finite element
|
||||
// spaces. In serial, face_nbr_vsize is set to 0.
|
||||
A_mat->OverrideSize(nrows, nrows + face_nbr_vsize);
|
||||
|
||||
EnsureCapacity(A_mat->GetMemoryI(), nrows + 1);
|
||||
EnsureCapacity(A_mat->GetMemoryJ(), nnz);
|
||||
EnsureCapacity(A_mat->GetMemoryData(), nnz);
|
||||
|
||||
Array<int> nbr_info(nel_ho*3*2*dim);
|
||||
auto h_nbr_info = Reshape(nbr_info.HostWrite(), nel_ho, 2*dim, 3);
|
||||
const int num_faces = fes_ho.GetMesh()->GetNumFaces();
|
||||
for (int f = 0; f < num_faces; f++)
|
||||
{
|
||||
Mesh::FaceInformation finfo = fes_ho.GetMesh()->GetFaceInformation(f);
|
||||
int e0 = finfo.element[0].index;
|
||||
int f0 = finfo.element[0].local_face_id;
|
||||
if (finfo.IsBoundary())
|
||||
{
|
||||
h_nbr_info(e0,f0,0) = -1;
|
||||
h_nbr_info(e0,f0,1)= -1;
|
||||
h_nbr_info(e0,f0,2)= -1;
|
||||
}
|
||||
else if (finfo.IsShared())
|
||||
{
|
||||
// Face neighbors elements are indexed after the last local element
|
||||
h_nbr_info(e0,f0,0) = nel_ho + finfo.element[1].index;
|
||||
h_nbr_info(e0,f0,1)= finfo.element[1].orientation;
|
||||
h_nbr_info(e0,f0,2)= finfo.element[1].local_face_id;
|
||||
}
|
||||
else if (finfo.IsInterior())
|
||||
{
|
||||
int e1 = finfo.element[1].index;
|
||||
int f1 = finfo.element[1].local_face_id;
|
||||
h_nbr_info(e0,f0,0) = e1;
|
||||
h_nbr_info(e0,f0,1)= finfo.element[1].orientation;
|
||||
h_nbr_info(e0,f0,2)= f1;
|
||||
h_nbr_info(e1,f1,0) = e0;
|
||||
h_nbr_info(e1,f1,1) = finfo.element[1].orientation;
|
||||
h_nbr_info(e1,f1,2) = f0;
|
||||
}
|
||||
};
|
||||
|
||||
auto h_I = A_mat->HostWriteI();
|
||||
h_I[0] = 0;
|
||||
for (int i = 0; i < nrows; ++i)
|
||||
{
|
||||
const int iel_ho = i / ndof_per_el;
|
||||
const int iloc = i % ndof_per_el;
|
||||
static const int lex_map_2[4] = {3, 1, 0, 2};
|
||||
static const int lex_map_3[6] = {4, 2, 1, 3, 0, 5};
|
||||
const int local_i[3] = {iloc % pp1, (iloc/pp1)%pp1, iloc/pp1/pp1};
|
||||
int bdr_count = 0;
|
||||
for (int n_idx = 0; n_idx < dim; ++n_idx)
|
||||
{
|
||||
for (int e_i = 0; e_i < 2; ++e_i)
|
||||
{
|
||||
const int j_lex = e_i + n_idx*2;
|
||||
const int f = (dim == 3) ? lex_map_3[j_lex]:lex_map_2[j_lex];
|
||||
const bool boundary = (local_i[n_idx] == e_i * p);
|
||||
if (boundary)
|
||||
{
|
||||
int neighbor_idx = h_nbr_info(iel_ho, f, 0);
|
||||
if (neighbor_idx == -1)
|
||||
{
|
||||
++bdr_count;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
h_I[i+1] = h_I[i] + (nnz_per_row - bdr_count);
|
||||
}
|
||||
|
||||
const auto V = Reshape(sparse_ij.Read(), nnz_per_row, ndof_per_el, nel_ho);
|
||||
auto J = A_mat->WriteJ();
|
||||
auto AV = A_mat->WriteData();
|
||||
auto I = A_mat->ReadI();
|
||||
|
||||
auto d_nbr_info = Reshape(nbr_info.Read(), nel_ho, 2*dim, 3);
|
||||
mfem::forall(nrows, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
const int e = i / ndof_per_el;
|
||||
const int iloc = i % ndof_per_el;
|
||||
const int local_x = iloc % pp1;
|
||||
const int local_y = (iloc/pp1)%pp1;
|
||||
const int local_z = iloc/pp1/pp1;
|
||||
const int local_i[3] = {local_x, local_y, local_z};
|
||||
int offset = I[i];
|
||||
static const int lex_map_2[4] = {3, 1, 0, 2};
|
||||
static const int lex_map_3[6] = {4,2,1,3,0,5};
|
||||
const int *lex_map = (dim == 2) ? lex_map_2 : lex_map_3;
|
||||
AV[offset] = V(0, iloc, e);
|
||||
J[offset] = i;
|
||||
++offset;
|
||||
for (int n_idx = 0; n_idx < dim; ++n_idx)
|
||||
{
|
||||
// qi is the face lexicographic index, obtained by taking the
|
||||
// lexicographic index of the coordinates ommiting n_idx.
|
||||
int qi = 0;
|
||||
int stride = 1;
|
||||
for (int d = 0; d < dim; ++d)
|
||||
{
|
||||
if (d != n_idx)
|
||||
{
|
||||
qi += local_i[d]*stride;
|
||||
stride *= pp1;
|
||||
}
|
||||
}
|
||||
for (int e_i = 0; e_i < 2; ++e_i)
|
||||
{
|
||||
const int j_lex = e_i + n_idx*2;
|
||||
const int f = lex_map[j_lex];
|
||||
const bool bdr = (local_i[n_idx] == e_i * p);
|
||||
if (bdr)
|
||||
{
|
||||
const int nbr_e = d_nbr_info(e, f, 0);
|
||||
const int nbr_ori = d_nbr_info(e, f, 1);
|
||||
const int nbr_f = d_nbr_info(e, f, 2);
|
||||
if (nbr_e != -1)
|
||||
{
|
||||
const int nbr_loc_idx = internal::FaceIdxToVolIdx(
|
||||
dim, qi, pp1, f, nbr_f, 1, nbr_ori);
|
||||
J[offset] = nbr_e*ndof_per_el + nbr_loc_idx;
|
||||
AV[offset] = V(f+1, iloc, e);
|
||||
++offset;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
int shift = (e_i == 0) ? -1 : 1;
|
||||
for (int n = 0; n < n_idx; ++n) { shift *= pp1; }
|
||||
J[offset] = i + shift;
|
||||
AV[offset] = V(f+1, iloc, e);
|
||||
++offset;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void BatchedLORAssembly::SparseIJToCSR(OperatorHandle &A) const
|
||||
{
|
||||
const int nvdof = fes_ho.GetVSize();
|
||||
@@ -372,12 +550,11 @@ void BatchedLORAssembly::SparseIJToCSR(OperatorHandle &A) const
|
||||
}
|
||||
|
||||
A_mat->OverrideSize(nvdof, nvdof);
|
||||
EnsureCapacity(A_mat->GetMemoryI(), nvdof + 1);
|
||||
|
||||
A_mat->GetMemoryI().New(nvdof+1, Device::GetDeviceMemoryType());
|
||||
int nnz = FillI(*A_mat);
|
||||
|
||||
A_mat->GetMemoryJ().New(nnz, Device::GetDeviceMemoryType());
|
||||
A_mat->GetMemoryData().New(nnz, Device::GetDeviceMemoryType());
|
||||
const int nnz = FillI(*A_mat);
|
||||
EnsureCapacity(A_mat->GetMemoryJ(), nnz);
|
||||
EnsureCapacity(A_mat->GetMemoryData(), nnz);
|
||||
FillJAndData(*A_mat);
|
||||
}
|
||||
|
||||
@@ -431,6 +608,19 @@ void BatchedLORAssembly::AssembleWithoutBC(BilinearForm &a, OperatorHandle &A)
|
||||
// Assemble the matrix, depending on what the form is.
|
||||
// This fills in the arrays sparse_ij and sparse_mapping.
|
||||
const FiniteElementCollection *fec = fes_ho.FEColl();
|
||||
|
||||
// Handle DG case separately, because assembly of CSR matrix requires
|
||||
// handling face terms.
|
||||
if (dynamic_cast<const DG_FECollection*>(fec))
|
||||
{
|
||||
if (HasIntegrators<DiffusionIntegrator, MassIntegrator>(a))
|
||||
{
|
||||
AssemblyKernel<BatchedLOR_DG>(a);
|
||||
}
|
||||
SparseIJToCSR_DG(A);
|
||||
return;
|
||||
}
|
||||
|
||||
if (dynamic_cast<const H1_FECollection*>(fec))
|
||||
{
|
||||
if (HasIntegrators<DiffusionIntegrator, MassIntegrator>(a))
|
||||
@@ -453,10 +643,47 @@ void BatchedLORAssembly::AssembleWithoutBC(BilinearForm &a, OperatorHandle &A)
|
||||
}
|
||||
}
|
||||
|
||||
return SparseIJToCSR(A);
|
||||
SparseIJToCSR(A);
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
void BatchedLORAssembly::ParAssemble_DG(SparseMatrix &A_local,
|
||||
OperatorHandle &A)
|
||||
{
|
||||
auto &par_fes = static_cast<ParFiniteElementSpace&>(fes_ho);
|
||||
|
||||
// handle the case when 'a' contains off-diagonal
|
||||
const int lvsize = par_fes.GetVSize();
|
||||
const Array<HYPRE_BigInt> &face_nbr_glob_ldof =
|
||||
par_fes.GetFaceNbrGlobalDofMapArray();
|
||||
const HYPRE_BigInt ldof_offset = par_fes.GetMyDofOffset();
|
||||
|
||||
const int nnz_local = A_local.NumNonZeroElems();
|
||||
Array<HYPRE_BigInt> glob_J(nnz_local);
|
||||
|
||||
const HYPRE_BigInt *d_face_nbr_glob_ldof = face_nbr_glob_ldof.Read();
|
||||
const int *d_J = A_local.ReadJ();
|
||||
HYPRE_BigInt *d_glob_J = glob_J.Write();
|
||||
|
||||
mfem::forall(nnz_local, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
if (d_J[i] < lvsize)
|
||||
{
|
||||
d_glob_J[i] = d_J[i] + ldof_offset;
|
||||
}
|
||||
else
|
||||
{
|
||||
d_glob_J[i] = d_face_nbr_glob_ldof[d_J[i] - lvsize];
|
||||
}
|
||||
});
|
||||
|
||||
A.Reset(new HypreParMatrix(
|
||||
par_fes.GetComm(), lvsize, par_fes.GlobalVSize(),
|
||||
par_fes.GlobalVSize(), A_local.HostReadWriteI(),
|
||||
glob_J.HostReadWrite(), A_local.HostReadWriteData(),
|
||||
par_fes.GetDofOffsets(), par_fes.GetDofOffsets()));
|
||||
}
|
||||
|
||||
void BatchedLORAssembly::ParAssemble(
|
||||
BilinearForm &a, const Array<int> &ess_dofs, OperatorHandle &A)
|
||||
{
|
||||
@@ -464,13 +691,18 @@ void BatchedLORAssembly::ParAssemble(
|
||||
OperatorHandle A_local;
|
||||
AssembleWithoutBC(a, A_local);
|
||||
|
||||
ParBilinearForm *pa =
|
||||
dynamic_cast<ParBilinearForm*>(&a);
|
||||
|
||||
pa->ParallelRAP(*A_local.As<SparseMatrix>(), A, true);
|
||||
|
||||
A.As<HypreParMatrix>()->EliminateBC(ess_dofs,
|
||||
Operator::DiagonalPolicy::DIAG_ONE);
|
||||
if (dynamic_cast<const DG_FECollection*>(fes_ho.FEColl()))
|
||||
{
|
||||
ParAssemble_DG(*A_local.As<SparseMatrix>(), A);
|
||||
}
|
||||
else
|
||||
{
|
||||
ParBilinearForm *pa =
|
||||
dynamic_cast<ParBilinearForm*>(&a);
|
||||
pa->ParallelRAP(*A_local.As<SparseMatrix>(), A, true);
|
||||
A.As<HypreParMatrix>()->EliminateBC(ess_dofs,
|
||||
Operator::DiagonalPolicy::DIAG_ONE);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -504,12 +736,22 @@ BatchedLORAssembly::BatchedLORAssembly(FiniteElementSpace &fes_ho_)
|
||||
FormLORVertexCoordinates(fes_ho, X_vert);
|
||||
}
|
||||
|
||||
IntegrationRule GetCollocatedIntRule(FiniteElementSpace &fes)
|
||||
IntegrationRule GetLobattoIntRule(Geometry::Type geom, int nd1d)
|
||||
{
|
||||
IntegrationRules irs(0, Quadrature1D::GaussLobatto);
|
||||
const Geometry::Type geom = fes.GetMesh()->GetTypicalElementGeometry();
|
||||
const int nd1d = fes.GetMaxElementOrder() + 1;
|
||||
return irs.Get(geom, 2*nd1d - 3);
|
||||
}
|
||||
|
||||
IntegrationRule GetCollocatedIntRule(FiniteElementSpace &fes)
|
||||
{
|
||||
const Geometry::Type geom = fes.GetMesh()->GetTypicalElementGeometry();
|
||||
return GetLobattoIntRule(geom, fes.GetMaxElementOrder() + 1);
|
||||
}
|
||||
|
||||
IntegrationRule GetCollocatedFaceIntRule(FiniteElementSpace &fes)
|
||||
{
|
||||
const Geometry::Type geom = fes.GetMesh()->GetTypicalFaceGeometry();
|
||||
return GetLobattoIntRule(geom, fes.GetMaxElementOrder() + 1);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
+32
-2
@@ -25,6 +25,7 @@ namespace mfem
|
||||
/// supported, currently:
|
||||
///
|
||||
/// - H1 diffusion + mass
|
||||
/// - DG diffusion + mass (in progress)
|
||||
/// - ND curl-curl + mass
|
||||
/// - RT div-div + mass
|
||||
///
|
||||
@@ -73,6 +74,9 @@ public:
|
||||
/// Return the vertices of the LOR mesh in E-vector format
|
||||
const Vector &GetLORVertexCoordinates() { return X_vert; }
|
||||
|
||||
/// Specialized implementation of SparseIJToCSR for DG spaces.
|
||||
void SparseIJToCSR_DG(OperatorHandle &A) const;
|
||||
|
||||
protected:
|
||||
/// After assembling the "sparse IJ" format, convert it to CSR.
|
||||
void SparseIJToCSR(OperatorHandle &A) const;
|
||||
@@ -105,6 +109,9 @@ public:
|
||||
void FillJAndData(SparseMatrix &A) const;
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// Assemble the parallel DG matrix (with shared faces).
|
||||
void ParAssemble_DG(SparseMatrix &A_local, OperatorHandle &A);
|
||||
|
||||
/// Assemble the system in parallel and place the result in @a A.
|
||||
void ParAssemble(BilinearForm &a, const Array<int> &ess_dofs,
|
||||
OperatorHandle &A);
|
||||
@@ -128,9 +135,8 @@ void EnsureCapacity(Memory<T> &mem, int capacity)
|
||||
|
||||
/// Return the first domain integrator in the form @a i of type @a T.
|
||||
template <typename T>
|
||||
static T *GetIntegrator(BilinearForm &a)
|
||||
static T *GetIntegrator(Array<BilinearFormIntegrator*> *integs)
|
||||
{
|
||||
Array<BilinearFormIntegrator*> *integs = a.GetDBFI();
|
||||
if (integs != NULL)
|
||||
{
|
||||
for (auto *i : *integs)
|
||||
@@ -144,8 +150,32 @@ static T *GetIntegrator(BilinearForm &a)
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static T *GetIntegrator(BilinearForm &a)
|
||||
{
|
||||
return GetIntegrator<T>(a.GetDBFI());
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static T *GetInteriorFaceIntegrator(BilinearForm &a)
|
||||
{
|
||||
return GetIntegrator<T>(a.GetFBFI());
|
||||
}
|
||||
|
||||
/// @brief Return the Gauss-Lobatto rule for geometry @a geom with @a nd1d
|
||||
/// points per dimension.
|
||||
IntegrationRule GetLobattoIntRule(Geometry::Type geom, int nd1d);
|
||||
|
||||
/// @brief Return the Gauss-Lobatto rule collocated with the element nodes.
|
||||
///
|
||||
/// Assumes @a fes uses Gauss-Lobatto basis.
|
||||
IntegrationRule GetCollocatedIntRule(FiniteElementSpace &fes);
|
||||
|
||||
/// @brief Return the Gauss-Lobatto rule collocated with face nodes.
|
||||
///
|
||||
/// Assumes @a fes uses Gauss-Lobatto basis.
|
||||
IntegrationRule GetCollocatedFaceIntRule(FiniteElementSpace &fes);
|
||||
|
||||
template <typename INTEGRATOR>
|
||||
void ProjectLORCoefficient(BilinearForm &a, CoefficientVector &coeff_vector)
|
||||
{
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_LOR_DG
|
||||
#define MFEM_LOR_DG
|
||||
|
||||
#include "lor_batched.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// BatchedLORKernel specialization for DG spaces. Not user facing. See the
|
||||
// classes BatchedLORAssembly and BatchedLORKernel .
|
||||
class BatchedLOR_DG : BatchedLORKernel
|
||||
{
|
||||
IntegrationRule ir_face; ///< Collocated Gauss-Lobatto face quadrature rule.
|
||||
real_t kappa; ///< DG penalty parameter.
|
||||
public:
|
||||
template <int ORDER, int SDIM> void Assemble2D();
|
||||
template <int ORDER> void Assemble3D();
|
||||
BatchedLOR_DG(BilinearForm &a,
|
||||
FiniteElementSpace &fes_ho_,
|
||||
Vector &X_vert_,
|
||||
Vector &sparse_ij_,
|
||||
Array<int> &sparse_mapping_)
|
||||
: BatchedLORKernel(fes_ho_, X_vert_, sparse_ij_, sparse_mapping_),
|
||||
ir_face(GetLobattoIntRule(fes_ho_.GetMesh()->GetTypicalFaceGeometry(),
|
||||
fes_ho_.GetMaxElementOrder() + 1))
|
||||
{
|
||||
ProjectLORCoefficient<MassIntegrator>(a, c1);
|
||||
ProjectLORCoefficient<DiffusionIntegrator>(a, c2);
|
||||
|
||||
auto *integ = GetInteriorFaceIntegrator<DGDiffusionIntegrator>(a);
|
||||
if (integ)
|
||||
{
|
||||
kappa = integ->GetPenaltyParameter();
|
||||
}
|
||||
else
|
||||
{
|
||||
kappa = 0.0;
|
||||
}
|
||||
}
|
||||
|
||||
/// @brief Compute and return the face info array.
|
||||
///
|
||||
/// The face info array has shape (6, nf), where @a nf is the number of
|
||||
/// faces. For each face @a i, the column (:,i) has entries (e0, f0, o0, e1,
|
||||
/// f1, o1), where @a e is adjacent element, @a f is the local face index,
|
||||
/// and @a o is the orientation. For boundary and shared faces, (e1, f1, o1)
|
||||
/// are all set to -1.
|
||||
Array<int> GetFaceInfo() const;
|
||||
|
||||
/// @brief Compute and return the boundary penalty factor.
|
||||
///
|
||||
/// The returned vector has shape (nq, nf), where @a nq is the number of
|
||||
/// nodes per face, and @a nf is the number of faces.
|
||||
///
|
||||
/// The boundary penalty factor is $J_f / h = J_f^2 / J_e$ (since $h = J_e /
|
||||
/// J_f$), where $J_f$ is the face Jacobian determinant, and $J_e$ is the
|
||||
/// element Jacobian determinant.
|
||||
Vector GetBdrPenaltyFactor() const;
|
||||
|
||||
/// Assemble the face penalty terms in the matrix @a sparse_ij.
|
||||
void AssembleFaceTerms();
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
#include "lor_dg_impl.hpp"
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,392 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "lor_dg.hpp"
|
||||
#include "../fe/face_map_utils.hpp"
|
||||
#include "../../linalg/dtensor.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
Array<int> BatchedLOR_DG::GetFaceInfo() const
|
||||
{
|
||||
Mesh &mesh = *fes_ho.GetMesh();
|
||||
const int nf = mesh.GetNumFaces();
|
||||
Array<int> face_info(nf * 6); // (e0, f0, o0, e1, f1, o1)
|
||||
auto h_face_info = Reshape(face_info.HostWrite(), 6, nf);
|
||||
for (int f = 0; f < nf; ++f)
|
||||
{
|
||||
auto finfo = mesh.GetFaceInformation(f);
|
||||
h_face_info(0, f) = finfo.element[0].index;
|
||||
h_face_info(1, f) = finfo.element[0].local_face_id;
|
||||
h_face_info(2, f) = finfo.element[0].orientation;
|
||||
if (finfo.IsLocal()) // Interior, non-shared face
|
||||
{
|
||||
h_face_info(3, f) = finfo.element[1].index;
|
||||
h_face_info(4, f) = finfo.element[1].local_face_id;
|
||||
h_face_info(5, f) = finfo.element[1].orientation;
|
||||
}
|
||||
else
|
||||
{
|
||||
h_face_info(3, f) = -1;
|
||||
h_face_info(4, f) = -1;
|
||||
h_face_info(5, f) = -1;
|
||||
}
|
||||
}
|
||||
return face_info;
|
||||
}
|
||||
|
||||
Vector BatchedLOR_DG::GetBdrPenaltyFactor() const
|
||||
{
|
||||
Mesh &mesh = *fes_ho.GetMesh();
|
||||
|
||||
const int nf = mesh.GetNumFaces();
|
||||
Array<int> f_int(mesh.GetNFbyType(FaceType::Interior));
|
||||
Array<int> f_bdr(mesh.GetNFbyType(FaceType::Boundary));
|
||||
{
|
||||
int i_int = 0;
|
||||
int i_bdr = 0;
|
||||
for (int i = 0; i < nf; ++i)
|
||||
{
|
||||
const auto f = mesh.GetFaceInformation(i);
|
||||
if (f.IsBoundary())
|
||||
{
|
||||
f_bdr[i_bdr] = i;
|
||||
++i_bdr;
|
||||
}
|
||||
else if (f.IsInterior())
|
||||
{
|
||||
f_int[i_int] = i;
|
||||
++i_int;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const auto geom = fes_ho.GetMesh()->GetGeometricFactors(
|
||||
ir, GeometricFactors::DETERMINANTS);
|
||||
|
||||
const int nq = ir_face.Size();
|
||||
Vector face_Jh(nq * nf);
|
||||
for (const FaceType ft : {FaceType::Interior, FaceType::Boundary})
|
||||
{
|
||||
const int nft = mesh.GetNFbyType(ft);
|
||||
auto *geom_face = mesh.GetFaceGeometricFactors(
|
||||
ir_face, FaceGeometricFactors::DETERMINANTS, ft);
|
||||
|
||||
const L2FaceValues fv = (ft == FaceType::Interior)
|
||||
? L2FaceValues::DoubleValued
|
||||
: L2FaceValues::SingleValued;
|
||||
const int m = (fv == L2FaceValues::DoubleValued) ? 2 : 1;
|
||||
|
||||
auto *r = fes_ho.GetFaceRestriction(ElementDofOrdering::LEXICOGRAPHIC, ft, fv);
|
||||
Vector detJ_r(nq * m * nft);
|
||||
r->Mult(geom->detJ, detJ_r);
|
||||
|
||||
const auto *d_i = (ft == FaceType::Interior) ? f_int.Read() : f_bdr.Read();
|
||||
const auto d_detJ_face = Reshape(geom_face->detJ.Read(), nq, nft);
|
||||
const auto d_detJ_r = Reshape(detJ_r.Read(), nq, m, nft);
|
||||
auto d_face_Jh = Reshape(face_Jh.Write(), nq, nf);
|
||||
|
||||
mfem::forall(nft * nq, [=] MFEM_HOST_DEVICE (int ii)
|
||||
{
|
||||
const int i = ii % nq;
|
||||
const int f = ii / nq;
|
||||
const real_t J_el = 0.5*(d_detJ_r(i, 0, f) + d_detJ_r(i, m==2?1:0, f));
|
||||
const real_t J_f = d_detJ_face(i, f);
|
||||
d_face_Jh(i, d_i[f]) = J_f * J_f / J_el;
|
||||
});
|
||||
}
|
||||
return face_Jh;
|
||||
}
|
||||
|
||||
void BatchedLOR_DG::AssembleFaceTerms()
|
||||
{
|
||||
Mesh &mesh = *fes_ho.GetMesh();
|
||||
|
||||
const int nnz_per_row = 1 + mesh.Dimension()*2;
|
||||
const int pp1 = fes_ho.GetMaxElementOrder() + 1;
|
||||
const int nel_ho = mesh.GetNE();
|
||||
const int nf = mesh.GetNumFaces();
|
||||
const int nd_face = ir_face.Size();
|
||||
const int nd = ir.Size();
|
||||
const int dim = mesh.Dimension();
|
||||
|
||||
Array<int> face_info = GetFaceInfo();
|
||||
const auto d_face_info = Reshape(face_info.Read(), 6, nf);
|
||||
|
||||
Vector face_Jh = GetBdrPenaltyFactor();
|
||||
const auto d_face_Jh = Reshape(face_Jh.Read(), nd_face, nf);
|
||||
|
||||
const auto *w_face = ir_face.GetWeights().Read();
|
||||
|
||||
// Penalty parameter (avoid capturing *this in lambda)
|
||||
const real_t d_kappa = kappa;
|
||||
|
||||
// Get diffusion coefficient
|
||||
const bool const_dq = c2.Size() == 1;
|
||||
const auto DQ = const_dq?Reshape(c2.Read(),1,1):Reshape(c2.Read(),nd,nel_ho);
|
||||
|
||||
// Sparse matrix entries
|
||||
auto V = Reshape(sparse_ij.ReadWrite(), nnz_per_row, nd, nel_ho);
|
||||
|
||||
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int f)
|
||||
{
|
||||
const int f_0 = d_face_info(1, f);
|
||||
const int f_1 = d_face_info(4, f);
|
||||
const int nsides = (f_1 >= 0) ? 2 : 1;
|
||||
for (int el_i = 0; el_i < nsides; ++el_i)
|
||||
{
|
||||
const int e = d_face_info(3*el_i, f);
|
||||
const int o = d_face_info(3*el_i + 2, f);
|
||||
const int v_idx = 1 + ((el_i == 0) ? f_0 : f_1);
|
||||
for (int i = 0; i < nd_face; ++i)
|
||||
{
|
||||
const int ii = internal::FaceIdxToVolIdx(dim, i, pp1, f_0, f_1, el_i, o);
|
||||
const real_t Jh = d_face_Jh(i, f);
|
||||
const real_t dq = const_dq ? DQ(0,0) : DQ(ii, e);
|
||||
V(v_idx, ii, e) = -dq*d_kappa*Jh*w_face[i];
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <int ORDER, int SDIM>
|
||||
void BatchedLOR_DG::Assemble2D()
|
||||
{
|
||||
MFEM_VERIFY(SDIM == 2, "Surface meshes not currently supported for LOR-DG.")
|
||||
|
||||
static constexpr int pp1 = ORDER + 1;
|
||||
static constexpr int ndof_per_el = pp1*pp1;
|
||||
static constexpr int nnz_per_row = 5;
|
||||
const int nel_ho = fes_ho.GetNE();
|
||||
|
||||
// Get element geometric factors; calling before AssembleFaceTerms, since
|
||||
// in AssembleFaceTerms, element Jacobian determinants are used, potentially
|
||||
// saving recomputation.
|
||||
const auto factors = GeometricFactors::DETERMINANTS |
|
||||
GeometricFactors::JACOBIANS;
|
||||
const auto *geom = fes_ho.GetMesh()->GetGeometricFactors(ir, factors);
|
||||
|
||||
// Sparse matrix entries
|
||||
sparse_ij.SetSize(nnz_per_row*ndof_per_el*nel_ho);
|
||||
sparse_ij.UseDevice(true);
|
||||
sparse_ij = 0.0;
|
||||
auto V = Reshape(sparse_ij.ReadWrite(), nnz_per_row, pp1, pp1, nel_ho);
|
||||
|
||||
AssembleFaceTerms();
|
||||
|
||||
// Populate Gauss-Lobatto quadrature rule of size (p+1)
|
||||
IntegrationRule ir_pp1;
|
||||
QuadratureFunctions1D::GaussLobatto(pp1, &ir_pp1);
|
||||
Vector glx_pp1(pp1), glw_pp1(pp1);
|
||||
for (int i = 0; i < pp1; ++i)
|
||||
{
|
||||
glx_pp1[i] = ir_pp1[i].x;
|
||||
glw_pp1[i] = ir_pp1[i].weight;
|
||||
}
|
||||
const auto *x_pp1 = glx_pp1.Read();
|
||||
const auto *w_1d = glw_pp1.Read();
|
||||
|
||||
// Get coefficients for mass and diffusion
|
||||
const bool const_mq = c1.Size() == 1;
|
||||
const auto MQ = const_mq
|
||||
? Reshape(c1.Read(), 1, 1, 1)
|
||||
: Reshape(c1.Read(), pp1, pp1, nel_ho);
|
||||
const bool const_dq = c2.Size() == 1;
|
||||
const auto DQ = const_dq
|
||||
? Reshape(c2.Read(), 1, 1, 1)
|
||||
: Reshape(c2.Read(), pp1, pp1, nel_ho);
|
||||
|
||||
const auto detJ = Reshape(geom->detJ.Read(), pp1, pp1, nel_ho);
|
||||
const auto J = Reshape(geom->J.Read(), pp1, pp1, 2, 2, nel_ho);
|
||||
const auto W = Reshape(ir.GetWeights().Read(), pp1, pp1);
|
||||
|
||||
mfem::forall(nel_ho, [=] MFEM_HOST_DEVICE (int iel_ho)
|
||||
{
|
||||
for (int iy = 0; iy < pp1; ++iy)
|
||||
{
|
||||
for (int ix = 0; ix < pp1; ++ix)
|
||||
{
|
||||
const real_t mq = const_mq ? MQ(0,0,0) : MQ(ix, iy, iel_ho);
|
||||
const real_t dq = const_dq ? DQ(0,0,0) : DQ(ix, iy, iel_ho);
|
||||
|
||||
for (int n_idx = 0; n_idx < 2; ++n_idx)
|
||||
{
|
||||
for (int e_i = 0; e_i < 2; ++e_i)
|
||||
{
|
||||
const int i_0 = (n_idx == 0) ? ix + e_i : ix;
|
||||
const int j_0 = (n_idx == 1) ? iy + e_i : iy;
|
||||
|
||||
const bool bdr = (n_idx == 0 && (i_0 == 0 || i_0 == pp1)) ||
|
||||
(n_idx == 1 && (j_0 == 0 || j_0 == pp1));
|
||||
|
||||
if (bdr) { continue; }
|
||||
|
||||
static constexpr int lex_map[] = {4, 2, 1, 3};
|
||||
const int v_idx_lex = e_i + n_idx*2;
|
||||
const int v_idx = lex_map[v_idx_lex];
|
||||
|
||||
const int w_idx = (n_idx == 0) ? iy : ix;
|
||||
const int x_idx = (n_idx == 0) ? i_0 : j_0;
|
||||
|
||||
const real_t J1 = J(ix, iy, n_idx, !n_idx, iel_ho);
|
||||
const real_t J2 = J(ix, iy, !n_idx, !n_idx, iel_ho);
|
||||
const real_t Jh = (J1*J1 + J2*J2) / detJ(ix, iy, iel_ho);
|
||||
|
||||
V(v_idx, ix, iy, iel_ho) =
|
||||
-dq * Jh * w_1d[w_idx] / (x_pp1[x_idx] - x_pp1[x_idx -1]);
|
||||
}
|
||||
}
|
||||
V(0, ix, iy, iel_ho) = mq * detJ(ix, iy, iel_ho) * W(ix, iy);
|
||||
for (int i = 1; i < nnz_per_row; ++i)
|
||||
{
|
||||
V(0, ix, iy, iel_ho) -= V(i, ix, iy, iel_ho);
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <int ORDER>
|
||||
void BatchedLOR_DG::Assemble3D()
|
||||
{
|
||||
static constexpr int pp1 = ORDER + 1;
|
||||
static constexpr int ndof_per_el = pp1*pp1*pp1;
|
||||
static constexpr int nnz_per_row = 7;
|
||||
const int nel_ho = fes_ho.GetNE();
|
||||
|
||||
// Get element geometric factors; calling before AssembleFaceTerms, since
|
||||
// in AssembleFaceTerms, element Jacobian determinants are used, potentially
|
||||
// saving recomputation.
|
||||
const auto factors = GeometricFactors::DETERMINANTS |
|
||||
GeometricFactors::JACOBIANS;
|
||||
const auto geom = fes_ho.GetMesh()->GetGeometricFactors(ir, factors);
|
||||
|
||||
sparse_ij.SetSize(nnz_per_row*ndof_per_el*nel_ho);
|
||||
sparse_ij.UseDevice(true);
|
||||
sparse_ij = 0.0;
|
||||
auto V = Reshape(sparse_ij.Write(), nnz_per_row, pp1, pp1, pp1, nel_ho);
|
||||
|
||||
AssembleFaceTerms();
|
||||
|
||||
// Populate Gauss-Lobatto quadrature rule of size (p+1)
|
||||
IntegrationRule ir_pp1;
|
||||
QuadratureFunctions1D::GaussLobatto(pp1, &ir_pp1);
|
||||
Vector glx_pp1(pp1), glw_pp1(pp1);
|
||||
for (int i = 0; i < pp1; ++i)
|
||||
{
|
||||
glx_pp1[i] = ir_pp1[i].x;
|
||||
glw_pp1[i] = ir_pp1[i].weight;
|
||||
}
|
||||
const auto *x_pp1 = glx_pp1.Read();
|
||||
const auto *w_1d = glw_pp1.Read();
|
||||
|
||||
const bool const_mq = c1.Size() == 1;
|
||||
const auto MQ = const_mq
|
||||
? Reshape(c1.Read(), 1, 1, 1, 1)
|
||||
: Reshape(c1.Read(), pp1, pp1, pp1, nel_ho);
|
||||
const bool const_dq = c2.Size() == 1;
|
||||
const auto DQ = const_dq
|
||||
? Reshape(c2.Read(), 1, 1, 1, 1)
|
||||
: Reshape(c2.Read(), pp1, pp1, pp1, nel_ho);
|
||||
const auto W = Reshape(ir.GetWeights().Read(), pp1, pp1, pp1);
|
||||
|
||||
const auto detJ = Reshape(geom->detJ.Read(), pp1, pp1, pp1, nel_ho);
|
||||
const auto J = Reshape(geom->J.Read(), pp1, pp1, pp1, 3, 3, nel_ho);
|
||||
|
||||
mfem::forall(nel_ho, [=] MFEM_HOST_DEVICE (int iel_ho)
|
||||
{
|
||||
for (int iz = 0; iz < pp1; ++iz)
|
||||
{
|
||||
for (int iy = 0; iy < pp1; ++iy)
|
||||
{
|
||||
for (int ix = 0; ix < pp1; ++ix)
|
||||
{
|
||||
const real_t mq = const_mq ? MQ(0,0,0,0) : MQ(ix, iy, iz, iel_ho);
|
||||
const real_t dq = const_dq ? DQ(0,0,0,0) : DQ(ix, iy, iz, iel_ho);
|
||||
|
||||
const real_t DETJ = detJ(ix, iy, iz, iel_ho);
|
||||
|
||||
for (int n_idx = 0; n_idx < 3; ++n_idx)
|
||||
{
|
||||
for (int e_i = 0; e_i < 2; ++e_i)
|
||||
{
|
||||
static constexpr int lex_map[] = {5,3,2,4,1,6};
|
||||
const int v_idx_lex = e_i + n_idx*2;
|
||||
const int v_idx = lex_map[v_idx_lex];
|
||||
|
||||
const int i_0 = (n_idx == 0) ? ix + e_i : ix;
|
||||
const int j_0 = (n_idx == 1) ? iy + e_i : iy;
|
||||
const int k_0 = (n_idx == 2) ? iz + e_i : iz;
|
||||
|
||||
const bool bdr =
|
||||
(n_idx == 0 && (i_0 == 0 || i_0 == pp1)) ||
|
||||
(n_idx == 1 && (j_0 == 0 || j_0 == pp1)) ||
|
||||
(n_idx == 2 && (k_0 == 0 || k_0 == pp1));
|
||||
|
||||
if (bdr) { continue; }
|
||||
|
||||
int x_idx = (n_idx == 0) ? i_0 : (n_idx == 1) ? j_0 : k_0;
|
||||
int w_idx_1 = (n_idx == 0) ? iy : (n_idx == 1) ? iz : ix;
|
||||
int w_idx_2 = (n_idx == 0) ? iz : (n_idx == 1) ? ix : iy;
|
||||
|
||||
const real_t J00 = J(ix, iy, iz, 0, 0, iel_ho);
|
||||
const real_t J01 = J(ix, iy, iz, 0, 1, iel_ho);
|
||||
const real_t J02 = J(ix, iy, iz, 0, 2, iel_ho);
|
||||
const real_t J10 = J(ix, iy, iz, 1, 0, iel_ho);
|
||||
const real_t J11 = J(ix, iy, iz, 1, 1, iel_ho);
|
||||
const real_t J12 = J(ix, iy, iz, 1, 2, iel_ho);
|
||||
const real_t J20 = J(ix, iy, iz, 2, 0, iel_ho);
|
||||
const real_t J21 = J(ix, iy, iz, 2, 1, iel_ho);
|
||||
const real_t J22 = J(ix, iy, iz, 2, 2, iel_ho);
|
||||
|
||||
real_t JinvJinvT_diag = 0.0;
|
||||
if (n_idx == 0)
|
||||
{
|
||||
JinvJinvT_diag = J02*J02*(J11*J11 + J21*J21) + (J12*J21 - J11*J22)*
|
||||
(J12*J21 - J11*J22) - 2*J01*J02*(J11*J12 + J21*J22) + J01*J01*
|
||||
(J12*J12 + J22*J22);
|
||||
}
|
||||
else if (n_idx == 1)
|
||||
{
|
||||
JinvJinvT_diag = J02*J02*(J10*J10 + J20*J20) + (J12*J20 - J10*J22)*
|
||||
(J12*J20 - J10*J22) - 2*J00*J02*(J10*J12 + J20*J22) + J00*J00*
|
||||
(J12*J12 + J22*J22);
|
||||
}
|
||||
else if (n_idx == 2)
|
||||
{
|
||||
JinvJinvT_diag = J01*J01*(J10*J10 + J20*J20) + (J11*J20 - J10*J21)*
|
||||
(J11*J20 - J10*J21) - 2*J00*J01*(J10*J11 + J20*J21) + J00*J00*
|
||||
(J11*J11 + J21*J21);
|
||||
}
|
||||
|
||||
const real_t Jh = JinvJinvT_diag / DETJ;
|
||||
|
||||
V(v_idx, ix, iy, iz, iel_ho) = -dq * Jh * w_1d[w_idx_1] * w_1d[w_idx_2] /
|
||||
(x_pp1[x_idx] - x_pp1[x_idx -1]);
|
||||
}
|
||||
}
|
||||
V(0, ix, iy, iz, iel_ho) = mq * DETJ * W(ix, iy, iz);
|
||||
for (int i = 1; i < 7; ++i)
|
||||
{
|
||||
V(0, ix, iy, iz, iel_ho) -= V(i, ix, iy, iz, iel_ho);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
+14
-3
@@ -788,12 +788,10 @@ BlockNonlinearForm::BlockNonlinearForm(Array<FiniteElementSpace *> &f) :
|
||||
}
|
||||
|
||||
void BlockNonlinearForm::SetEssentialBC(
|
||||
const Array<Array<int> *> &bdr_attr_is_ess, Array<Vector *> &rhs)
|
||||
const Array<Array<int>*> &bdr_attr_is_ess, Array<Vector*> &rhs)
|
||||
{
|
||||
for (int s = 0; s < fes.Size(); ++s)
|
||||
{
|
||||
ess_tdofs[s]->SetSize(ess_tdofs.Size());
|
||||
|
||||
fes[s]->GetEssentialTrueDofs(*bdr_attr_is_ess[s], *ess_tdofs[s]);
|
||||
|
||||
if (rhs[s])
|
||||
@@ -803,6 +801,19 @@ void BlockNonlinearForm::SetEssentialBC(
|
||||
}
|
||||
}
|
||||
|
||||
void BlockNonlinearForm::SetEssentialTrueDofs(
|
||||
const Array<Array<int>*> &ess_tdof_list, Array<Vector*> &rhs)
|
||||
{
|
||||
for (int s = 0; s < fes.Size(); ++s)
|
||||
{
|
||||
*ess_tdofs[s] = *ess_tdof_list[s];
|
||||
if (rhs[s])
|
||||
{
|
||||
rhs[s]->SetSubVector(*ess_tdofs[s], 0.0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
real_t BlockNonlinearForm::GetEnergyBlocked(const BlockVector &bx) const
|
||||
{
|
||||
Array<Array<int> *> vdofs(fes.Size());
|
||||
|
||||
+33
-2
@@ -363,8 +363,39 @@ public:
|
||||
Array<int> &bdr_marker)
|
||||
{ bfnfi.Append(nlfi); bfnfi_marker.Append(&bdr_marker); }
|
||||
|
||||
virtual void SetEssentialBC(const Array<Array<int> *>&bdr_attr_is_ess,
|
||||
Array<Vector *> &rhs);
|
||||
/** @brief Set essential boundary conditions to each finite element space
|
||||
using boundary attribute markers.
|
||||
|
||||
This method calls `FiniteElementSpace::GetEssentialTrueDofs()` for each
|
||||
space and stores ess_tdof_lists internally.
|
||||
|
||||
If `rhs` vectors are non-null, the entries corresponding to these
|
||||
essential DoFs are set to zero. This ensures compatibility with the
|
||||
output of the `Mult()` method, which also zeroes out these entries.
|
||||
|
||||
@param[in] bdr_attr_is_ess A list of boundary attribute markers for each
|
||||
space.
|
||||
@param[in,out] rhs An array of optional right-hand side vectors.
|
||||
If a vector at `rhs[i]` is non-null, its essential DoFs will be set
|
||||
to zero. */
|
||||
virtual void SetEssentialBC(const Array<Array<int>*> &bdr_attr_is_ess,
|
||||
Array<Vector*> &rhs);
|
||||
|
||||
/** @brief Set essential boundary conditions to each finite element space
|
||||
using essential true dof lists.
|
||||
|
||||
This method stores a copy of the provided essential true dof lists.
|
||||
|
||||
If `rhs` vectors are non-null, the entries corresponding to these
|
||||
essential DoFs are set to zero. This ensures compatibility with the
|
||||
output of the `Mult()` method, which also zeroes out these entries.
|
||||
|
||||
@param[in] ess_tdof_list A list of essential true dofs for each space.
|
||||
@param[in,out] rhs An array of optional right-hand side vectors.
|
||||
If a vector at `rhs[i]` is non-null, its essential DoFs will be set
|
||||
to zero. */
|
||||
virtual void SetEssentialTrueDofs(const Array<Array<int>*> &ess_tdof_list,
|
||||
Array<Vector*> &rhs);
|
||||
|
||||
virtual real_t GetEnergy(const Vector &x) const;
|
||||
|
||||
|
||||
@@ -332,7 +332,7 @@ ParDerefineMatrixOp::ParDerefineMatrixOp(ParFiniteElementSpace &fespace_,
|
||||
pack_col_idcs.SetSize(send_len);
|
||||
// memory manager doesn't appear to have a graceful fallback for
|
||||
// HOST_PINNED if not built with CUDA or HIP
|
||||
#if defined(MFEM_USE_CUDA) or defined(MFEM_USE_HIP)
|
||||
#if defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP)
|
||||
xghost_send.SetSize(send_len * fespace->GetVDim(),
|
||||
Device::GetGPUAwareMPI() ? MemoryType::DEFAULT
|
||||
: MemoryType::HOST_PINNED);
|
||||
|
||||
@@ -481,6 +481,7 @@ public:
|
||||
that the number of DOFs is @a ndofs. */
|
||||
const FiniteElement *GetFaceNbrFE(int i, int ndofs = 0) const;
|
||||
const FiniteElement *GetFaceNbrFaceFE(int i) const;
|
||||
const Array<HYPRE_BigInt> &GetFaceNbrGlobalDofMapArray() { return face_nbr_glob_dof_map; }
|
||||
const HYPRE_BigInt *GetFaceNbrGlobalDofMap() { return face_nbr_glob_dof_map; }
|
||||
ElementTransformation *GetFaceNbrElementTransformation(int i) const
|
||||
{ return pmesh->GetFaceNbrElementTransformation(i); }
|
||||
|
||||
+19
-3
@@ -199,9 +199,8 @@ const ParFiniteElementSpace *ParBlockNonlinearForm::ParFESpace(int k) const
|
||||
}
|
||||
|
||||
// Here, rhs is a true dof vector
|
||||
void ParBlockNonlinearForm::SetEssentialBC(const
|
||||
Array<Array<int> *>&bdr_attr_is_ess,
|
||||
Array<Vector *> &rhs)
|
||||
void ParBlockNonlinearForm::SetEssentialBC(
|
||||
const Array<Array<int>*> &bdr_attr_is_ess, Array<Vector*> &rhs)
|
||||
{
|
||||
Array<Vector *> nullarray(fes.Size());
|
||||
nullarray = NULL;
|
||||
@@ -217,6 +216,23 @@ void ParBlockNonlinearForm::SetEssentialBC(const
|
||||
}
|
||||
}
|
||||
|
||||
void ParBlockNonlinearForm::SetEssentialTrueDofs(
|
||||
const Array<Array<int>*> &ess_tdof_list, Array<Vector*> &rhs)
|
||||
{
|
||||
Array<Vector *> nullarray(fes.Size());
|
||||
nullarray = nullptr;
|
||||
|
||||
BlockNonlinearForm::SetEssentialTrueDofs(ess_tdof_list, nullarray);
|
||||
|
||||
for (int s = 0; s < fes.Size(); ++s)
|
||||
{
|
||||
if (rhs[s])
|
||||
{
|
||||
rhs[s]->SetSubVector(*ess_tdofs[s], 0.0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
real_t ParBlockNonlinearForm::GetEnergy(const Vector &x) const
|
||||
{
|
||||
// xs_true is not modified, so const_cast is okay
|
||||
|
||||
+33
-3
@@ -102,9 +102,39 @@ public:
|
||||
gradient-type (if different from the default) must be set again. */
|
||||
void SetParSpaces(Array<ParFiniteElementSpace *> &pf);
|
||||
|
||||
// Here, rhs is a true dof vector
|
||||
void SetEssentialBC(const Array<Array<int> *>&bdr_attr_is_ess,
|
||||
Array<Vector *> &rhs) override;
|
||||
/** @brief Set essential boundary conditions to each finite element space
|
||||
using boundary attribute markers.
|
||||
|
||||
This method calls `FiniteElementSpace::GetEssentialTrueDofs()` for each
|
||||
space and stores ess_tdof_lists internally.
|
||||
|
||||
If `rhs` vectors are non-null, the entries corresponding to these
|
||||
essential DoFs are set to zero. This ensures compatibility with the
|
||||
output of the `Mult()` method, which also zeroes out these entries.
|
||||
|
||||
@param[in] bdr_attr_is_ess A list of boundary attribute markers for each
|
||||
space.
|
||||
@param[in,out] rhs An array of optional right-hand side vectors.
|
||||
If a vector at `rhs[i]` is non-null, its essential DoFs will be set
|
||||
to zero. */
|
||||
virtual void SetEssentialBC(const Array<Array<int>*> &bdr_attr_is_ess,
|
||||
Array<Vector*> &rhs) override;
|
||||
|
||||
/** @brief Set essential boundary conditions to each finite element space
|
||||
using essential true dof lists.
|
||||
|
||||
This method stores a copy of the provided essential true dof lists.
|
||||
|
||||
If `rhs` vectors are non-null, the entries corresponding to these
|
||||
essential DoFs are set to zero. This ensures compatibility with the
|
||||
output of the `Mult()` method, which also zeroes out these entries.
|
||||
|
||||
@param[in] ess_tdof_list A list of essential true dofs for each space.
|
||||
@param[in,out] rhs An array of optional right-hand side vectors.
|
||||
If a vector at `rhs[i]` is non-null, its essential DoFs will be set
|
||||
to zero. */
|
||||
virtual void SetEssentialTrueDofs(const Array<Array<int>*> &ess_tdof_list,
|
||||
Array<Vector*> &rhs) override;
|
||||
|
||||
/// Block T-Vector to Block T-Vector
|
||||
void Mult(const Vector &x, Vector &y) const override;
|
||||
|
||||
@@ -37,6 +37,8 @@ void InitDetKernels()
|
||||
k::Specialization<3,3,3,3>::Add();
|
||||
k::Specialization<3,3,3,5>::Add();
|
||||
k::Specialization<3,3,3,6>::Add();
|
||||
k::Specialization<3,3,4,6>::Add();
|
||||
k::Specialization<3,3,3,4>::Add();
|
||||
}
|
||||
|
||||
} // namespace quadrature_interpolator
|
||||
|
||||
@@ -28,6 +28,7 @@ void InitEvalByNodesKernels()
|
||||
k::Specialization<2,QVectorLayout::byNODES,1,2,4>::Opt<1>::Add();
|
||||
k::Specialization<2,QVectorLayout::byNODES,1,3,2>::Opt<1>::Add();
|
||||
k::Specialization<2,QVectorLayout::byNODES,1,3,4>::Opt<1>::Add();
|
||||
k::Specialization<2,QVectorLayout::byNODES,1,3,6>::Opt<1>::Add();
|
||||
k::Specialization<2,QVectorLayout::byNODES,1,4,3>::Opt<1>::Add();
|
||||
k::Specialization<2,QVectorLayout::byNODES,1,4,4>::Opt<1>::Add();
|
||||
|
||||
|
||||
@@ -30,6 +30,7 @@ void InitEvalByVDimKernels()
|
||||
k::Specialization<2,QVectorLayout::byVDIM,2,2,4>::Opt<8>::Add();
|
||||
k::Specialization<2,QVectorLayout::byVDIM,2,3,4>::Opt<8>::Add();
|
||||
k::Specialization<2,QVectorLayout::byVDIM,2,3,6>::Opt<4>::Add();
|
||||
k::Specialization<2,QVectorLayout::byVDIM,2,4,6>::Opt<2>::Add();
|
||||
k::Specialization<2,QVectorLayout::byVDIM,2,4,8>::Opt<2>::Add();
|
||||
// 3D
|
||||
k::Specialization<3,QVectorLayout::byVDIM,1,2,4>::Opt<1>::Add();
|
||||
@@ -47,6 +48,9 @@ void InitEvalByVDimKernels()
|
||||
k::Specialization<3,QVectorLayout::byVDIM,3,7,7>::Opt<1>::Add();
|
||||
k::Specialization<3,QVectorLayout::byVDIM,3,8,8>::Opt<1>::Add();
|
||||
k::Specialization<3,QVectorLayout::byVDIM,3,9,9>::Opt<1>::Add();
|
||||
|
||||
k::Specialization<3,QVectorLayout::byVDIM,3,4,6>::Opt<1>::Add();
|
||||
k::Specialization<3,QVectorLayout::byVDIM,3,3,4>::Opt<1>::Add();
|
||||
}
|
||||
|
||||
} // namespace quadrature_interpolator
|
||||
|
||||
@@ -154,6 +154,21 @@ int Array<T>::IsSorted() const
|
||||
return 1;
|
||||
}
|
||||
|
||||
template <class T>
|
||||
bool Array<T>::IsConstant() const
|
||||
{
|
||||
if (size < 2) { return true; }
|
||||
const T v0 = data[0];
|
||||
for (int i = 1; i < size; i++)
|
||||
{
|
||||
if (data[i] != v0)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
template <class T>
|
||||
void Array2D<T>::Load(const char *filename, int fmt)
|
||||
|
||||
+34
-1
@@ -217,7 +217,7 @@ public:
|
||||
/// Reduces the capacity of the array to exactly match the current size.
|
||||
inline void ShrinkToFit();
|
||||
|
||||
/// Create a copy of the internal array to the provided @a copy.
|
||||
/// Create a copy of the internal array to the provided @a copy.
|
||||
inline void Copy(Array ©) const;
|
||||
|
||||
/// Make this Array a reference to a pointer.
|
||||
@@ -302,6 +302,9 @@ public:
|
||||
/// Does the Array have Size zero.
|
||||
bool IsEmpty() const { return Size() == 0; }
|
||||
|
||||
/// Return true if all entries of the array are the same.
|
||||
bool IsConstant() const;
|
||||
|
||||
/// Fill the entries of the array with the cumulative sum of the entries.
|
||||
void PartialSum();
|
||||
|
||||
@@ -412,10 +415,13 @@ private:
|
||||
|
||||
public:
|
||||
Array2D() { M = N = 0; }
|
||||
|
||||
/// Construct an m x n 2D array.
|
||||
Array2D(int m, int n) : array1d(m*n) { M = m; N = n; }
|
||||
|
||||
Array2D(const Array2D &) = default;
|
||||
|
||||
/// Set the 2D array size to m x n.
|
||||
void SetSize(int m, int n) { array1d.SetSize(m*n); M = m; N = n; }
|
||||
|
||||
int NumRows() const { return M; }
|
||||
@@ -472,9 +478,11 @@ public:
|
||||
void Load(int new_size0,int new_size1, std::istream &in)
|
||||
{ SetSize(new_size0,new_size1); Load(in, 1); }
|
||||
|
||||
/// Create a copy of the internal array to the provided @a copy.
|
||||
void Copy(Array2D ©) const
|
||||
{ copy.M = M; copy.N = N; array1d.Copy(copy.array1d); }
|
||||
|
||||
/// Set all entries of the array to the provided constant.
|
||||
inline void operator=(const T &a)
|
||||
{ array1d = a; }
|
||||
|
||||
@@ -489,6 +497,14 @@ public:
|
||||
|
||||
/// Prints array to stream with width elements per row
|
||||
void Print(std::ostream &out = mfem::out, int width = 4);
|
||||
|
||||
/** @brief Find the maximal element in the array, using the comparison
|
||||
operator `<` for class T. */
|
||||
T Max() const { return array1d.Max(); }
|
||||
|
||||
/** @brief Find the minimal element in the array, using the comparison
|
||||
operator `<` for class T. */
|
||||
T Min() const { return array1d.Min(); }
|
||||
};
|
||||
|
||||
|
||||
@@ -501,15 +517,32 @@ private:
|
||||
|
||||
public:
|
||||
Array3D() { N2 = N3 = 0; }
|
||||
|
||||
/// Construct a 3D array of size n1 x n2 x n3.
|
||||
Array3D(int n1, int n2, int n3)
|
||||
: array1d(n1*n2*n3) { N2 = n2; N3 = n3; }
|
||||
|
||||
/// Set the 3D array size to n1 x n2 x n3.
|
||||
void SetSize(int n1, int n2, int n3)
|
||||
{ array1d.SetSize(n1*n2*n3); N2 = n2; N3 = n3; }
|
||||
|
||||
/// Get the 3D array size in the first dimension.
|
||||
int GetSize1() const
|
||||
{
|
||||
const int size = array1d.Size();
|
||||
return size == 0 ? 0 : size / (N2 * N3);
|
||||
}
|
||||
|
||||
/// Get the 3D array size in the second dimension.
|
||||
int GetSize2() const { return N2; }
|
||||
|
||||
/// Get the 3D array size in the third dimension.
|
||||
int GetSize3() const { return N3; }
|
||||
|
||||
inline const T &operator()(int i, int j, int k) const;
|
||||
inline T &operator()(int i, int j, int k);
|
||||
|
||||
/// Set all entries of the array to the provided constant.
|
||||
inline void operator=(const T &a)
|
||||
{ array1d = a; }
|
||||
};
|
||||
|
||||
@@ -466,7 +466,7 @@ template<class B, class R> struct reduction_kernel
|
||||
/// helper for computing the reduction block size
|
||||
static int block_log2(unsigned N)
|
||||
{
|
||||
#if defined(__GNUC__) or defined(__clang__)
|
||||
#if defined(__GNUC__) || defined(__clang__)
|
||||
return N ? (sizeof(unsigned) * 8 - __builtin_clz(N)) : 0;
|
||||
#elif defined(_MSC_VER)
|
||||
return sizeof(unsigned) * 8 - __lzclz(N);
|
||||
|
||||
@@ -68,7 +68,7 @@ void MagmaBatchedLinAlg::AddMult(const DenseTensor &A, const Vector &x,
|
||||
auto d_x = x.Read(); // Shape (n, k, n_mat);
|
||||
auto d_y = beta == 0.0 ? y.Write() : y.ReadWrite(); // Shape (m, k, n_mat);
|
||||
|
||||
magma_trans_t magma_op = tr ? MagmaNoTrans : MagmaTrans;
|
||||
magma_trans_t magma_op = tr ? MagmaTrans : MagmaNoTrans;
|
||||
|
||||
MFEM_MAGMABLAS_PREFIX(gemm_batched_strided)(
|
||||
magma_op, MagmaNoTrans, m, k, n, alpha, d_A, m, m*n, d_x, n, n*k,
|
||||
@@ -167,7 +167,7 @@ void MagmaBatchedLinAlg::Invert(DenseTensor &A) const
|
||||
magma_int_t status;
|
||||
|
||||
status = MFEM_MAGMA_PREFIX(getrf_batched)(
|
||||
n, n, d_A_ptrs, n, d_P_ptrs, info_array.Write(), n_mat,
|
||||
n, n, d_LU_ptrs, n, d_P_ptrs, info_array.Write(), n_mat,
|
||||
Magma::Queue());
|
||||
MFEM_VERIFY(status == MAGMA_SUCCESS, "");
|
||||
|
||||
|
||||
+2
-2
@@ -2327,8 +2327,8 @@ void HypreParMatrix::Threshold(real_t threshold)
|
||||
/* TODO: GenerateDiagAndOffd() uses an int array of size equal to the number
|
||||
of columns in csr_A_wo_z which is the global number of columns in A. This
|
||||
does not scale well. */
|
||||
ierr += GenerateDiagAndOffd(csr_A_wo_z,parcsr_A_ptr,
|
||||
col_start,col_end);
|
||||
ierr += hypre_GenerateDiagAndOffd(csr_A_wo_z,parcsr_A_ptr,
|
||||
col_start,col_end);
|
||||
|
||||
ierr += hypre_CSRMatrixDestroy(csr_A_wo_z);
|
||||
|
||||
|
||||
@@ -25,11 +25,18 @@
|
||||
#define HYPRE_TIMING
|
||||
|
||||
// hypre header files
|
||||
#if MFEM_HYPRE_VERSION < 30000
|
||||
#include <seq_mv.h>
|
||||
#include <temp_multivector.h>
|
||||
#else
|
||||
#include <_hypre_seq_mv.h>
|
||||
#include <_hypre_lobpcg_temp_multivector.h>
|
||||
#endif
|
||||
#include <_hypre_parcsr_mv.h>
|
||||
#include <_hypre_parcsr_ls.h>
|
||||
|
||||
#include <HYPRE_parcsr_ls.h>
|
||||
|
||||
#ifdef HYPRE_COMPLEX
|
||||
#error "MFEM does not work with HYPRE's complex numbers support"
|
||||
#endif
|
||||
@@ -53,6 +60,10 @@
|
||||
#error "MFEM_USE_HIP=YES is required when HYPRE is built with HIP!"
|
||||
#endif
|
||||
|
||||
#if MFEM_HYPRE_VERSION > 21500
|
||||
#define HYPRE_AssumedPartitionCheck() 1
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
|
||||
@@ -1916,9 +1916,9 @@ hypre_ParCSRMatrixAdd(hypre_ParCSRMatrix *A,
|
||||
/* FIXME: GenerateDiagAndOffd() uses an int array of size equal to the
|
||||
number of columns in csr_C_temp which is the global number of columns
|
||||
in A and B. This does not scale well. */
|
||||
ierr += GenerateDiagAndOffd(csr_C_temp, C,
|
||||
hypre_ParCSRMatrixFirstColDiag(A),
|
||||
hypre_ParCSRMatrixLastColDiag(A));
|
||||
ierr += hypre_GenerateDiagAndOffd(csr_C_temp, C,
|
||||
hypre_ParCSRMatrixFirstColDiag(A),
|
||||
hypre_ParCSRMatrixLastColDiag(A));
|
||||
|
||||
/* delete CSR version of C */
|
||||
ierr += hypre_CSRMatrixDestroy(csr_C_temp);
|
||||
|
||||
@@ -21,6 +21,10 @@
|
||||
// hypre header files
|
||||
#include <_hypre_parcsr_mv.h>
|
||||
|
||||
#if MFEM_HYPRE_VERSION < 30000
|
||||
#define hypre_GenerateDiagAndOffd GenerateDiagAndOffd
|
||||
#endif
|
||||
|
||||
// Older hypre versions do not define HYPRE_BigInt and HYPRE_MPI_BIG_INT, so we
|
||||
// define them here for backward compatibility.
|
||||
#if MFEM_HYPRE_VERSION < 21600
|
||||
|
||||
+1
-1
@@ -1019,7 +1019,7 @@ MMA::MMA(MPI_Comm comm_, int nVar, int nCon, real_t *xval, int iter)
|
||||
mSubProblem.reset(new MMA::MMASubSvanberg(*this, nVar, nCon));
|
||||
}
|
||||
|
||||
MMA::MMA(MPI_Comm comm_, const int & nVar, const int & nCon,
|
||||
MMA::MMA(MPI_Comm comm_, const int nVar, const int nCon,
|
||||
const Vector & xval, int iter) : MMA(comm_, nVar, nCon, xval.GetData(), iter)
|
||||
{}
|
||||
#endif
|
||||
|
||||
+166
-31
@@ -25,25 +25,42 @@ namespace mfem
|
||||
// forward declaration
|
||||
class Vector;
|
||||
|
||||
/** \brief MMA (Method of Moving Asymptotes) solves an optimization problem
|
||||
* of the form:
|
||||
/** \brief MMA (Method of Moving Asymptotes) solves a nonlinear optimization
|
||||
* problem involving an objective function, inequality constraints,
|
||||
* and variable bounds.
|
||||
*
|
||||
* Find x that minimizes the objective function F(x),
|
||||
* subject to C(x)_i <= 0, for all i = 1, ... m
|
||||
* x_lo <= x <= x_hi.
|
||||
* \details
|
||||
* This class finds ${\bf x} \in R^n$ that solves the following nonlinear
|
||||
* program:
|
||||
* $$
|
||||
* \begin{array}{ll}
|
||||
* \min_{{\bf x} \in R^n} & F({\bf x})\\
|
||||
* \textrm{subject to} & C({\bf x})_i \leq 0,\quad
|
||||
* \textrm{for all}\quad i = 1,\ldots m\\
|
||||
* & {\bf x}_{\textrm{lo}} \leq {\bf x} \leq
|
||||
* {\bf x}_{\textrm{hi}}.
|
||||
* \end{array}
|
||||
* $$
|
||||
* Here $F : R^n \to R$ is the objective function, and
|
||||
* $C : R^n \to R^m$ is a set of $m$ inequality constraints. The
|
||||
* variable bounds are sometimes called box constraints. By
|
||||
* convention, the routine seeks ${\bf x}$ that minimizes the
|
||||
* objective function, $F$. Maximization problems should be
|
||||
* reformulated as a minimization of $-F$.
|
||||
*
|
||||
* The objective functions are replaced by convex functions
|
||||
* chosen based on gradient information, and solved using a dual method.
|
||||
* The unique optimal solution of this subproblem is returned as the next
|
||||
* iteration point. Optimality is determined by the KKT conditions.
|
||||
*
|
||||
* The "Update" function in MMA advances the optimization and must be called
|
||||
* in every optimization iteration. Current and previous iteration points
|
||||
* construct the "moving asymptotes". The design variables, objective function,
|
||||
* constraints are passed to an approximating subproblem. The design variables
|
||||
* are updated and returned. Its implementation closely follows the original
|
||||
* formulation of 'Svanberg, K. (2007). MMA and GCMMA-two methods
|
||||
* for nonlinear optimization. vol, 1, 1-15.'
|
||||
* The "Update" function in MMA advances the optimization and must be
|
||||
* called in every optimization iteration. Current and previous iteration
|
||||
* points construct the "moving asymptotes". The design variables,
|
||||
* objective function, constraints are passed to an approximating
|
||||
* subproblem. The design variables are updated and returned. Its
|
||||
* implementation closely follows the original formulation of <a
|
||||
* href="https://people.kth.se/~krille/mmagcmma.pdf">'Svanberg, K. (2007).
|
||||
* MMA and GCMMA-two methods for nonlinear optimization. vol, 1, 1-15.'</a>
|
||||
*
|
||||
* When used in parallel, all Vectors are assumed to be true dof vectors,
|
||||
* and the operators are expected to be defined for tdof vectors.
|
||||
@@ -52,46 +69,164 @@ class Vector;
|
||||
class MMA
|
||||
{
|
||||
public:
|
||||
/// Serial constructor:
|
||||
/// nVar - number of design parameters;
|
||||
/// nCon - number of constraints;
|
||||
/// xval[nVar] - initial parameter values
|
||||
/**
|
||||
* \brief Serial constructor
|
||||
* \param nVar total number of design parameters
|
||||
* \param nCon number of inequality constraints (i.e., $C$)
|
||||
* \param xval initial values for design parameters (a pointer
|
||||
* to \p nVar doubles). Caller retains ownership of
|
||||
* this pointer/data.
|
||||
* \param iterationNumber the starting iteration number
|
||||
*/
|
||||
MMA(int nVar, int nCon, real_t *xval, int iterationNumber = 0);
|
||||
|
||||
/**
|
||||
* \brief Serial constructor
|
||||
* \param nVar total number of design parameters
|
||||
* \param nCon number of inequality constraints (i.e., $C$)
|
||||
* \param xval initial values for design parameters (size should
|
||||
* be \p nVar). Caller retains ownership of
|
||||
* this Vector.
|
||||
* \param iterationNumber the starting iteration number
|
||||
*/
|
||||
MMA(const int nVar, int nCon, Vector & xval, int iterationNumber = 0);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// Parallel constructor:
|
||||
/// comm_ - communicator
|
||||
/**
|
||||
* \brief Parallel constructor
|
||||
* \param comm_ the MPI communicator participating in the NLP solve
|
||||
* \param nVar number of design parameters on this MPI rank
|
||||
* \param nCon total number of inequality constraints (i.e., $C$).
|
||||
* Every MPI rank provides the same value here.
|
||||
* \param xval initial values for design parameters on this MPI rank
|
||||
* (a pointer to \p nVar doubles). Caller retains ownership
|
||||
* of this pointer/data.
|
||||
* \param iterationNumber the starting iteration number. All MPI ranks
|
||||
* should pass in the same value here.
|
||||
*
|
||||
* \details
|
||||
* Each MPI rank has a subset of the total design variable vector, and
|
||||
* calls for that MPI rank always address its subset of the design
|
||||
* variable vector and gradients with respect to its subset of the design
|
||||
* variable vector.
|
||||
*
|
||||
* If you wanted to determine the global number of design variables, it
|
||||
* would be determined as follows:
|
||||
* \code{.cpp}
|
||||
* int globalDesignVars;
|
||||
* MPI_Allreduce(&nVar, &globalDesignVars, 1, MPI_INT, MPI_SUM, comm_);
|
||||
* \endcode
|
||||
*/
|
||||
MMA(MPI_Comm comm_, int nVar, int nCon, real_t *xval,
|
||||
int iterationNumber = 0);
|
||||
MMA(MPI_Comm comm_, const int & nVar, const int & nCon, const Vector & xval,
|
||||
/**
|
||||
* \brief Parallel constructor
|
||||
* \param comm_ the MPI communicator participating in the NLP solve
|
||||
* \param nVar number of design parameters on this MPI rank
|
||||
* \param nCon total number of inequality constraints (i.e., $C$).
|
||||
* Every MPI rank provides the same value here.
|
||||
* \param xval initial values for design parameters (size should
|
||||
* be \p nVar). Caller retains ownership of
|
||||
* this Vector.
|
||||
* \param iterationNumber the starting iteration number. All MPI ranks
|
||||
* should pass in the same value here.
|
||||
*
|
||||
* \details
|
||||
* Each MPI rank has a subset of the total design variable vector, and
|
||||
* calls for that MPI rank always address its subset of the design
|
||||
* variable vector and gradients with respect to its subset of the design
|
||||
* variable vector.
|
||||
*
|
||||
* If you wanted to determine the global number of design variables, it
|
||||
* would be determined as follows:
|
||||
* \code{.cpp}
|
||||
* int globalDesignVars;
|
||||
* MPI_Allreduce(&nVar, &globalDesignVars, 1, MPI_INT, MPI_SUM, comm_);
|
||||
* \endcode
|
||||
*/
|
||||
MMA(MPI_Comm comm_, const int nVar, const int nCon, const Vector & xval,
|
||||
int iterationNumber = 0);
|
||||
#endif
|
||||
|
||||
/// Destructor
|
||||
~MMA();
|
||||
|
||||
/// Update the optimization parameters
|
||||
/// dfdx[nVar] - gradients of the objective
|
||||
/// gx[nCon] - values of the constraints
|
||||
/// dgdx[nCon*nVar] - gradients of the constraints ordered
|
||||
/// constraint by constraint, e.g. {dg0dx0, dg0dx1, ... ,}
|
||||
/// {dg1dx0, dg1dx1, ... ,}
|
||||
/// xmin[nVar] - lower bounds
|
||||
/// xmax[nVar] - upper bounds
|
||||
/// xval[nVar] - input/output for optimization parameters
|
||||
/**
|
||||
* \brief Update the optimization parameters for a constrained
|
||||
* nonlinear program
|
||||
* \param dfdx vector of size nVar holding the gradients of the
|
||||
* objective function with respect to
|
||||
* the design variables,
|
||||
* $\frac{\partial F}{\partial {\bf x}_i}$
|
||||
* for each variable on this rank.
|
||||
* \param gx vector of size nCon holding the values of the
|
||||
* inequality constraints. Every MPI rank should
|
||||
* pass in the same values here.
|
||||
* \param dgdx vector of size $\textrm{nCon}\cdot\textrm{nVar}$
|
||||
* holding the gradients of the constraints in
|
||||
* row-major order. For example, {dg0dx0, dg0dx1, ...,}
|
||||
* {dg1dx0, dg1dx1, ..., }, ...
|
||||
* \param xmin vector of size nVar holding the lower bounds on
|
||||
* the design values. \p xmin and \p xmax are
|
||||
* the box constraints.
|
||||
* \param xmax vector of size nVar holding the upper bounds on
|
||||
* the design values. \p xmin and \p xmax are
|
||||
* the box constraints.
|
||||
* \param xval vector of size nVar. On entry, this holds the
|
||||
* value of the design variables where the objective,
|
||||
* constraints, and their gradients were evaluated.
|
||||
* On exit, this holds the result of the MMA iteration,
|
||||
* the next design variable value to use.
|
||||
*
|
||||
* \details
|
||||
* The caller retains ownership of all Vectors passed into this method.
|
||||
*/
|
||||
void Update(const Vector& dfdx,
|
||||
const Vector& gx, const Vector& dgdx,
|
||||
const Vector& xmin, const Vector& xmax,
|
||||
Vector& xval);
|
||||
/// Unconstrained
|
||||
|
||||
/**
|
||||
* \brief Update the optimization parameters for an unconstrained
|
||||
* nonlinear program
|
||||
* \param dfdx vector of size nVar holding the gradients of the
|
||||
* objective function with respect to
|
||||
* the design variables,
|
||||
* $\frac{\partial F}{\partial {\bf x}_i}$
|
||||
* for each variable on this rank.
|
||||
* \param xmin vector of size nVar holding the lower bounds on
|
||||
* the design values. \p xmin and \p xmax are
|
||||
* the box constraints.
|
||||
* \param xmax vector of size nVar holding the upper bounds on
|
||||
* the design values. \p xmin and \p xmax are
|
||||
* the box constraints.
|
||||
* \param xval vector of size nVar. On entry, this holds the
|
||||
* value of the design variables where the objective,
|
||||
* constraints, and their gradients were evaluated.
|
||||
* On exit, this holds the result of the MMA iteration,
|
||||
* the next design variable value to use.
|
||||
*
|
||||
* \details
|
||||
* The caller retains ownership of all Vectors passed into this method.
|
||||
* This should be used when the number of inequality constraints is zero.
|
||||
*/
|
||||
void Update( const Vector& dfdx,
|
||||
const Vector& xmin, const Vector& xmax,
|
||||
Vector& xval);
|
||||
|
||||
/**
|
||||
* \brief Change the iteration number
|
||||
* \param iterationNumber the new iteration number
|
||||
*/
|
||||
void SetIteration( int iterationNumber ) { iter = iterationNumber; };
|
||||
int GetIteration() { return iter; };
|
||||
|
||||
/// Return the current iteration number
|
||||
int GetIteration() const { return iter; };
|
||||
|
||||
/**
|
||||
* \brief change the print level
|
||||
* \param print_lvl the new print level
|
||||
*/
|
||||
void SetPrintLevel(int print_lvl) { print_level = print_lvl; }
|
||||
|
||||
protected:
|
||||
@@ -123,7 +258,7 @@ private:
|
||||
/// KKT norm
|
||||
real_t kktnorm;
|
||||
|
||||
/// intialization state
|
||||
/// initialization state
|
||||
bool isInitialized = false;
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
@@ -1356,6 +1356,7 @@ void PetscParMatrix::MakeWrapper(MPI_Comm comm, const Operator* op, Mat *A)
|
||||
PETSC_DECIDE,PETSC_DECIDE); PCHKERRQ(A,ierr);
|
||||
ierr = MatSetType(*A,MATSHELL); PCHKERRQ(A,ierr);
|
||||
ierr = MatShellSetContext(*A,(void *)op); PCHKERRQ(A,ierr);
|
||||
#if PETSC_VERSION_LT(3,24,0)
|
||||
ierr = MatShellSetOperation(*A,MATOP_MULT,
|
||||
(void (*)())__mfem_mat_shell_apply);
|
||||
PCHKERRQ(A,ierr);
|
||||
@@ -1367,6 +1368,19 @@ void PetscParMatrix::MakeWrapper(MPI_Comm comm, const Operator* op, Mat *A)
|
||||
PCHKERRQ(A,ierr);
|
||||
ierr = MatShellSetOperation(*A,MATOP_DESTROY,
|
||||
(void (*)())__mfem_mat_shell_destroy);
|
||||
#else
|
||||
ierr = MatShellSetOperation(*A,MATOP_MULT,
|
||||
(PetscErrorCodeFn*)__mfem_mat_shell_apply);
|
||||
PCHKERRQ(A,ierr);
|
||||
ierr = MatShellSetOperation(*A,MATOP_MULT_TRANSPOSE,
|
||||
(PetscErrorCodeFn*)__mfem_mat_shell_apply_transpose);
|
||||
PCHKERRQ(A,ierr);
|
||||
ierr = MatShellSetOperation(*A,MATOP_COPY,
|
||||
(PetscErrorCodeFn*)__mfem_mat_shell_copy);
|
||||
PCHKERRQ(A,ierr);
|
||||
ierr = MatShellSetOperation(*A,MATOP_DESTROY,
|
||||
(PetscErrorCodeFn*)__mfem_mat_shell_destroy);
|
||||
#endif
|
||||
#if defined(_USE_DEVICE)
|
||||
MemoryType mt = GetMemoryType(op->GetMemoryClass());
|
||||
if (mt == MemoryType::DEVICE || mt == MemoryType::MANAGED)
|
||||
|
||||
+73
-117
@@ -46,6 +46,12 @@
|
||||
#define MFEM_GPUSPARSE_ALG HIPSPARSE_CSRMV_ALG1
|
||||
#endif // defined(MFEM_USE_CUDA)
|
||||
|
||||
#if defined(MFEM_USE_SINGLE)
|
||||
#define MFEM_CUDA_or_HIP_REAL_T MFEM_CUDA_or_HIP(_R_32F)
|
||||
#elif defined(MFEM_USE_DOUBLE)
|
||||
#define MFEM_CUDA_or_HIP_REAL_T MFEM_CUDA_or_HIP(_R_64F)
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
@@ -57,8 +63,10 @@ int SparseMatrix::SparseMatrixCount = 0;
|
||||
/// @cond Suppress_Doxygen_warnings
|
||||
MFEM_cu_or_hip(sparseHandle_t) SparseMatrix::handle = nullptr;
|
||||
/// @endcond
|
||||
#ifndef MFEM_CUDA_1897_WORKAROUND
|
||||
size_t SparseMatrix::bufferSize = 0;
|
||||
void * SparseMatrix::dBuffer = nullptr;
|
||||
#endif
|
||||
#endif // MFEM_USE_CUDA_OR_HIP
|
||||
|
||||
void SparseMatrix::InitGPUSparse()
|
||||
@@ -464,109 +472,67 @@ void SparseMatrix::SortColumnIndices()
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_CUDA_OR_HIP
|
||||
if ( Device::Allows( Backend::CUDA_MASK ))
|
||||
if (Device::Allows(Backend::CUDA_MASK) || Device::Allows(Backend::HIP_MASK))
|
||||
{
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
size_t pBufferSizeInBytes = 0;
|
||||
void *pBuffer = NULL;
|
||||
|
||||
const int n = Height();
|
||||
const int m = Width();
|
||||
const int m = Height();
|
||||
const int n = Width();
|
||||
const int nnzA = J.Capacity();
|
||||
real_t * d_a_sorted = ReadWriteData();
|
||||
const int * d_ia = ReadI();
|
||||
int * d_ja_sorted = ReadWriteJ();
|
||||
csru2csrInfo_t sortInfoA;
|
||||
const int *d_ia = ReadI();
|
||||
int *d_ja = ReadWriteJ();
|
||||
|
||||
cusparseMatDescr_t matA_descr;
|
||||
cusparseCreateMatDescr( &matA_descr );
|
||||
cusparseSetMatIndexBase( matA_descr, CUSPARSE_INDEX_BASE_ZERO );
|
||||
cusparseSetMatType( matA_descr, CUSPARSE_MATRIX_TYPE_GENERAL );
|
||||
// Get size of temporary buffer needed to sort the column indices,
|
||||
// allocate the temporary buffer.
|
||||
size_t pBufferSizeInBytes;
|
||||
MFEM_cu_or_hip(sparseXcsrsort_bufferSizeExt)(handle, m, n, nnzA, d_ia,
|
||||
d_ja, &pBufferSizeInBytes);
|
||||
void *pBuffer = MFEM_Cu_or_Hip(MemAlloc)(&pBuffer, pBufferSizeInBytes);
|
||||
|
||||
cusparseCreateCsru2csrInfo( &sortInfoA );
|
||||
// Create matrix descriptor, will have default values
|
||||
// CUSPARSE_INDEX_BASE_ZERO and CUSPARSE_MATRIX_TYPE_GENERAL.
|
||||
MFEM_cu_or_hip(sparseMatDescr_t) matA_descr;
|
||||
MFEM_cu_or_hip(sparseCreateMatDescr)(&matA_descr);
|
||||
|
||||
#ifdef MFEM_USE_SINGLE
|
||||
cusparseScsru2csr_bufferSizeExt( handle, n, m, nnzA, d_a_sorted, d_ia,
|
||||
d_ja_sorted, sortInfoA,
|
||||
&pBufferSizeInBytes);
|
||||
#elif defined MFEM_USE_DOUBLE
|
||||
cusparseDcsru2csr_bufferSizeExt( handle, n, m, nnzA, d_a_sorted, d_ia,
|
||||
d_ja_sorted, sortInfoA,
|
||||
&pBufferSizeInBytes);
|
||||
#else
|
||||
MFEM_ABORT("Floating point type undefined");
|
||||
#endif
|
||||
// Initialize permutation to identity
|
||||
Array<int> P(nnzA);
|
||||
int *d_P = P.Write();
|
||||
mfem::forall(nnzA, [=] MFEM_HOST_DEVICE (int i) { d_P[i] = i; });
|
||||
|
||||
CuMemAlloc( &pBuffer, pBufferSizeInBytes );
|
||||
// Sort the column indices. The array d_ja will now be sorted. The
|
||||
// permutation required to sort the values will be returned in d_P.
|
||||
MFEM_cu_or_hip(sparseXcsrsort)(handle, m, n, nnzA, matA_descr, d_ia, d_ja,
|
||||
d_P, pBuffer);
|
||||
|
||||
#ifdef MFEM_USE_SINGLE
|
||||
cusparseScsru2csr( handle, n, m, nnzA, matA_descr, d_a_sorted, d_ia,
|
||||
d_ja_sorted, sortInfoA, pBuffer);
|
||||
#elif defined MFEM_USE_DOUBLE
|
||||
cusparseDcsru2csr( handle, n, m, nnzA, matA_descr, d_a_sorted, d_ia,
|
||||
d_ja_sorted, sortInfoA, pBuffer);
|
||||
#else
|
||||
MFEM_ABORT("Floating point type undefined");
|
||||
#endif
|
||||
// Create a copy of the unsorted matrix values.
|
||||
real_t *d_a = ReadWriteData();
|
||||
void *d_a_unsorted = MFEM_Cu_or_Hip(MemAlloc)(&d_a_unsorted,
|
||||
nnzA * sizeof(real_t));
|
||||
MFEM_Cu_or_Hip(MemcpyDtoD)(d_a_unsorted, d_a, nnzA * sizeof(real_t));
|
||||
|
||||
// The above call is (at least in some cases) asynchronous, so we need to
|
||||
// wait for it to finish before we can free device temporaries.
|
||||
// Create the (input) dense vector with the unsorted values.
|
||||
MFEM_cu_or_hip(sparseDnVecDescr_t) d_a_dense;
|
||||
MFEM_cu_or_hip(sparseCreateDnVec)(&d_a_dense, nnzA, d_a_unsorted,
|
||||
MFEM_CUDA_or_HIP_REAL_T);
|
||||
|
||||
// Create the (output) sparse vector that will have the sorted values.
|
||||
MFEM_cu_or_hip(sparseSpVecDescr_t) d_a_sparse;
|
||||
MFEM_cu_or_hip(sparseCreateSpVec)(&d_a_sparse, nnzA, nnzA, d_P, d_a,
|
||||
MFEM_CU_or_HIP(SPARSE_INDEX_32I),
|
||||
MFEM_CU_or_HIP(SPARSE_INDEX_BASE_ZERO),
|
||||
MFEM_CUDA_or_HIP_REAL_T);
|
||||
|
||||
// Sort the matrix values using the permutation vector.
|
||||
MFEM_cu_or_hip(sparseGather)(handle, d_a_dense, d_a_sparse);
|
||||
|
||||
// The above calls may be asynchronous, so we need to wait for them to
|
||||
// finish before we can free memory.
|
||||
MFEM_STREAM_SYNC;
|
||||
|
||||
cusparseDestroyCsru2csrInfo( sortInfoA );
|
||||
cusparseDestroyMatDescr( matA_descr );
|
||||
MFEM_cu_or_hip(sparseDestroyDnVec)(d_a_dense);
|
||||
MFEM_cu_or_hip(sparseDestroySpVec)(d_a_sparse);
|
||||
MFEM_cu_or_hip(sparseDestroyMatDescr)(matA_descr);
|
||||
|
||||
CuMemFree( pBuffer );
|
||||
#endif
|
||||
}
|
||||
else if ( Device::Allows( Backend::HIP_MASK ))
|
||||
{
|
||||
#if defined(MFEM_USE_HIP)
|
||||
size_t pBufferSizeInBytes = 0;
|
||||
void *pBuffer = NULL;
|
||||
int *P = NULL;
|
||||
|
||||
const int n = Height();
|
||||
const int m = Width();
|
||||
const int nnzA = J.Capacity();
|
||||
real_t * d_a_sorted = ReadWriteData();
|
||||
const int * d_ia = ReadI();
|
||||
int * d_ja_sorted = ReadWriteJ();
|
||||
|
||||
hipsparseMatDescr_t descrA;
|
||||
hipsparseCreateMatDescr( &descrA );
|
||||
// FIXME: There is not in-place version of csr sort in hipSPARSE currently, so we make
|
||||
// a temporary copy of the data for gthr, sort that, and then copy the sorted values
|
||||
// back to the array being returned. Where there is an in-place version available,
|
||||
// we should use it.
|
||||
Array< real_t > a_tmp( nnzA );
|
||||
real_t *d_a_tmp = a_tmp.Write();
|
||||
|
||||
hipsparseXcsrsort_bufferSizeExt(handle, n, m, nnzA, d_ia, d_ja_sorted,
|
||||
&pBufferSizeInBytes);
|
||||
|
||||
HipMemAlloc( &pBuffer, pBufferSizeInBytes );
|
||||
HipMemAlloc( (void**)&P, nnzA * sizeof(int) );
|
||||
|
||||
hipsparseCreateIdentityPermutation(handle, nnzA, P);
|
||||
hipsparseXcsrsort(handle, n, m, nnzA, descrA, d_ia, d_ja_sorted, P, pBuffer);
|
||||
|
||||
#if defined(MFEM_USE_SINGLE)
|
||||
hipsparseSgthr(handle, nnzA, d_a_sorted, d_a_tmp, P,
|
||||
HIPSPARSE_INDEX_BASE_ZERO);
|
||||
#elif defined(MFEM_USE_DOUBLE)
|
||||
hipsparseDgthr(handle, nnzA, d_a_sorted, d_a_tmp, P,
|
||||
HIPSPARSE_INDEX_BASE_ZERO);
|
||||
#else
|
||||
MFEM_ABORT("Unsupported floating point type!");
|
||||
#endif
|
||||
|
||||
A.CopyFrom( a_tmp.GetMemory(), nnzA );
|
||||
hipsparseDestroyMatDescr( descrA );
|
||||
|
||||
HipMemFree( pBuffer );
|
||||
HipMemFree( P );
|
||||
#endif
|
||||
MFEM_Cu_or_Hip(MemFree)(d_a_unsorted);
|
||||
MFEM_Cu_or_Hip(MemFree)(pBuffer);
|
||||
}
|
||||
else
|
||||
#endif // MFEM_USE_CUDA_OR_HIP
|
||||
@@ -821,27 +787,15 @@ void SparseMatrix::AddMult(const Vector &x, Vector &y, const real_t a) const
|
||||
MFEM_CU_or_HIP(SPARSE_INDEX_32I),
|
||||
MFEM_CU_or_HIP(SPARSE_INDEX_32I),
|
||||
MFEM_CU_or_HIP(SPARSE_INDEX_BASE_ZERO),
|
||||
#ifdef MFEM_USE_SINGLE
|
||||
MFEM_CUDA_or_HIP(_R_32F));
|
||||
#else
|
||||
MFEM_CUDA_or_HIP(_R_64F));
|
||||
#endif
|
||||
MFEM_CUDA_or_HIP_REAL_T);
|
||||
|
||||
// Create handles for input/output vectors
|
||||
MFEM_cu_or_hip(sparseCreateDnVec)(&vecX_descr,
|
||||
x.Size(),
|
||||
const_cast<real_t *>(d_x),
|
||||
#ifdef MFEM_USE_SINGLE
|
||||
MFEM_CUDA_or_HIP(_R_32F));
|
||||
#else
|
||||
MFEM_CUDA_or_HIP(_R_64F));
|
||||
#endif
|
||||
MFEM_CUDA_or_HIP_REAL_T);
|
||||
MFEM_cu_or_hip(sparseCreateDnVec)(&vecY_descr, y.Size(), d_y,
|
||||
#ifdef MFEM_USE_SINGLE
|
||||
MFEM_CUDA_or_HIP(_R_32F));
|
||||
#else
|
||||
MFEM_CUDA_or_HIP(_R_64F));
|
||||
#endif
|
||||
MFEM_CUDA_or_HIP_REAL_T);
|
||||
#else
|
||||
cusparseCreateMatDescr(&matA_descr);
|
||||
cusparseSetMatIndexBase(matA_descr, CUSPARSE_INDEX_BASE_ZERO);
|
||||
@@ -860,11 +814,7 @@ void SparseMatrix::AddMult(const Vector &x, Vector &y, const real_t a) const
|
||||
vecX_descr,
|
||||
&beta,
|
||||
vecY_descr,
|
||||
#ifdef MFEM_USE_SINGLE
|
||||
MFEM_CUDA_or_HIP(_R_32F),
|
||||
#else
|
||||
MFEM_CUDA_or_HIP(_R_64F),
|
||||
#endif
|
||||
MFEM_CUDA_or_HIP_REAL_T,
|
||||
MFEM_GPUSPARSE_ALG,
|
||||
&newBufferSize);
|
||||
|
||||
@@ -891,11 +841,7 @@ void SparseMatrix::AddMult(const Vector &x, Vector &y, const real_t a) const
|
||||
vecX_descr,
|
||||
&beta,
|
||||
vecY_descr,
|
||||
#ifdef MFEM_USE_SINGLE
|
||||
MFEM_CUDA_or_HIP(_R_32F),
|
||||
#else
|
||||
MFEM_CUDA_or_HIP(_R_64F),
|
||||
#endif
|
||||
MFEM_CUDA_or_HIP_REAL_T,
|
||||
MFEM_GPUSPARSE_ALG,
|
||||
dBuffer);
|
||||
#else
|
||||
@@ -4372,6 +4318,14 @@ SparseMatrix::~SparseMatrix()
|
||||
#ifdef MFEM_USE_CUDA_OR_HIP
|
||||
if (Device::Allows(Backend::CUDA_MASK | Backend::HIP_MASK))
|
||||
{
|
||||
#ifdef MFEM_CUDA_1897_WORKAROUND
|
||||
if (dBuffer)
|
||||
{
|
||||
MFEM_Cu_or_Hip(MemFree)(dBuffer);
|
||||
dBuffer = nullptr;
|
||||
bufferSize = 0;
|
||||
}
|
||||
#endif
|
||||
if (SparseMatrixCount==1)
|
||||
{
|
||||
if (handle)
|
||||
@@ -4379,12 +4333,14 @@ SparseMatrix::~SparseMatrix()
|
||||
MFEM_cu_or_hip(sparseDestroy)(handle);
|
||||
handle = nullptr;
|
||||
}
|
||||
#ifndef MFEM_CUDA_1897_WORKAROUND
|
||||
if (dBuffer)
|
||||
{
|
||||
MFEM_Cu_or_Hip(MemFree)(dBuffer);
|
||||
dBuffer = nullptr;
|
||||
bufferSize = 0;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
SparseMatrixCount--;
|
||||
}
|
||||
|
||||
@@ -98,9 +98,17 @@ protected:
|
||||
#ifdef MFEM_USE_CUDA_OR_HIP
|
||||
// common for hipSPARSE and cuSPARSE
|
||||
static int SparseMatrixCount;
|
||||
mutable bool initBuffers = false;
|
||||
|
||||
#if defined(MFEM_USE_CUDA) && CUDA_VERSION >= 12300 && CUDA_VERSION < 12602
|
||||
// Workaround for bug CUSPARSE-1897
|
||||
#define MFEM_CUDA_1897_WORKAROUND
|
||||
mutable size_t bufferSize = 0;
|
||||
mutable void *dBuffer = nullptr;
|
||||
#else
|
||||
static size_t bufferSize;
|
||||
static void *dBuffer;
|
||||
mutable bool initBuffers = false;
|
||||
#endif
|
||||
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
cusparseStatus_t status;
|
||||
|
||||
@@ -377,7 +377,7 @@ MFEM_CONFIG_VARS = MFEM_CXX MFEM_HOST_CXX MFEM_CPPFLAGS MFEM_CXXFLAGS\
|
||||
MFEM_INC_DIR MFEM_TPLFLAGS MFEM_INCFLAGS MFEM_PICFLAG MFEM_FLAGS MFEM_LIB_DIR\
|
||||
MFEM_EXT_LIBS MFEM_LIBS MFEM_LIB_FILE MFEM_STATIC MFEM_SHARED MFEM_BUILD_TAG\
|
||||
MFEM_PREFIX MFEM_CONFIG_EXTRA MFEM_MPIEXEC MFEM_MPIEXEC_NP MFEM_MPI_NP\
|
||||
MFEM_TEST_MK
|
||||
MFEM_TEST_MK MFEM_XLINKER
|
||||
|
||||
# Config vars: values of the form @VAL@ are replaced by $(VAL) in config.mk
|
||||
MFEM_CPPFLAGS ?= $(CPPFLAGS)
|
||||
@@ -394,6 +394,7 @@ MFEM_BUILD_TAG ?= $(shell uname -snm)
|
||||
MFEM_PREFIX ?= $(PREFIX)
|
||||
MFEM_INC_DIR ?= $(if $(CONFIG_FILE_DEF),@MFEM_BUILD_DIR@,@MFEM_DIR@)
|
||||
MFEM_LIB_DIR ?= $(if $(CONFIG_FILE_DEF),@MFEM_BUILD_DIR@,@MFEM_DIR@)
|
||||
MFEM_XLINKER ?= $(XLINKER)
|
||||
MFEM_TEST_MK ?= @MFEM_DIR@/config/test.mk
|
||||
# Use "\n" (interpreted by sed) to add a newline.
|
||||
MFEM_CONFIG_EXTRA ?= $(if $(CONFIG_FILE_DEF),MFEM_BUILD_DIR ?= @MFEM_DIR@,)
|
||||
|
||||
@@ -20,6 +20,7 @@ set(SRCS
|
||||
mesh_operators.cpp
|
||||
mesh_readers.cpp
|
||||
ncmesh.cpp
|
||||
ncnurbs.cpp
|
||||
nurbs.cpp
|
||||
point.cpp
|
||||
pyramid.cpp
|
||||
@@ -48,6 +49,7 @@ set(HDRS
|
||||
mesh_headers.hpp
|
||||
mesh_operators.hpp
|
||||
ncmesh.hpp
|
||||
ncnurbs.hpp
|
||||
nurbs.hpp
|
||||
point.hpp
|
||||
pyramid.hpp
|
||||
|
||||
+264
-125
@@ -3496,7 +3496,6 @@ void Mesh::FinalizeHexMesh(int generate_edges, int refine, bool fix_orientation)
|
||||
void Mesh::FinalizeMesh(int refine, bool fix_orientation)
|
||||
{
|
||||
FinalizeTopology();
|
||||
|
||||
Finalize(refine, fix_orientation);
|
||||
}
|
||||
|
||||
@@ -4159,6 +4158,8 @@ void Mesh::Make3D24TetsFromHex(int nx, int ny, int nz,
|
||||
ind[5] = VertexIndex(x+1, y , z+1);
|
||||
ind[6] = VertexIndex(x+1, y+1, z+1);
|
||||
ind[7] = VertexIndex( x, y+1, z+1);
|
||||
// *INDENT-ON*
|
||||
|
||||
AddHexAs24TetsWithPoints(ind, hex_face_verts, 1);
|
||||
}
|
||||
}
|
||||
@@ -4179,8 +4180,8 @@ void Mesh::Make3D24TetsFromHex(int nx, int ny, int nz,
|
||||
|
||||
auto get3array = [](Array<int> v)
|
||||
{
|
||||
v.Sort();
|
||||
return std::array<int, 3>{v[0], v[1], v[2]};
|
||||
v.Sort();
|
||||
return std::array<int, 3> {v[0], v[1], v[2]};
|
||||
};
|
||||
|
||||
Array<int> el_faces;
|
||||
@@ -4188,32 +4189,32 @@ void Mesh::Make3D24TetsFromHex(int nx, int ny, int nz,
|
||||
Array<int> vertidxs;
|
||||
for (int i = 0; i < el_to_face->Size(); i++)
|
||||
{
|
||||
el_to_face->GetRow(i, el_faces);
|
||||
for (int j = 0; j < el_faces.Size(); j++)
|
||||
{
|
||||
GetFaceVertices(el_faces[j], vertidxs);
|
||||
auto t = get3array(vertidxs);
|
||||
auto it = tet_face_count.find(t);
|
||||
if (it == tet_face_count.end()) //edge does not already exist
|
||||
{
|
||||
tet_face_count.insert({t, 1});
|
||||
face_count_map.insert({t, el_faces[j]});
|
||||
}
|
||||
else
|
||||
{
|
||||
it->second++; // increase edge count value by 1.
|
||||
}
|
||||
}
|
||||
el_to_face->GetRow(i, el_faces);
|
||||
for (int j = 0; j < el_faces.Size(); j++)
|
||||
{
|
||||
GetFaceVertices(el_faces[j], vertidxs);
|
||||
auto t = get3array(vertidxs);
|
||||
auto it = tet_face_count.find(t);
|
||||
if (it == tet_face_count.end()) //edge does not already exist
|
||||
{
|
||||
tet_face_count.insert({t, 1});
|
||||
face_count_map.insert({t, el_faces[j]});
|
||||
}
|
||||
else
|
||||
{
|
||||
it->second++; // increase edge count value by 1.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (const auto &edge : tet_face_count)
|
||||
{
|
||||
if (edge.second == 1) //if this only appears once, it is a boundary edge
|
||||
{
|
||||
int facenum = (face_count_map.find(edge.first))->second;
|
||||
GetFaceVertices(facenum, vertidxs);
|
||||
AddBdrTriangle(vertidxs, 1);
|
||||
}
|
||||
if (edge.second == 1) //if this only appears once, it is a boundary edge
|
||||
{
|
||||
int facenum = (face_count_map.find(edge.first))->second;
|
||||
GetFaceVertices(facenum, vertidxs);
|
||||
AddBdrTriangle(vertidxs, 1);
|
||||
}
|
||||
}
|
||||
|
||||
#if 0
|
||||
@@ -4454,7 +4455,7 @@ void Mesh::Make1D(int n, real_t sx)
|
||||
}
|
||||
|
||||
Mesh::Mesh(const Mesh &mesh, bool copy_nodes)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
{
|
||||
Dim = mesh.Dim;
|
||||
spaceDim = mesh.spaceDim;
|
||||
@@ -4634,7 +4635,7 @@ Mesh Mesh::MakeCartesian3D(
|
||||
}
|
||||
|
||||
Mesh Mesh::MakeCartesian3DWith24TetsPerHex(int nx, int ny, int nz,
|
||||
real_t sx, real_t sy, real_t sz)
|
||||
real_t sx, real_t sy, real_t sz)
|
||||
{
|
||||
Mesh mesh;
|
||||
mesh.Make3D24TetsFromHex(nx, ny, nz, sx, sy, sz);
|
||||
@@ -4679,7 +4680,7 @@ Mesh Mesh::MakeRefined(Mesh &orig_mesh, const Array<int> &ref_factors,
|
||||
|
||||
Mesh::Mesh(const std::string &filename, int generate_edges, int refine,
|
||||
bool fix_orientation)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
{
|
||||
// Initialization as in the default constructor
|
||||
SetEmpty();
|
||||
@@ -4698,7 +4699,7 @@ Mesh::Mesh(const std::string &filename, int generate_edges, int refine,
|
||||
|
||||
Mesh::Mesh(std::istream &input, int generate_edges, int refine,
|
||||
bool fix_orientation)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
{
|
||||
SetEmpty();
|
||||
Load(input, generate_edges, refine, fix_orientation);
|
||||
@@ -4734,7 +4735,7 @@ Mesh::Mesh(real_t *vertices_, int num_vertices,
|
||||
int *boundary_indices, Geometry::Type boundary_type,
|
||||
int *boundary_attributes, int num_boundary_elements,
|
||||
int dimension, int space_dimension)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
{
|
||||
if (space_dimension == -1)
|
||||
{
|
||||
@@ -4772,7 +4773,7 @@ Mesh::Mesh(real_t *vertices_, int num_vertices,
|
||||
}
|
||||
|
||||
Mesh::Mesh( const NURBSExtension& ext )
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
{
|
||||
SetEmpty();
|
||||
/// make an internal copy of the NURBSExtension
|
||||
@@ -5050,9 +5051,13 @@ void Mesh::Loader(std::istream &input, int generate_edges,
|
||||
{
|
||||
ReadNURBSMesh(input, curved, read_gf);
|
||||
}
|
||||
else if (mesh_type == "MFEM NURBS NC-patch mesh v1.0")
|
||||
{
|
||||
ReadNURBSMesh(input, curved, read_gf, true, true); // Spacing is required
|
||||
}
|
||||
else if (mesh_type == "MFEM NURBS mesh v1.1")
|
||||
{
|
||||
ReadNURBSMesh(input, curved, read_gf, true);
|
||||
ReadNURBSMesh(input, curved, read_gf, true);
|
||||
}
|
||||
else if (mesh_type == "MFEM INLINE mesh v1.0")
|
||||
{
|
||||
@@ -5160,11 +5165,24 @@ void Mesh::Loader(std::istream &input, int generate_edges,
|
||||
"invalid mesh: end of file tag not found");
|
||||
}
|
||||
|
||||
if (NURBSext && NURBSext->NonconformingPatches())
|
||||
{
|
||||
string ident;
|
||||
skip_comment_lines(input, '#');
|
||||
// Check for the optional section "patch_cp"
|
||||
if (input.peek() == 'p')
|
||||
{
|
||||
input >> ident;
|
||||
MFEM_VERIFY(ident == "patch_cp", "Invalid mesh format");
|
||||
NURBSext->ReadCoarsePatchCP(input);
|
||||
}
|
||||
}
|
||||
|
||||
// Finalize(...) should be called after this, if needed.
|
||||
}
|
||||
|
||||
Mesh::Mesh(Mesh *mesh_array[], int num_pieces)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
{
|
||||
int i, j, ie, ib, iv, *v, nv;
|
||||
Element *el;
|
||||
@@ -5891,14 +5909,15 @@ Array<int> Mesh::MakeSimplicial_(const Mesh &orig_mesh, int *vglobal)
|
||||
}
|
||||
|
||||
|
||||
void Mesh::MakeHigherOrderSimplicial_(const Mesh &orig_mesh, const Array<int> &parent_elements)
|
||||
void Mesh::MakeHigherOrderSimplicial_(const Mesh &orig_mesh,
|
||||
const Array<int> &parent_elements)
|
||||
{
|
||||
// Higher order associated to vertices are unchanged, and those for
|
||||
// previously existing edges. DOFs associated to new elements need to be set.
|
||||
const int sdim = orig_mesh.SpaceDimension();
|
||||
auto *orig_fespace = orig_mesh.GetNodes()->FESpace();
|
||||
SetCurvature(orig_fespace->GetMaxElementOrder(), orig_fespace->IsDGSpace(),
|
||||
orig_mesh.SpaceDimension(), orig_fespace->GetOrdering());
|
||||
orig_mesh.SpaceDimension(), orig_fespace->GetOrdering());
|
||||
|
||||
// The dofs associated with vertices are unchanged, but there can be new dofs
|
||||
// associated to edges, faces and volumes. Additionally, because we know that
|
||||
@@ -5921,7 +5940,8 @@ void Mesh::MakeHigherOrderSimplicial_(const Mesh &orig_mesh, const Array<int> &p
|
||||
// of child element
|
||||
DenseMatrix shape; // ndof_coarse x nnode_refined.
|
||||
DenseMatrix point_matrix; // sdim x nnode_refined
|
||||
IntegrationRule child_nodes_in_parent; // The parent nodes that correspond to the child nodes
|
||||
IntegrationRule
|
||||
child_nodes_in_parent; // The parent nodes that correspond to the child nodes
|
||||
for (int i = 0; i < parent_elements.Size(); i++)
|
||||
{
|
||||
const int ip = parent_elements[i];
|
||||
@@ -5939,72 +5959,77 @@ void Mesh::MakeHigherOrderSimplicial_(const Mesh &orig_mesh, const Array<int> &p
|
||||
case Geometry::Type::PRISM : // fall through
|
||||
case Geometry::Type::PYRAMID : // fall through
|
||||
case Geometry::Type::SQUARE :
|
||||
{
|
||||
// Extract the vertices of parent and child, can then form the
|
||||
// map from child reference coordinates to parent reference
|
||||
// coordinates. Exploit the fact that for Nodes, the vertex
|
||||
// entries come first, and their indexing matches the vertex
|
||||
// numbering. Thus we have already have an inverse index map.
|
||||
orig_mesh.GetElementVertices(ip, parent_vertices);
|
||||
GetElementVertices(i, child_vertices);
|
||||
node_map.SetSize(0);
|
||||
for (auto cv : child_vertices)
|
||||
for (int ipv = 0; ipv < parent_vertices.Size(); ipv++)
|
||||
if (cv == parent_vertices[ipv])
|
||||
{
|
||||
node_map.Append(ipv);
|
||||
break;
|
||||
}
|
||||
MFEM_ASSERT(node_map.Size() == Geometry::NumVerts[GetElementBaseGeometry(i)],
|
||||
"!");
|
||||
// node_map now says which of the parent vertex nodes map to each
|
||||
// of the child vertex nodes. Using this can build a basis in the
|
||||
// parent element from child Node values, exploit the linearity
|
||||
// to then transform all nodes.
|
||||
child_nodes_in_parent.SetSize(0);
|
||||
const auto *orig_FE = orig_mesh.GetNodes()->FESpace()->GetFE(ip);
|
||||
for (auto pn : node_map)
|
||||
{
|
||||
// Extract the vertices of parent and child, can then form the
|
||||
// map from child reference coordinates to parent reference
|
||||
// coordinates. Exploit the fact that for Nodes, the vertex
|
||||
// entries come first, and their indexing matches the vertex
|
||||
// numbering. Thus we have already have an inverse index map.
|
||||
orig_mesh.GetElementVertices(ip, parent_vertices);
|
||||
GetElementVertices(i, child_vertices);
|
||||
node_map.SetSize(0);
|
||||
for (auto cv : child_vertices)
|
||||
for (int ipv = 0; ipv < parent_vertices.Size(); ipv++)
|
||||
if (cv == parent_vertices[ipv])
|
||||
{
|
||||
node_map.Append(ipv);
|
||||
break;
|
||||
}
|
||||
MFEM_ASSERT(node_map.Size() == Geometry::NumVerts[GetElementBaseGeometry(i)], "!");
|
||||
// node_map now says which of the parent vertex nodes map to each
|
||||
// of the child vertex nodes. Using this can build a basis in the
|
||||
// parent element from child Node values, exploit the linearity
|
||||
// to then transform all nodes.
|
||||
child_nodes_in_parent.SetSize(0);
|
||||
const auto *orig_FE = orig_mesh.GetNodes()->FESpace()->GetFE(ip);
|
||||
for (auto pn : node_map)
|
||||
{
|
||||
child_nodes_in_parent.Append(orig_FE->GetNodes()[pn]);
|
||||
}
|
||||
const auto *simplex_FE = GetNodes()->FESpace()->GetFE(i);
|
||||
shape.SetSize(orig_FE->GetDof(), simplex_FE->GetDof()); // One set of evaluations per simplex dof.
|
||||
Vector col;
|
||||
for (int j = 0; j < simplex_FE->GetNodes().Size(); j++)
|
||||
{
|
||||
const auto &simplex_node = simplex_FE->GetNodes()[j];
|
||||
IntegrationPoint simplex_node_in_orig;
|
||||
// Handle the 2D vs 3D case by multiplying .z by zero.
|
||||
simplex_node_in_orig.Set3(
|
||||
child_nodes_in_parent[0].x +
|
||||
simplex_node.x * (child_nodes_in_parent[1].x - child_nodes_in_parent[0].x)
|
||||
+ simplex_node.y * (child_nodes_in_parent[2].x - child_nodes_in_parent[0].x)
|
||||
+ simplex_node.z * (child_nodes_in_parent[(sdim > 2) ? 3 : 0].x - child_nodes_in_parent[0].x),
|
||||
child_nodes_in_parent[0].y +
|
||||
simplex_node.x * (child_nodes_in_parent[1].y - child_nodes_in_parent[0].y)
|
||||
+ simplex_node.y * (child_nodes_in_parent[2].y - child_nodes_in_parent[0].y)
|
||||
+ simplex_node.z * (child_nodes_in_parent[(sdim > 2) ? 3 : 0].y - child_nodes_in_parent[0].y),
|
||||
child_nodes_in_parent[0].z +
|
||||
simplex_node.x * (child_nodes_in_parent[1].z - child_nodes_in_parent[0].z)
|
||||
+ simplex_node.y * (child_nodes_in_parent[2].z - child_nodes_in_parent[0].z)
|
||||
+ simplex_node.z * (child_nodes_in_parent[(sdim > 2) ? 3 : 0].z - child_nodes_in_parent[0].z));
|
||||
shape.GetColumnReference(j, col);
|
||||
orig_FE->CalcShape(simplex_node_in_orig, col);
|
||||
}
|
||||
// All the non-simplex basis functions have now been evaluated at
|
||||
// all the simplex basis function node locations. Now evaluate
|
||||
// the summations and place back into the Nodes vector.
|
||||
orig_mesh.GetNodes()->GetElementDofValues(ip, edofvals);
|
||||
// Dof values are always returned as
|
||||
// [[x_1,x_2,x_3,...],
|
||||
// [y_1,y_2,y_3,...],
|
||||
// [z_1,z_2,z_3,...]]
|
||||
DenseMatrix edofvals_mat(edofvals.GetData(), orig_FE->GetDof(), sdim);
|
||||
point_matrix.SetSize(simplex_FE->GetDof(), sdim);
|
||||
MultAtB(shape, edofvals_mat, point_matrix);
|
||||
GetNodes()->FESpace()->GetElementVDofs(i, edofs);
|
||||
GetNodes()->SetSubVector(edofs, point_matrix.GetData());
|
||||
child_nodes_in_parent.Append(orig_FE->GetNodes()[pn]);
|
||||
}
|
||||
break;
|
||||
const auto *simplex_FE = GetNodes()->FESpace()->GetFE(i);
|
||||
shape.SetSize(orig_FE->GetDof(),
|
||||
simplex_FE->GetDof()); // One set of evaluations per simplex dof.
|
||||
Vector col;
|
||||
for (int j = 0; j < simplex_FE->GetNodes().Size(); j++)
|
||||
{
|
||||
const auto &simplex_node = simplex_FE->GetNodes()[j];
|
||||
IntegrationPoint simplex_node_in_orig;
|
||||
// Handle the 2D vs 3D case by multiplying .z by zero.
|
||||
simplex_node_in_orig.Set3(
|
||||
child_nodes_in_parent[0].x +
|
||||
simplex_node.x * (child_nodes_in_parent[1].x - child_nodes_in_parent[0].x)
|
||||
+ simplex_node.y * (child_nodes_in_parent[2].x - child_nodes_in_parent[0].x)
|
||||
+ simplex_node.z * (child_nodes_in_parent[(sdim > 2) ? 3 : 0].x -
|
||||
child_nodes_in_parent[0].x),
|
||||
child_nodes_in_parent[0].y +
|
||||
simplex_node.x * (child_nodes_in_parent[1].y - child_nodes_in_parent[0].y)
|
||||
+ simplex_node.y * (child_nodes_in_parent[2].y - child_nodes_in_parent[0].y)
|
||||
+ simplex_node.z * (child_nodes_in_parent[(sdim > 2) ? 3 : 0].y -
|
||||
child_nodes_in_parent[0].y),
|
||||
child_nodes_in_parent[0].z +
|
||||
simplex_node.x * (child_nodes_in_parent[1].z - child_nodes_in_parent[0].z)
|
||||
+ simplex_node.y * (child_nodes_in_parent[2].z - child_nodes_in_parent[0].z)
|
||||
+ simplex_node.z * (child_nodes_in_parent[(sdim > 2) ? 3 : 0].z -
|
||||
child_nodes_in_parent[0].z));
|
||||
shape.GetColumnReference(j, col);
|
||||
orig_FE->CalcShape(simplex_node_in_orig, col);
|
||||
}
|
||||
// All the non-simplex basis functions have now been evaluated at
|
||||
// all the simplex basis function node locations. Now evaluate
|
||||
// the summations and place back into the Nodes vector.
|
||||
orig_mesh.GetNodes()->GetElementDofValues(ip, edofvals);
|
||||
// Dof values are always returned as
|
||||
// [[x_1,x_2,x_3,...],
|
||||
// [y_1,y_2,y_3,...],
|
||||
// [z_1,z_2,z_3,...]]
|
||||
DenseMatrix edofvals_mat(edofvals.GetData(), orig_FE->GetDof(), sdim);
|
||||
point_matrix.SetSize(simplex_FE->GetDof(), sdim);
|
||||
MultAtB(shape, edofvals_mat, point_matrix);
|
||||
GetNodes()->FESpace()->GetElementVDofs(i, edofs);
|
||||
GetNodes()->SetSubVector(edofs, point_matrix.GetData());
|
||||
}
|
||||
break;
|
||||
case Geometry::Type::POINT : // fall through
|
||||
case Geometry::Type::INVALID :
|
||||
case Geometry::Type::NUM_GEOMETRIES :
|
||||
@@ -6287,6 +6312,11 @@ void Mesh::KnotRemove(Array<Vector *> &kv)
|
||||
UpdateNURBS();
|
||||
}
|
||||
|
||||
void Mesh::RefineNURBSWithKVFactors(int rf, const std::string &kvf)
|
||||
{
|
||||
RefineNURBS(true, 0.0, Array<int>(&rf, 1), kvf);
|
||||
}
|
||||
|
||||
void Mesh::NURBSUniformRefinement(int rf, real_t tol)
|
||||
{
|
||||
Array<int> rf_array(Dim);
|
||||
@@ -6299,8 +6329,13 @@ void Mesh::NURBSUniformRefinement(Array<int> const& rf, real_t tol)
|
||||
MFEM_VERIFY(rf.Size() == Dim,
|
||||
"Refinement factors must be defined for each dimension");
|
||||
|
||||
MFEM_VERIFY(NURBSext, "NURBSUniformRefinement is only for NURBS meshes");
|
||||
RefineNURBS(false, tol, rf, "");
|
||||
}
|
||||
|
||||
void Mesh::RefineNURBS(bool usingKVF, real_t tol, const Array<int> &rf,
|
||||
const std::string &kvf)
|
||||
{
|
||||
MFEM_VERIFY(NURBSext, "This type of refinement is only for NURBS meshes");
|
||||
NURBSext->ConvertToPatches(*Nodes);
|
||||
|
||||
Array<int> cf;
|
||||
@@ -6312,17 +6347,19 @@ void Mesh::NURBSUniformRefinement(Array<int> const& rf, real_t tol)
|
||||
cf1 = (cf1 && f == 1);
|
||||
}
|
||||
|
||||
if (cf1)
|
||||
if (!cf1 && NURBSext->NonconformingPatches())
|
||||
{
|
||||
NURBSext->UniformRefinement(rf);
|
||||
NURBSext->FullyCoarsen();
|
||||
last_operation = Mesh::NONE; // FiniteElementSpace::Update is not supported
|
||||
}
|
||||
else
|
||||
else if (!cf1 && !NURBSext->NonconformingPatches())
|
||||
{
|
||||
MFEM_VERIFY(!usingKVF, "This refinement type is not supported for this"
|
||||
" NURBS mesh type");
|
||||
NURBSext->Coarsen(cf, tol);
|
||||
|
||||
last_operation = Mesh::NONE; // FiniteElementSpace::Update is not supported
|
||||
sequence++;
|
||||
|
||||
UpdateNURBS();
|
||||
|
||||
NURBSext->ConvertToPatches(*Nodes);
|
||||
@@ -6330,6 +6367,18 @@ void Mesh::NURBSUniformRefinement(Array<int> const& rf, real_t tol)
|
||||
NURBSext->UniformRefinement(cf);
|
||||
}
|
||||
|
||||
if (cf1 || NURBSext->NonconformingPatches())
|
||||
{
|
||||
if (usingKVF || NURBSext->NonconformingPatches())
|
||||
{
|
||||
NURBSext->RefineWithKVFactors(rf[0], kvf, !cf1);
|
||||
}
|
||||
else
|
||||
{
|
||||
NURBSext->UniformRefinement(rf);
|
||||
}
|
||||
}
|
||||
|
||||
last_operation = Mesh::NONE; // FiniteElementSpace::Update is not supported
|
||||
sequence++;
|
||||
|
||||
@@ -6545,7 +6594,7 @@ void Mesh::GetEdgeToUniqueKnotvector(Array<int> &edge_to_ukv,
|
||||
{
|
||||
const int ri = get_root(i);
|
||||
const int rj = get_root(j);
|
||||
if (ri == rj) return;
|
||||
if (ri == rj) { return; }
|
||||
// keep the lowest index
|
||||
(ri < rj) ? pkv_map[rj] = ri : pkv_map[ri] = rj;
|
||||
};
|
||||
@@ -6606,6 +6655,52 @@ void Mesh::GetEdgeToUniqueKnotvector(Array<int> &edge_to_ukv,
|
||||
}
|
||||
}
|
||||
|
||||
void Mesh::LoadNonconformingPatchTopo(std::istream &input,
|
||||
Array<int> &edge_to_ukv)
|
||||
{
|
||||
SetEmpty();
|
||||
|
||||
// Read MFEM NURBS NC-patch mesh v1.0 format
|
||||
int curved = 0;
|
||||
int is_nc = 1;
|
||||
|
||||
ncmesh = new NCMesh(input, 10, curved, is_nc);
|
||||
|
||||
InitFromNCMesh(*ncmesh);
|
||||
|
||||
skip_comment_lines(input, '#');
|
||||
|
||||
string ident;
|
||||
int inputNumOfEdges = -1;
|
||||
|
||||
input >> ident; // 'edges'
|
||||
input >> inputNumOfEdges;
|
||||
|
||||
MFEM_VERIFY(NumOfEdges == inputNumOfEdges, "");
|
||||
|
||||
edge_to_ukv.SetSize(NumOfEdges);
|
||||
for (int j = 0; j < NumOfEdges; j++)
|
||||
{
|
||||
int v[2]; // Vertex indices
|
||||
int ukv; // Unique KnotVector index
|
||||
input >> ukv >> v[0] >> v[1];
|
||||
|
||||
for (int i=0; i<2; ++i)
|
||||
{
|
||||
v[i] = ncmesh->vertex_nodeId[v[i]];
|
||||
}
|
||||
|
||||
if (v[0] > v[1])
|
||||
{
|
||||
ukv = -1 - ukv;
|
||||
}
|
||||
edge_to_ukv[j] = ukv;
|
||||
}
|
||||
|
||||
FinalizeTopology();
|
||||
CheckBdrElementOrientation(); // check and fix boundary element orientation
|
||||
}
|
||||
|
||||
void XYZ_VectorFunction(const Vector &p, Vector &v)
|
||||
{
|
||||
if (p.Size() >= v.Size())
|
||||
@@ -7848,14 +7943,14 @@ void Mesh::GetBdrElementAdjacentElement2(
|
||||
|
||||
void Mesh::SetAttribute(int i, int attr)
|
||||
{
|
||||
elements[i]->SetAttribute(attr);
|
||||
if (elem_attrs_cache.Size() == GetNE())
|
||||
{
|
||||
// update the existing cache instead of deleting it
|
||||
elem_attrs_cache.HostReadWrite();
|
||||
elem_attrs_cache[i] = attr;
|
||||
}
|
||||
if (ncmesh) ncmesh->SetAttribute(i, attr);
|
||||
elements[i]->SetAttribute(attr);
|
||||
if (elem_attrs_cache.Size() == GetNE())
|
||||
{
|
||||
// update the existing cache instead of deleting it
|
||||
elem_attrs_cache.HostReadWrite();
|
||||
elem_attrs_cache[i] = attr;
|
||||
}
|
||||
if (ncmesh) { ncmesh->SetAttribute(i, attr); }
|
||||
}
|
||||
|
||||
Element::Type Mesh::GetElementType(int i) const
|
||||
@@ -10924,7 +11019,7 @@ void Mesh::InitFromNCMesh(const NCMesh &ncmesh_)
|
||||
}
|
||||
|
||||
Mesh::Mesh(const NCMesh &ncmesh_)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
: attribute_sets(attributes), bdr_attribute_sets(bdr_attributes)
|
||||
{
|
||||
Init();
|
||||
InitTables();
|
||||
@@ -11884,6 +11979,7 @@ void Mesh::Printer(std::ostream &os, std::string section_delimiter,
|
||||
os << '\n';
|
||||
Nodes->Save(os);
|
||||
|
||||
NURBSext->PrintCoarsePatches(os);
|
||||
// patch-wise format
|
||||
// NURBSext->ConvertToPatches(*Nodes);
|
||||
// NURBSext->Print(os);
|
||||
@@ -11918,10 +12014,10 @@ void Mesh::Printer(std::ostream &os, std::string section_delimiter,
|
||||
|
||||
// serial/parallel conforming mesh format
|
||||
const bool set_names = attribute_sets.SetsExist() ||
|
||||
bdr_attribute_sets.SetsExist();
|
||||
bdr_attribute_sets.SetsExist();
|
||||
os << (!set_names && section_delimiter.empty()
|
||||
? "MFEM mesh v1.0\n" :
|
||||
(!set_names ? "MFEM mesh v1.2\n" : "MFEM mesh v1.3\n"));
|
||||
(!set_names ? "MFEM mesh v1.2\n" : "MFEM mesh v1.3\n"));
|
||||
|
||||
if (set_names && section_delimiter.empty())
|
||||
{
|
||||
@@ -11953,8 +12049,8 @@ void Mesh::Printer(std::ostream &os, std::string section_delimiter,
|
||||
|
||||
if (set_names)
|
||||
{
|
||||
os << "\nattribute_sets\n";
|
||||
attribute_sets.Print(os);
|
||||
os << "\nattribute_sets\n";
|
||||
attribute_sets.Print(os);
|
||||
}
|
||||
|
||||
os << "\nboundary\n" << NumOfBdrElements << '\n';
|
||||
@@ -11965,8 +12061,8 @@ void Mesh::Printer(std::ostream &os, std::string section_delimiter,
|
||||
|
||||
if (set_names)
|
||||
{
|
||||
os << "\nbdr_attribute_sets\n";
|
||||
bdr_attribute_sets.Print(os);
|
||||
os << "\nbdr_attribute_sets\n";
|
||||
bdr_attribute_sets.Print(os);
|
||||
}
|
||||
|
||||
os << "\nvertices\n" << NumOfVertices << '\n';
|
||||
@@ -11998,9 +12094,9 @@ void Mesh::Printer(std::ostream &os, std::string section_delimiter,
|
||||
}
|
||||
|
||||
void Mesh::PrintTopo(std::ostream &os, const Array<int> &e_to_k,
|
||||
const int version, const std::string &comments) const
|
||||
const int version, const std::string &comments) const
|
||||
{
|
||||
MFEM_VERIFY(version == 10 || version == 11, "Invalid NURBS mesh version");
|
||||
MFEM_VERIFY(version == 10 || version == 11, "Invalid NURBS mesh version");
|
||||
|
||||
int i;
|
||||
Array<int> vert;
|
||||
@@ -12030,8 +12126,16 @@ void Mesh::PrintTopo(std::ostream &os, const Array<int> &e_to_k,
|
||||
PrintElement(boundary[i], os);
|
||||
}
|
||||
|
||||
PrintTopoEdges(os, e_to_k);
|
||||
}
|
||||
|
||||
void Mesh::PrintTopoEdges(std::ostream &os, const Array<int> &e_to_k,
|
||||
bool vmap) const
|
||||
{
|
||||
Array<int> vert;
|
||||
|
||||
os << "\nedges\n" << NumOfEdges << '\n';
|
||||
for (i = 0; i < NumOfEdges; i++)
|
||||
for (int i = 0; i < NumOfEdges; i++)
|
||||
{
|
||||
edge_vertex->GetRow(i, vert);
|
||||
int ki = e_to_k[i];
|
||||
@@ -12039,9 +12143,30 @@ void Mesh::PrintTopo(std::ostream &os, const Array<int> &e_to_k,
|
||||
{
|
||||
ki = -1 - ki;
|
||||
}
|
||||
|
||||
if (vmap)
|
||||
{
|
||||
for (int j=0; j<2; ++j)
|
||||
{
|
||||
vert[j] = ncmesh->vertex_nodeId[vert[j]];
|
||||
}
|
||||
|
||||
if (e_to_k[i] < 0)
|
||||
{
|
||||
// Swap the entries of vert
|
||||
const int s = vert[0];
|
||||
vert[0] = vert[1];
|
||||
vert[1] = s;
|
||||
}
|
||||
}
|
||||
|
||||
os << ki << ' ' << vert[0] << ' ' << vert[1] << '\n';
|
||||
}
|
||||
os << "\nvertices\n" << NumOfVertices << '\n';
|
||||
|
||||
if (!vmap)
|
||||
{
|
||||
os << "\nvertices\n" << NumOfVertices << '\n';
|
||||
}
|
||||
}
|
||||
|
||||
void Mesh::Save(const std::string &fname, int precision) const
|
||||
@@ -15321,6 +15446,20 @@ Mesh *Extrude2D(Mesh *mesh, const int nz, const real_t sz)
|
||||
return mesh3d;
|
||||
}
|
||||
|
||||
bool Mesh::Conforming() const
|
||||
{
|
||||
if (NURBSext)
|
||||
{
|
||||
// NURBS meshes are always conforming (element-wise). NURBS patch
|
||||
// conformity is indicated by NURBSExtension::NonconformingPatches.
|
||||
return true;
|
||||
}
|
||||
else
|
||||
{
|
||||
return ncmesh == NULL;
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef MFEM_DEBUG
|
||||
void Mesh::DebugDump(std::ostream &os) const
|
||||
{
|
||||
|
||||
+28
-5
@@ -65,6 +65,7 @@ class Mesh
|
||||
{
|
||||
friend class NCMesh;
|
||||
friend class NURBSExtension;
|
||||
friend class NCNURBSExtension;
|
||||
#ifdef MFEM_USE_MPI
|
||||
friend class ParMesh;
|
||||
friend class ParNCMesh;
|
||||
@@ -359,7 +360,7 @@ protected:
|
||||
void ReadXML_VTKMesh(std::istream &input, int &curved, int &read_gf,
|
||||
bool &finalize_topo, const std::string &xml_prefix="");
|
||||
void ReadNURBSMesh(std::istream &input, int &curved, int &read_gf,
|
||||
bool spacing=false);
|
||||
bool spacing=false, bool nc=false);
|
||||
void ReadInlineMesh(std::istream &input, bool generate_edges = false);
|
||||
void ReadGmshMesh(std::istream &input, int &curved, int &read_gf);
|
||||
|
||||
@@ -492,8 +493,24 @@ protected:
|
||||
/// Read NURBS patch/macro-element mesh
|
||||
void LoadPatchTopo(std::istream &input, Array<int> &edge_to_ukv);
|
||||
|
||||
/// Read NURBS patch/macro-element mesh (MFEM NURBS NC-patch mesh format)
|
||||
void LoadNonconformingPatchTopo(std::istream &input,
|
||||
Array<int> &edge_to_ukv);
|
||||
|
||||
/// Update this NURBS Mesh and its NURBS data structures after a change, such
|
||||
/// as refinement, derefinement, or degree change.
|
||||
void UpdateNURBS();
|
||||
|
||||
/** @brief Refine the NURBS mesh with default refinement factors in @a rf for
|
||||
each dimension.
|
||||
|
||||
Optionally, if @a usingKVF is true, use refinement factors specified for
|
||||
particular KnotVectors, from the file with name in @a kvf. When
|
||||
coarsening by knot removal is necessary for non-nested spacing formulas,
|
||||
tolerance @a tol is used (see NURBSPatch::KnotRemove()). */
|
||||
void RefineNURBS(bool usingKVF, real_t tol, const Array<int> &rf,
|
||||
const std::string &kvf);
|
||||
|
||||
/** @brief Write the beginning of a NURBS mesh to @a os, specifying the NURBS
|
||||
patch topology. Optional file comments can be provided in @a comments.
|
||||
|
||||
@@ -506,6 +523,10 @@ protected:
|
||||
const int version,
|
||||
const std::string &comment = "") const;
|
||||
|
||||
/// Write the patch topology edges of a NURBS mesh (see PrintTopo()).
|
||||
void PrintTopoEdges(std::ostream &out, const Array<int> &e_to_k,
|
||||
bool vmap = false) const;
|
||||
|
||||
/// Used in GetFaceElementTransformations (...)
|
||||
void GetLocalPtToSegTransformation(IsoparametricTransformation &,
|
||||
int i) const;
|
||||
@@ -2410,6 +2431,10 @@ public:
|
||||
virtual void NURBSUniformRefinement(int rf = 2, real_t tol = 1.0e-12);
|
||||
virtual void NURBSUniformRefinement(const Array<int> &rf, real_t tol=1.e-12);
|
||||
|
||||
/** @a brief Use knotvector refinement factors loaded from the file with name
|
||||
in @a kvf. Everywhere else, use the default refinement factor @a rf. */
|
||||
virtual void RefineNURBSWithKVFactors(int rf, const std::string &kvf);
|
||||
|
||||
/// Coarsening for a NURBS mesh, with an optional coarsening factor @a cf > 1
|
||||
/// which divides the number of elements in each dimension.
|
||||
void NURBSCoarsening(int cf = 2, real_t tol = 1.0e-12);
|
||||
@@ -2465,10 +2490,8 @@ public:
|
||||
(default) or nonconforming. */
|
||||
void EnsureNCMesh(bool simplices_nonconforming = false);
|
||||
|
||||
/// Return a bool indicating whether this mesh is conforming.
|
||||
bool Conforming() const { return ncmesh == NULL; }
|
||||
/// Return a bool indicating whether this mesh is nonconforming.
|
||||
bool Nonconforming() const { return ncmesh != NULL; }
|
||||
bool Conforming() const;
|
||||
bool Nonconforming() const { return !Conforming(); }
|
||||
|
||||
/** Designate this mesh for output as "NC mesh v1.1", meaning it is
|
||||
nonconforming with nonuniform refinement spacings. */
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "mesh_headers.hpp"
|
||||
#include "ncnurbs.hpp"
|
||||
#include "../fem/fem.hpp"
|
||||
#include "../general/binaryio.hpp"
|
||||
#include "../general/text.hpp"
|
||||
@@ -772,7 +773,7 @@ struct BufferReader : BufferReaderBase
|
||||
int header_entry_size = HeaderEntrySize();
|
||||
int nblocks = ReadHeaderEntry(header_buf);
|
||||
header_buf += header_entry_size;
|
||||
std::vector<int> header(nblocks + 2);
|
||||
std::vector<size_t> header(nblocks + 2);
|
||||
for (int i=0; i<nblocks+2; ++i)
|
||||
{
|
||||
header[i] = ReadHeaderEntry(header_buf);
|
||||
@@ -791,7 +792,7 @@ struct BufferReader : BufferReaderBase
|
||||
dest_ptr += dest_len;
|
||||
source_ptr += source_len;
|
||||
}
|
||||
MFEM_VERIFY(int(sizeof(F)*n) == (dest_ptr - dest_start),
|
||||
MFEM_VERIFY(size_t(sizeof(F)*n) == (dest_ptr - dest_start),
|
||||
"AppendedData: wrong data size");
|
||||
buf = uncompressed_data.data();
|
||||
#else
|
||||
@@ -1309,9 +1310,10 @@ void Mesh::ReadVTKMesh(std::istream &input, int &curved, int &read_gf,
|
||||
} // end ReadVTKMesh
|
||||
|
||||
void Mesh::ReadNURBSMesh(std::istream &input, int &curved, int &read_gf,
|
||||
bool spacing)
|
||||
bool spacing, bool nc)
|
||||
{
|
||||
NURBSext = new NURBSExtension(input, spacing);
|
||||
NURBSext = nc ? new NCNURBSExtension(input, spacing):
|
||||
new NURBSExtension(input, spacing);
|
||||
|
||||
Dim = NURBSext->Dimension();
|
||||
NumOfVertices = NURBSext->GetNV();
|
||||
|
||||
+125
-2
@@ -5104,6 +5104,55 @@ void NCMesh::GetPointMatrix(Geometry::Type geom, const char* ref_path,
|
||||
}
|
||||
}
|
||||
|
||||
void RemapKnotIndex(bool rev, const Array<int> &rf, int &k);
|
||||
std::pair<int, int> QuadrupleToPair(const std::array<int, 4> &q);
|
||||
|
||||
void NCMesh::RefineVertexToKnotSpan(const std::vector<Array<int>> &kvf,
|
||||
const Array<KnotVector*> &kvext,
|
||||
std::map<std::pair<int, int>,
|
||||
std::array<int, 2>> &parentToKV)
|
||||
{
|
||||
// Note that entries 1 and 2 of vertex_to_knotspan are (k1, k2), which are knot
|
||||
// span (element) indices in the two dimensions of a patch face.
|
||||
|
||||
for (int i=0; i<vertex_to_knotspan.Size(); ++i)
|
||||
{
|
||||
if (Dim == 3)
|
||||
{
|
||||
int tv;
|
||||
std::array<int, 2> ks;
|
||||
std::array<int, 4> pv;
|
||||
vertex_to_knotspan.GetVertex3D(i, tv, ks, pv);
|
||||
|
||||
bool edgeReverse[2];
|
||||
for (int j=0; j<2; ++j)
|
||||
{
|
||||
const bool ascending = pv[j+1] > pv[j];
|
||||
edgeReverse[j] = !ascending;
|
||||
}
|
||||
|
||||
// The parent face is defined with vertices (pv0, pv1, pv2, pv3).
|
||||
const std::pair<int, int> parentPair = QuadrupleToPair(pv);
|
||||
const std::array<int, 2> kv = parentToKV.at(parentPair);
|
||||
RemapKnotIndex(edgeReverse[0], kvf[kv[0]], ks[0]);
|
||||
RemapKnotIndex(edgeReverse[1], kvf[kv[1]], ks[1]);
|
||||
vertex_to_knotspan.SetKnotSpans3D(i, ks);
|
||||
}
|
||||
else // 2D
|
||||
{
|
||||
int tv, ks;
|
||||
std::array<int, 2> pv;
|
||||
vertex_to_knotspan.GetVertex2D(i, tv, ks, pv);
|
||||
const bool rev = pv[1] < pv[0];
|
||||
const std::pair<int, int> parentPair(rev ? pv[1] : pv[0], rev ? pv[0] : pv[1]);
|
||||
const std::array<int, 2> kv = parentToKV.at(parentPair);
|
||||
const int kvId = kv[0];
|
||||
RemapKnotIndex(rev, kvf[kvId], ks);
|
||||
vertex_to_knotspan.SetKnotSpan2D(i, ks);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void NCMesh::MarkCoarseLevel()
|
||||
{
|
||||
coarse_elements.SetSize(leaf_elements.Size());
|
||||
@@ -6116,6 +6165,60 @@ void NCMesh::LoadVertexParents(std::istream &input)
|
||||
}
|
||||
}
|
||||
|
||||
void NCMesh::LoadVertexToKnotSpan(std::istream &input)
|
||||
{
|
||||
if (Dim == 2) { LoadVertexToKnotSpan2D(input); }
|
||||
else { LoadVertexToKnotSpan3D(input); }
|
||||
}
|
||||
|
||||
void NCMesh::LoadVertexToKnotSpan2D(std::istream &input)
|
||||
{
|
||||
int nv;
|
||||
input >> nv;
|
||||
MFEM_VERIFY(0 <= nv, "Invalid vertex-to-knot data");
|
||||
vertex_to_knotspan.SetSize(2, nv);
|
||||
for (int i=0; i<nv; ++i)
|
||||
{
|
||||
int id, ks;
|
||||
std::array<int, 2> pv;
|
||||
input >> id >> ks >> pv[0] >> pv[1];
|
||||
|
||||
const bool idsExist = nodes.IdExists(id) && nodes.IdExists(pv[0])
|
||||
&& nodes.IdExists(pv[1]);
|
||||
|
||||
MFEM_VERIFY(idsExist && 0 < ks, "Invalid index");
|
||||
vertex_to_knotspan.SetVertex2D(i, id, ks, pv);
|
||||
}
|
||||
}
|
||||
|
||||
void NCMesh::LoadVertexToKnotSpan3D(std::istream &input)
|
||||
{
|
||||
int nv;
|
||||
input >> nv;
|
||||
MFEM_VERIFY(0 <= nv, "Invalid vertex-to-knot data");
|
||||
vertex_to_knotspan.SetSize(3, nv);
|
||||
for (int i=0; i<nv; ++i)
|
||||
{
|
||||
int id;
|
||||
std::array<int, 2> ks;
|
||||
std::array<int, 4> pv; // Parent vertex indices
|
||||
input >> id >> ks[0] >> ks[1] >> pv[0] >> pv[1] >> pv[2] >> pv[3];
|
||||
|
||||
#ifdef MFEM_DEBUG
|
||||
bool idsExist = nodes.IdExists(id);
|
||||
for (int j=0; j<4; ++j)
|
||||
{
|
||||
idsExist = idsExist && nodes.IdExists(pv[j]);
|
||||
}
|
||||
|
||||
const bool validKnotIds = (0 <= ks[0] || 0 <= ks[1]) &&
|
||||
(0 < ks[0] || 0 < ks[1]);
|
||||
MFEM_ASSERT(idsExist && validKnotIds, "Invalid index");
|
||||
#endif
|
||||
vertex_to_knotspan.SetVertex3D(i, id, ks, pv);
|
||||
}
|
||||
}
|
||||
|
||||
int NCMesh::PrintBoundary(std::ostream *os) const
|
||||
{
|
||||
static const int nfv2geom[5] =
|
||||
@@ -6243,9 +6346,14 @@ bool NCMesh::ZeroRootStates() const
|
||||
return true;
|
||||
}
|
||||
|
||||
void NCMesh::Print(std::ostream &os, const std::string &comments) const
|
||||
void NCMesh::Print(std::ostream &os, const std::string &comments,
|
||||
bool nurbs) const
|
||||
{
|
||||
if (using_scaling)
|
||||
if (nurbs)
|
||||
{
|
||||
os << "MFEM NURBS NC-patch mesh v1.0\n\n";
|
||||
}
|
||||
else if (using_scaling)
|
||||
{
|
||||
os << "MFEM NC mesh v1.1\n\n";
|
||||
}
|
||||
@@ -6321,6 +6429,12 @@ void NCMesh::Print(std::ostream &os, const std::string &comments) const
|
||||
}
|
||||
}
|
||||
|
||||
if (nurbs && vertex_to_knotspan.Size() > 0)
|
||||
{
|
||||
os << "\nvertex_to_knotspan\n";
|
||||
vertex_to_knotspan.Print(os);
|
||||
}
|
||||
|
||||
if (coordinates.Size())
|
||||
{
|
||||
os << "\n# top-level node coordinates";
|
||||
@@ -6513,6 +6627,15 @@ NCMesh::NCMesh(std::istream &input, int version, int &curved, int &is_nc)
|
||||
input >> ident;
|
||||
}
|
||||
|
||||
// load map from hanging patch vertices to patch edge knots
|
||||
if (ident == "vertex_to_knotspan")
|
||||
{
|
||||
LoadVertexToKnotSpan(input);
|
||||
|
||||
skip_comment_lines(input, '#');
|
||||
input >> ident;
|
||||
}
|
||||
|
||||
// load root states
|
||||
if (ident == "root_state")
|
||||
{
|
||||
|
||||
+72
-2
@@ -114,7 +114,57 @@ void Swap(CoarseFineTransformations &a, CoarseFineTransformations &b);
|
||||
|
||||
struct MatrixMap; // for internal use
|
||||
|
||||
/** \brief A class for non-conforming AMR. The class is not used directly by the
|
||||
/** @brief For a NURBS mesh with nonconforming patch topology, this struct
|
||||
provides a map from hanging vertices in the patch topology to the knotvector
|
||||
of a neighboring patch. This facilitates ensuring mesh conformity.
|
||||
*/
|
||||
class VertexToKnotSpan
|
||||
{
|
||||
public:
|
||||
/// Set the spatial dimension and number of vertices.
|
||||
void SetSize(int dimension, int numVertices);
|
||||
|
||||
// The following set and get functions are for a single entry in the array of
|
||||
// data, for a hanging vertex in the patch topology, with the given 'index'.
|
||||
// The vertex index is 'v', parent vertices are 'pv', and knot-span is 'ks'.
|
||||
|
||||
/// Set the data for a vertex in 2D.
|
||||
void SetVertex2D(int index, int v, int ks,
|
||||
const std::array<int, 2> &pv);
|
||||
|
||||
/// Set the data for a vertex in 3D.
|
||||
void SetVertex3D(int index, int v, const std::array<int, 2> &ks,
|
||||
const std::array<int, 4> &pv);
|
||||
|
||||
/// Set the knot-span index for a vertex in 2D.
|
||||
void SetKnotSpan2D(int index, int ks);
|
||||
|
||||
/// Set the knot-span indices for a vertex in 3D.
|
||||
void SetKnotSpans3D(int index, const std::array<int, 2> &ks);
|
||||
|
||||
/// Get the data for a vertex in 2D.
|
||||
void GetVertex2D(int index, int &v, int &ks,
|
||||
std::array<int, 2> &pv) const;
|
||||
|
||||
/// Get the data for a vertex in 3D.
|
||||
void GetVertex3D(int index, int &v, std::array<int, 2> &ks,
|
||||
std::array<int, 4> &pv) const;
|
||||
|
||||
/// Print all the data.
|
||||
void Print(std::ostream &os) const;
|
||||
|
||||
/// Return the number of vertices.
|
||||
int Size() const { return data.NumRows(); }
|
||||
|
||||
/// Return the vertex pair representing the parent edge (2D) or face (3D).
|
||||
std::pair<int, int> GetVertexParentPair(int index) const;
|
||||
|
||||
private:
|
||||
int dim; /// Spatial dimension
|
||||
Array2D<int> data; /// Row-wise data for each vertex.
|
||||
};
|
||||
|
||||
/** @brief A class for non-conforming AMR. The class is not used directly by the
|
||||
* user, rather it is an extension of the Mesh class.
|
||||
*
|
||||
* In general, the class is used by MFEM as follows:
|
||||
@@ -348,6 +398,16 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
const VertexToKnotSpan& GetVertexToKnotSpan() const
|
||||
{
|
||||
return vertex_to_knotspan;
|
||||
}
|
||||
|
||||
/// Remap knot-span indices @a vertex_to_knotspan after refinement.
|
||||
void RefineVertexToKnotSpan(const std::vector<Array<int>> &kvf,
|
||||
const Array<KnotVector*> &kvext,
|
||||
std::map<std::pair<int, int>,
|
||||
std::array<int, 2>> &parentToKV);
|
||||
|
||||
// coarse/fine transforms
|
||||
|
||||
@@ -467,7 +527,8 @@ public:
|
||||
/** I/O: Print the mesh in "MFEM NC mesh v1.0" format. If @a comments is
|
||||
non-empty, it will be printed after the first line of the file, and each
|
||||
line should begin with '#'. */
|
||||
void Print(std::ostream &out, const std::string &comments = "") const;
|
||||
void Print(std::ostream &out, const std::string &comments = "",
|
||||
bool nurbs=false) const;
|
||||
|
||||
/// I/O: Return true if the mesh was loaded from the legacy v1.1 format.
|
||||
bool IsLegacyLoaded() const { return Legacy; }
|
||||
@@ -1300,6 +1361,12 @@ protected:
|
||||
/// Load the vertex parent hierarchy from a mesh file.
|
||||
void LoadVertexParents(std::istream &input);
|
||||
|
||||
/// Load VertexToKnotSpan data for the NC patch topology mesh of a 2D or 3D
|
||||
/// MFEM NURBS NC-patch mesh.
|
||||
void LoadVertexToKnotSpan(std::istream &input);
|
||||
void LoadVertexToKnotSpan2D(std::istream &input);
|
||||
void LoadVertexToKnotSpan3D(std::istream &input);
|
||||
|
||||
/** Print the "boundary" section of the mesh file. If out == NULL, only
|
||||
return the number of boundary elements. */
|
||||
int PrintBoundary(std::ostream *out) const;
|
||||
@@ -1342,6 +1409,9 @@ protected:
|
||||
|
||||
static GeomInfo GI[Geometry::NumGeom];
|
||||
|
||||
/// This is used for a NURBS mesh with this NCMesh as its patch topology.
|
||||
VertexToKnotSpan vertex_to_knotspan;
|
||||
|
||||
#ifdef MFEM_DEBUG
|
||||
public:
|
||||
void DebugLeafOrder(std::ostream &out) const;
|
||||
|
||||
+3866
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user